diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml index e3daad8e8..c25acd068 100644 --- a/.github/workflows/check.yml +++ b/.github/workflows/check.yml @@ -23,7 +23,9 @@ jobs: linux: name: Linux runs-on: ubuntu-latest - timeout-minutes: 15 + # make check takes about 14 minutes here now (14.9 at worst this week); + # at 15 the job was cut off with nothing failed. + timeout-minutes: 25 steps: - uses: actions/checkout@v4 - name: make check @@ -128,13 +130,21 @@ jobs: python c/tools/make_edge_tiny_tokenizer.py --vocab-size 281 /tmp/glm53_stream-i4 cd c && COLI_GLM53_FIXTURE=/tmp/glm53_stream-i4 python -m unittest -v tests.test_glm53_dashboard COLI_GLM53_FIXTURE=/tmp/glm53_stream-i4 python -m unittest -v tests.test_glm53_context_exceeded + # The two stdlib-only oracles used to be argparse scripts that `make + # test-python` collected as zero tests (#1700). They run here, on the + # fixtures generated above, token-exact against transformers in f32. + - name: Run the GLM-5.3 tiny oracles + run: | + cd c && GLM53_TINY=/tmp/glm53_tiny GLM53_MM_TINY=/tmp/glm53_mm python -m unittest -v tests.test_glm53_oracles macos: # clang; libomp for the threaded path (Makefile falls back to # single-threaded automatically if it's ever missing). name: macOS (colibri + V4 platform gate) runs-on: macos-latest - timeout-minutes: 15 + # `make check` here now takes about 14.5 minutes; at 15 the job was cut + # off inside the Python suite with no test failed, and showed as red. + timeout-minutes: 25 steps: - uses: actions/checkout@v4 - name: install libomp diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c9af31a35..1471b14da 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -147,7 +147,7 @@ jobs: musl: name: musl libc (Alpine) runs-on: ubuntu-latest - container: alpine:3.21 + container: alpine:3.24 steps: # #1430: colibri did not compile against musl, because malloc_trim is a # glibc extension guarded on __linux__ rather than on __GLIBC__. Nothing in @@ -547,6 +547,31 @@ jobs: echo "cap=$cap: identical ($(cat ids_${cap}_1.txt))" done make tests/test_expert_ffn && ./tests/test_expert_ffn + - name: Mixed expert layout (int4 gs64 gate/up, int8 down) loads and is cap-independent + run: | + cd c + # The #1370 experiment knob: one slab per expert with int4 gate/up and int8 + # down (2*inter*hidden bytes). The engine must tell it apart from int4 and + # int8 by size, read every matrix in its own format, and give the same ids + # at every cache capacity (cap=1 recycles the single slot after each expert, + # which is where a wrong slab offset would show). The ids are not compared to + # the torch oracle: int4 gate/up drift from f32 by design, as in the A/B above. + python3 tools/convert_qwen36.py --model qwen36_tiny64 --out qwen36_tiny64_d8 --ebits 4 --gs 64 --down-bits 8 + python3 - <<'PY' + import json; m = json.load(open("qwen36_tiny64_d8/qwen36_meta.json")) + assert m["expert_down_bits"] == 8 and m["expert_down_gs"] == 0 and m["expert_gs"] == 64, m + PY + for cap in 1 2 8; do + COLI_DENSE_I8=0 SNAP=qwen36_tiny64_d8 ./qwen36 "$cap" 4 qwen36_tiny64/ref_full.json > mixed_$cap.log 2>&1 || true + grep -q "expert format on disk: int4 gate/up + int8 down" mixed_$cap.log || { echo "FAIL: mixed layout not detected at cap=$cap"; cat mixed_$cap.log; exit 1; } + grep -E "^C engine" mixed_$cap.log > mixed_ids_$cap.txt + test -s mixed_ids_$cap.txt || { echo "FAIL: no ids at cap=$cap"; cat mixed_$cap.log; exit 1; } + done + cmp mixed_ids_1.txt mixed_ids_8.txt && cmp mixed_ids_2.txt mixed_ids_8.txt || { echo "FAIL: ids differ across caps"; cat mixed_ids_*.txt; exit 1; } + echo "mixed layout: identical ids at cap 1/2/8 ($(cat mixed_ids_8.txt))" + # COLI_CUDA=1 on a mixed container is refused with a line, the CPU path stands + COLI_CUDA=1 COLI_DENSE_I8=0 SNAP=qwen36_tiny64_d8 ./qwen36 8 4 qwen36_tiny64/ref_full.json > mixed_cuda.log 2>&1 || true + grep -q "COLI_CUDA=1 ignored: the VRAM expert tier does not take the mixed layout" mixed_cuda.log || { echo "FAIL: tier refusal line missing"; cat mixed_cuda.log; exit 1; } - name: A malformed container is refused, not read run: | cd c @@ -843,7 +868,9 @@ jobs: # the oracle fixture has no tokenizer; serve mode needs one python3 tools/make_edge_tiny_tokenizer.py --vocab-size "$(python3 -c 'import json;print(json.load(open("qwen38_tiny/config.json"))["vocab_size"])')" qwen38_tiny make qwen38 >/dev/null - QWEN38_TINY=qwen38_tiny python3 -m unittest -v tests.test_qwen38_dashboard + QWEN38_TINY=qwen38_tiny python3 -m unittest -v tests.test_qwen38_dashboard tests.test_qwen38_brio + python3 tools/make_edge_tiny_tokenizer.py --vocab-size 64 qwen38_tiny_fp8 + QWEN38_TINY=qwen38_tiny_fp8 python3 -m unittest -v tests.test_qwen38_brio inkling-oracle: name: Inkling oracle (token-exact vs transformers) @@ -923,10 +950,15 @@ jobs: /tmp/tike - name: Generate the glm_tiny fixture run: cd c && python3 tools/make_glm_oracle.py - - name: Token-exact oracle (teacher forcing) - # The fixture is only trustworthy if the engine reproduces it, so assert - # that before reading anything else out of a run. - run: cd c && SNAP=./glm_tiny TF=1 COLI_TEMP=0 ./colibri 64 16 16 + - name: Oracle (30–32/32 teacher forcing + exact greedy) + # Preserve CONTRIBUTING.md's two TF near-tie allowance. Greedy remains + # exact, and invalid/non-finite results cannot consume the allowance. + run: | + cd c + SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16 + SNAP=./glm_tiny COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16 + - name: Oracle allowance and failure regressions + run: cd c && python3 tests/test_glm_oracle.py - name: Structural efficiency tests # test_cpu_vs_cpu_tok_s_stability is NOT in this list: it is a tok/s # bound (two runs within 25%) over a ~15 ms tiny replay -- it measured @@ -1204,6 +1236,30 @@ jobs: - name: Python test suite run: cd c && python3 -m unittest discover -s tests -p 'test_*.py' + gguf: + name: GGUF reader + converter tests + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + cache: pip + cache-dependency-path: c/tools/requirements-gguf.txt + # This job is what makes the numerical GGUF evidence real: without these + # deps those tests skip at import, so the generic `python` job alone only + # exercises test_gguf_reader.py (pure stdlib). + - name: Install GGUF test dependencies + run: pip install -r c/tools/requirements-gguf.txt + - name: GGUF reader, dequant, profile and converter tests + run: | + cd c + python3 -m unittest -v \ + tests.test_gguf_reader \ + tests.test_gguf_dequant \ + tests.test_gguf_olmoe_profile \ + tests.test_convert_gguf_to_olmoe + windows-python-focused: name: Python tests (Windows focused) runs-on: windows-latest @@ -1229,7 +1285,7 @@ jobs: # This job builds every engine on arm64 (NEON compile coverage), runs the # integer-kernel bit-exactness gate with the NEON branches live, and replays # the glm_tiny teacher-forcing oracle against a fixture generated on THIS - # runner (same-machine torch reference, so no cross-ISA float excuses). + # runner. As on x86, allow two TF near ties; greedy remains exact. oracle-arm: name: ARM (engines + NEON kernel exactness + tiny oracle) runs-on: ubuntu-24.04-arm @@ -1245,6 +1301,12 @@ jobs: run: pip install -r c/tools/oracle-requirements.txt - name: Build every engine (NEON branches must compile) run: make -C c colibri inkling kimi_k3 olmoe + - name: "DeepSeek V4 FP4 expert kernels: NEON arm bit-exact against the scalar arm (#1696)" + run: | + cd c + # batch (prefill) and S=1 (the matvec the decode uses) + bash tools/bench_fp4_matmul.sh 32 256 128 1 + bash tools/bench_fp4_matmul.sh 1 256 128 1 - name: Integer-kernel exactness, NEON branches live (#1081) run: | cd c @@ -1252,5 +1314,10 @@ jobs: /tmp/tike - name: Generate the glm_tiny fixture on this runner run: cd c && python3 tools/make_glm_oracle.py - - name: Token-exact oracle on ARM (teacher forcing) - run: cd c && SNAP=./glm_tiny TF=1 COLI_TEMP=0 ./colibri 64 16 16 + - name: Oracle on ARM (30–32/32 teacher forcing + exact greedy) + run: | + cd c + SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16 + SNAP=./glm_tiny COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16 + - name: Oracle allowance and failure regressions on ARM + run: cd c && python3 tests/test_glm_oracle.py diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 1ef3d3af7..4afa56adc 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -84,7 +84,7 @@ jobs: - uses: actions/setup-node@v4 with: - node-version: '20' + node-version: '22' cache: npm cache-dependency-path: web/package-lock.json diff --git a/.gitignore b/.gitignore index d1564447e..eeba7b078 100644 --- a/.gitignore +++ b/.gitignore @@ -30,6 +30,8 @@ c/deepseek_v4.exe c/deepseek_v41 c/deepseek_v41.exe c/COLI_V4_UNIT_*.o +c/deepseek_v4.cflags +c/deepseek_v4.cudaflags # ...and the ownership-test objects, which the same Makefile puts in a build/ # subdirectory (V4_OWN_DIR) rather than next to the sources. #868 caught the # twelve in c/, not the four in here, so `make check` still left `?? c/build/`. diff --git a/CHANGELOG.md b/CHANGELOG.md index 39d9bb452..e6a5f546d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,6 +3,395 @@ All notable changes to colibrì are documented here. Format follows [Keep a Changelog](https://keepachangelog.com/). +## [1.12.1] — 2026-09-24 + +95 pull requests since v1.12.0, 82 of them from contributors. Two tokenizers +brought back to the reference, brio on the ninth engine, `coli chat` working +again at the default context on two families, and a placement decision that +is now measured on the card in front of it instead of predicted. + +### Tokenizers, measured against the reference + +- **#1654**: qwen36 tokenized differently from HF `tokenizers` in two ways. + An added token right after punctuation was encoded as text (`X.<|im_end|>` + was 7 tokens instead of 3, every chat turn ending in punctuation paid +4, + #1653), and a whitespace run followed by a non-space was one piece where + the regex's `\s+(?!\S)` leaves the last char to the next one, so every + indented line of code tokenized differently. Measured on the real + vocabulary: 2,803 lines and blocks of code, Markdown, Chinese and + Japanese went from 757 identical to 2,803, with 5.4% fewer tokens. +- **#1656**: OLMoE's `tokenizer.json` has no Split, a bare ByteLevel with + `use_regex`, for which HF runs the original GPT-2 pattern; `tok.h` applied + cl100k. A GPT-2 family in `tok.h`: 1,560/1,708 identical before, 1,708/1,708 + after. The same measurement on GLM-5.2/5.3/5.3-Flash, DeepSeek V4 and V4.1, + Inkling and Qwen3.8 came back identical on every case. + +### Brio and the serve contract + +- **#1662**: `POST /v1/systemone`, the request and the reply of TypeSafe's + Jev API, served by the brio channel: a client written for it points at + colibri and changes the base URL. `noul` is a yes/no question, `choice` + scores the labels with their descriptions in the text, `score` the level + numbers with the expected value and the legend; `confidence` by their + documented formula. Any `model` name is accepted on that route. Measured on + the real Qwen3.6: the three-question example of the docs in 1m46 with the + state read once. +- **#1655**: the DeepSeek V4 engine speaks the numeric channel (`logprobs=k`, + `pin=1`, `max_tokens=0`), so `/v1/brio` works on the ninth engine instead + of answering 500 (#1648). The head that used to keep only its argmax now + returns the whole row; `ECHO` per prompt position during prefill, the + prompt-end scores kept with a state snapshot on `pin`, and a logprob tail + on every `DATA` frame during generation. The tiny fixture pins that the + best `ECHO` token equals the greedy token from the same prefix, and that + the pinned predictor equals a cold prefill's. +- **#1659**: qwen36 and qwen38 refused a request when `prompt + max_tokens` + exceeded the context, and the gateway's default budget for these two + families is 8192, the whole default context: every request without + `max_tokens` and every `coli chat` message answered 400 on a two-token + prompt (#1641). `max_tokens` is now a ceiling, clamped to the room the + prompt leaves, as GLM and DeepSeek V4 already did; only a prompt that does + not fit is refused. `docs/api.md` states the rule, `docs/qwen38.md` names + `Q38_MAXT` as the variable `--ctx` becomes. + +### The dense trunk in VRAM, measured before it is placed + +- **#1657**: qwen36 offers the rest of its dense trunk to the VRAM placer: + the DeltaNet out_proj (`dnout`), the attention q/k/v/o (`attnproj`) and + the shared expert (`shexp`), about 650 MB more of int8 on the 35B beside + `lmhead` and `dnproj`. Measured on four Tesla M10 by the reporter of + #1652, every placed component ran slower than the CPU (lm_head 68.8 ms + against 41.7), so the engine now times one GEMV both ways at startup and + withdraws the whole automatic placement when the GPU loses, giving the + VRAM back to the experts: `auto` equals `off` on that box, byte-identical + output. A hand-written `COLI_PLACE` stands; `COLI_TRUNK_PROBE=0` trusts + the placer. + +### Performance + +- **#1664**: qwen36's dense trunk and routed experts multiply with integer + dot products. The activation is quantized to int8 once per call and the + weights, int8 rows or int4 planar blocks, meet it with maddubs / vpdpbusd + instead of a float conversion per weight; the integer kernels move from + `quant.h` into `idot.h`, shared by every engine. Measured on the 35B, 8 + threads, every expert resident: decode 6.71 to 8.23 tok/s (+22.6%), lm_head + 12.6 to 10.1 ms/token, the expert compute 22.7 to 15.6, for +1.3% + perplexity on 4 x 512 tokens. Both are the default (`COLI_DENSE_IDOT=0`, + `QWEN_EXPERT_ACT=f32` restore the f32 kernels). `COLI_DENSE_BITS=4` with + `COLI_DENSE_INT4=` stores part of the trunk as int4 in blocks + of 64: opt-in, with the perplexity it costs per component in the docs + (lm_head alone +2.4%, everything +10%). +- **#1668**: qwen38's dense trunk (553 matrices, 3.6 G weights, 8 GiB of + BF16 read on every token, more than the ten routed experts) is kept on the + CPU as int8 rows with the BF16 copy released, and multiplied with the same + integer kernels; the routed experts' e4m3 blocks are decoded eight at a + time in registers and multiplied with FMA instead of one table lookup per + weight. Measured on the released Qwen3.8-Flash-Next-FP8, 8 threads, RAM + LRU 96 per layer: decode 0.61 to 1.42 tok/s, the trunk 434 to 85 ms/token, + lm_head 76 to 13, the expert GEMVs 388 to 140, peak RSS 32.2 to 28.5 GB; + prefill of 512 tokens 495 to 149 s. Perplexity on 4 x 512 tokens +0.5% + (two chunks lower, two higher); the vector FP8 kernel alone reproduces the + BF16 run to four decimals. Both are the default (`Q38_TRUNK_CPU_INT8=0` + keeps the BF16 trunk, `Q38_FP8_KERNEL=scalar` the table kernel). + +### Performance, from contributors + +- **#1606**: the K1b grouped int4 family gets a multi-row tile and AVX-512 + and AMX arms, and is no longer switched off on AVX-512 builds; exact on + all eight engines on a 16-core AVX-512 host. +- **#1239**: an SSE4.1 tier for the olmoe and qwen36 int8 GEMV, for hosts + with SSE4.1 but no AVX2; on the Sandy Bridge of #1652 decode went from + 2.45 to 3.54 tok/s. +- **#1313**: `matmul_fp8` computes four output rows per pass under clang, + where the contraction makes it bit-exact; GCC keeps the one-row kernel. +- **#1612**: qwen36 gains the GLM engine's `CACHE_ROUTE` lever, with the + VRAM tier as the first residency level, opt-in. +- **#906**: `DEGRADE_ZERO`, an opt-in policy that zero-fills a missed + expert slot below a gate-weight threshold instead of blocking on the + load (#865). +- **#1677**: qwen36 projects a prompt's DeltaNet inputs (qkv and z) on the + card in blocks of up to 256 rows instead of one row at a time; a 259-row + prefill makes 2 projection calls instead of 259, with the convolution + history and the recurrent state checked against the CPU run. The paired + microbenchmark on an RTX 4070 read 4 to 10x per projection. +- **#1674**: qwen36's attention projections the tier placed in VRAM answer + a whole prompt batch with one call per matrix; a failed call turns only + that handle off and the prompt continues on the CPU. +- **#1676**: Kimi K3's streaming CUDA expert keeps the gate, up and SiTU + intermediates on the device and applies down there (one fused entry + point, optional in the DLL: an older backend keeps the three-call path). +- **#1673**: the streaming MXFP4 matmul reuses one grow-only device scratch + per card instead of allocating and freeing weights and scales on every + call. +- **#1559** (kreuzzelg): `convert_qwen36.py --down-bits 8` writes the mixed + expert layout, int4 gs64 gate/up and int8 down in one slab (5.7 bits per + weight against gs64's 4.5); the engine tells it apart by size and reads + each matrix in its own format on the CPU path, and refuses the VRAM tier + with a line. It is the knob behind the #1370 numbers: on wikitext-2 the + int8 down alone recovers a quarter of the gap between gs64 and all-int8, + the rest sits in gate/up. A measurement tool and a middle step, not the + answer to the gap. +- **#1730** (mfethe1): DeepSeek V4's FP4 expert kernels, the prefill batch + and the decode matvec, get a NEON arm; arm64 used to take the scalar + arm, which is why Apple Silicon prefilled at decode speed (#1696). Bit + identical to the scalar arm, and the ARM CI job now checks that on every + change; 17 to 22x on the kernel at the V4 expert shapes on an M-series + Mac, as measured by the author. +- **#1716** (jtinbergen): qwen36 quantizes its dense weights to int8 while + loading instead of keeping an f32 copy first, and converts f16/bf16 with + SIMD. On the 35B the resident set after load goes from 9.2 to 4.8 GB; + the generated text and the perplexity are identical to before. +- **#1286** (cameron): the grouped int4 GEMV and the fused gate/up GEMV get + an SSE4.1 arm for CPUs without AVX2 (Ivy Bridge and older). It vectorizes + across output rows, so each lane runs the scalar row's exact sequence and + the result is bit-identical; forced-SSE4.1 tests at -O1, -O3 and without + FP contraction pin it. 2.2 to 2.7x on the isolated kernel on a dual + E5-2680 v2. +- **#1686** (DebugSultan): qwen38's prefill chunk (`Q38_PREFILL_BATCH_ROWS`) + and workspace (`Q38_PREFILL_WORKSPACE_MIB`) are runtime knobs, the expert + load batch is no longer capped at top-k, and the QSA ranking and + attention run per position in parallel at prefill. Measured on the + released checkpoint on top of the int8 trunk: output byte-identical at + every chunk width, no speed change on our 16-core server; the knobs are + there for hardware where the chunk binds. + +### Fixed + +- **#1650** (bokiko): a Qwen3.8 pin snapshots the recurrent and PLE state + but reuses the live attention and indexer rows; after an unrelated prompt + overwrote those rows, returning to the pin could change brio logprobs + without a warning. The engine now records the token identity of the live + rows (`kv_prefix.h`) and refuses a stale pin or prefix restore; image rows + are tainted. Wire regressions run on the BF16 and FP8 fixtures. +- **#1626**: `SNAP` is the model directory for every non-GLM engine, so + `coli run` stops handing them a leftover environment (#1600). +- **#1604**: glm53 honours `Mat.resident` in the Vulkan gate and frees the + Vulkan handle in `mat_release`. +- **#1321**: glm53 sizes its expert cache around the model rather than + around `MemAvailable`, which the page cache had been inflating. +- **#1588**: qwen36 refuses loudly on a failed encode-buffer realloc instead + of writing through NULL. +- **#1546**: `coli convert` routes OLMoE to `convert_olmoe_merged.py`. +- **#1630**: olmoe emits the `ROUTE_TRACE` records it announced; the stream + used to be a zero-byte file. +- **#1610**: `v41_dsml.py` is staged during installation (and the nix flake + bumped). +- **#1511**: the GPU test suite builds under HIP on gfx1151. +- **#1658**: `test_mem_available` compared two reads of available memory + with `==` and failed on a busy Windows runner; a quarter of a GB of + tolerance. +- **#1670**: DeepSeek V4.1 read only its argv cache cap, so `RAM_GB=120` + on a 128 GB box left the engine at eight expert slots per layer and + 23.8 GB of RSS (#1666). With `--cap` omitted, `coli chat`, `coli serve` + and `coli web` now size the cache from the resource plan, with `RAM_GB` + or `--ram` as the budget; an explicit `--cap`, a measured profile and an + auto-tier plan keep precedence. +- **#1671**: DeepSeek V4.1 treats `max_tokens` as a ceiling like the other + engines (#1641): a fitting prompt with a large request generates what + the context leaves, a score-only prompt may fill the context, and a + prompt one token over it is refused instead of silently truncated. +- **#1675**: resident MXFP4 tensors on CUDA carried O float scales where + the kernel reads O x ceil(I/32) exponent bytes: short buffers were + over-read and long ones truncated. One format-aware size for upload, + refresh, accounting and release. +- **#1678**, **#1679**, **#1680**, **#1682**, **#1683**, **#1684**: the + Qwen CUDA tier's lifecycle, end to end. Shutdown releases every resident + expert, projection handle and host table after parked callers resume; + a failed gate, up or down upload frees what it had already allocated; + the expert budget charges the three scale buffers at their own sizes + (two experts used to be admitted where one fit); a failed result + collection stops inference instead of publishing a partial MoE sum; a + failed or explicitly disabled tier start unwinds its storage and + synchronization objects, and a second init cannot overwrite a running + tier; the CUDA backend validates the whole device list before touching + state and keeps live contexts on a repeated init. Fault-injected on the + fake backend, then run together on an RTX 4070 under compute-sanitizer + with zero errors and zero bytes leaked. +- **#1669**: `test_systemone_api` imports its scoring engine relative to + its package, so an installed `tests` package no longer breaks discovery. +- **#1697** (kevin9327): the dashboard redesign had dropped the reasoning + stream: thinking tokens arrived on `delta.reasoning_content` and vanished, + the bubble stayed empty until the answer and a stop during thinking lost + the turn. The stream is read again, rendered as its own folding block, + counted in the rate and the time to first token, and a unit test pins the + split. +- **#1693** (namespaceMarcello): with `PILOT` on, OLMoE could read the same + expert twice, once from the prefetcher and once from the forward pass, + into two slots; a slot being read now keeps a reservation in the index + (the `colibri.c` pattern) and the second caller waits for the first read + to publish. Three model-free scenarios pin it. +- **#1695** (namespaceMarcello): the prefill echo state and `serve_echo` sit + under the same `QWEN36_NO_MAIN` guard, so the segment build no longer + warns about a function it never gets; the full build is byte-identical. +- **#1724** (GenericRikka): Qwen3.6 decoded ``, `` and the + tool tags to nothing, because they live only in the tokenizer's + `added_tokens`; with thinking on, the closing tag never reached the + gateway and the whole answer came back as `reasoning_content`. The + non-special added tokens are decoded now; special ones such as + `<|im_start|>` still decode to nothing. +- **#1734** (tarazum): stopping `coli serve` closes the engine's stdin and + waits for it to exit on its own before the hard-stop ladder, so the + engine's teardown runs; qwen36 never saved its `HEAT_FILE` under `coli + serve` (#1733). On Windows the gateway handles SIGBREAK and the engine + runs in its own process group. +- **#1726** (kevin9327): Inkling measured no RAM on Windows and sized its + expert cache to 16 per layer; it uses the shared probe now, which on + Linux and macOS reads the same numbers as before. +- **#1735**: on GNU Make 3.81, the system make on macOS, `.build-config` + was never written and every build relinked (#1732). +- **#1731** (bokiko): the DeepSeek V4 CUDA object rebuilds when the nvcc + command changes, so a new `CUDA_ARCH` no longer links the old object. +- **#1728** (crichalchemist): `make test-c VK=1` built 43 test binaries + without the Vulkan object; they link it now. +- **#1712** (kevin9327): Kimi K3, Inkling and OLMoE now treat `max_tokens` + as a ceiling like the other engines; `coli chat`'s default of 16384 + answered 400 on every Kimi and Inkling message against their 8192-token + window. +- **#1713** (kevin9327): `coli plan`, `doctor` and `--auto-tier` export the + variable that actually sizes the expert cache on Kimi K3 + (`K3_EXPERT_GB`) and GLM-5.3 (`GLM53_EXPERT_GB`); only `RAM_GB` was + exported, which neither engine reads as the cache size. +- **#1711** (kevin9327): on Windows, a Kimi K3 `CUDA_DLL` build and a HIP + host were refused by `--gpu` as CPU-only; the probe reads the backend DLL + name the host was built with, and the launcher maps `--gpu` onto Kimi's + `K3_CUDA`. +- **#1710** (kevin9327): a `tools[]` entry whose `function` is not an object + answered HTTP 500 from the GLM and DeepSeek renderers; it is the 400 that + `generation_options` already had. +- **#1721** (monotophic): every frame the gateway writes to the engine is + checked, short writes are completed, and a failed `CANCEL` or `STOP` + drops the request's pending entry and answers a named 500 instead of a + silent close. +- **#1714**, **#1719** (benmaster82): the brio options form pins the shared + state, so per-question requests on one document read it once; the web + page can stop a scoring run, and duplicate options are removed on both + clients. +- **#1709** (kevin9327): regenerating a turn with pictures sends them again + and leaves the composer alone. +- **#1646** (Stamina9): qwen38 says once, on stderr, why the parallel expert + read path is not taken (disabled, cache smaller than the route, no FP8 + scale bank, repeated expert, converted layout). +- **#1708** (wittchen): every `VK=1` build of glm53 failed to compile on a + misplaced parenthesis. +- **#1707** (namespaceMarcello): the DeepSeek V4 unit objects rebuild when + the build flags change, so a CUDA engine build followed by `make test-c` + no longer links the wrong objects (#1702). +- **#1715** (namespaceMarcello): seven GLM-5.3 harnesses matched the + unittest glob and counted as zero tests; they are renamed, a skip exits 2, + the two tiny oracles run in CI, and a discovery test catches the next + empty module (#1700). +- **#1622**: DeepSeek V4's REAP checkpoints store each expert as six + per-matrix records; the engine read them through buffered pread and + counted every one as a direct-I/O fallback (36% of expert reads on the + 150B, #1615). Each segment now goes through the aligned direct window, + with a regression on a generated per-matrix fixture. +- **#1597**: a replayed tool call whose `arguments` parsed as JSON but was + not an object (`"[1, 2]"`, `"5"`) answered HTTP 500 from the GLM renderers + before the engine was asked anything; it renders the call without + arguments, as every other renderer already did. +- **#1624**: the five gcc 13 warnings left in `make check` are gone, and + `st_index_load` refuses an index path that would not fit its buffer + instead of opening a truncated one, with a long-path case in the tests. +- **#1651**: a `pyflakes` pass over the launcher, autotune, the family + registry and the qwen36 converter: a `measure()` defined twice, a + `readline` import without a fallback, a stray f-string, dead variables. +- **#1580**: `make qwen36 CUDA_DLL=1` on Windows reached GNU make's implicit + rule and built a CPU-only binary; a bare `qwen36` alias, a `.build-config` + prerequisite so a CUDA_DLL change rebuilds, a loader-against-header parity + test, and the Windows CUDA tier documented. +- **#1556**: `coli plan` on macOS said "no supported GPU detected" on every + Mac; it now lists the Metal device by name, as identity only, without + pretending unified memory is a VRAM budget. +- **#1691**: the installed launcher invoked as `/bin/coli` or `/sbin/coli` + on a merged-/usr system derived `/libexec/colibri` instead of + `/usr/libexec/colibri`, because `abspath` kept the alias (#1689, + florin65's patch): `realpath` first. A test runs the launcher through + such an alias, and another checks that every root module the launcher + reaches is in the `make install` list, the gap #1610 closed by hand. + +### Tools and the gateway + +- **#1425**: a general GGUF reader, pure stdlib, and a converter from GGUF + OLMoE checkpoints to a colibri container, with the numerical evidence in + its own CI job. +- **#1497**: opt-in prompt-injected tool calling for the families without + native tool tokens (OLMoE, Qwen3.6), behind `COLI_TOOL_FALLBACK=1`, with a + two-turn end-to-end test. +- **#1355**, **#1357**: durable per-request results and strict stdout + classification in the eval harness, and the logprob-gap check gated on + the engine preamble. +- **#1687**: `GET /metrics` in Prometheus text format, behind the API key: + four gauges, six outcome counters and four histograms (queue wait, slot + occupancy, first output, engine call), no request labels, no new + dependency. The admission scheduler distinguishes completion, failure + and cancellation, lets a request use a free slot that no earlier waiter + reserved, and joins the keepalive pump before the slot is released. +- **#1717** (enitimeago): the web chat offers Continue on the last + assistant message when it stopped at the token limit, by hand or on an + error, and only when `/health` says the server continues assistant + turns (#1699). +- **#1402** (enitimeago): a request whose last message is a non-empty + `assistant` turn continues that turn instead of answering in a new one, on + `/v1/chat/completions` and `/v1/messages`, for all nine families (Kimi K3 + frames the open turn engine-side); the prompt ends inside the turn as the + official template renders it without a generation cue. On by default, + `COLI_CONTINUE_ASSISTANT=0` restores the old behaviour; refused together + with tools or a turn ending in whitespace, with a 400 that says why. Each + renderer is pinned against the vendored template (#1401). +- **#1102** (monotophic): checkpoint-faithful FP8 containers that store + `kv_b_proj` as fmt=8 could load but not decode attention; the absorb path + now decodes fmt=8 on CPU (bit-exact against the reference) and CUDA + (within the documented tolerance), and the kv_b sharding refuses by name + the formats it cannot serve, which also closes two silent misreads of + fmt=5 and fmt=6. +- **#1395** (rybruscoe): `COLI_EXACT_VERIFY=1` makes the speculative verify + batch token-exact against sequential decode, at a measured cost on the + dot itself; off by default, the default path is unchanged. +- **#1720** (monotophic): a request carrying `seed` is accepted and the seed + ignored, as `docs/api.md` now says, instead of a 400; no determinism is + implied. +- **#1605**: `ORACLE_STRICT=1` makes a GLM oracle comparison exit non-zero + when it fails, token-exact by default with `ORACLE_TF_MAX_MISMATCHES` for + the documented teacher-forcing allowance; references are validated before + the comparison and non-finite logits cannot pass. Both oracle CI jobs run + real-process regressions against it. +- **#1705**: `tools/benchmark_baseline.py`, a collection protocol on top of + the HTTP harness for a repeated three-engine serving baseline: one frozen + manifest (hardware, model and template identity, per-engine launch + settings, cache and speculation policy), a rotating plan over a + concurrency matrix, one collector per engine and round that manages no + server, and a comparison that keeps failed and missing cells visible and + distinguishes matched artifacts from deployment comparisons. No results + are bundled and no ranking is emitted. +- **#1688**: `tools/benchmark_http_serving.py`, a stdlib HTTP streaming + benchmark over fixed JSONL conversations: closed-loop or paced arrivals + (periodic or Poisson, seeded), warmup separated from measurement, + first-output and duration SLOs, latency percentiles and usage-based + token throughput; a truncated or malformed stream is a failure, not a + sample. + +### Docs + +- **#1639**, **#1644**: the README shows brio mode and the dashboard as it + is: the workspace, the Brain page (the measured expert atlas as a cortex, + and a region inside it) and the Profiling page, in four languages; + `docs/api.md` describes the four pages instead of the old console. +- **#1492**, **#1643**: a Japanese README, and its banner at the shipping + version, which the banner test now checks in every language. +- **#1634**: the multi-disk guide states measured gains and limits instead + of "twice the bandwidth", with Bash and PowerShell examples. +- **#1617**: connecting the pi coding agent to `coli serve`. +- **#1619**: `expected_bytes` identity versus physical extent for + int4-rans256-g0 (#1273). +- **#1618** (bherald): `docs/qwen38.md` no longer calls the engine text-only; + the vision tower and the gateway image path shipped in 1.12.0. +- **#1649**, **#1647** (Suraj2105-1): the musl CI job runs on Alpine 3.24 + and the release pipeline on Node.js 22, ahead of the 3.21 and Node 20 + end of life. +- **#1568** (Yoruxyv): an Indonesian translation of the dashboard. +- **#1681** (XBold): the README and `docs/qwen38.md` no longer say Qwen3.8 + has no GPU backend; the CUDA VRAM expert tier and the int8 trunk in VRAM + shipped in 1.12.0. + ## [1.12.0] — 2026-09-20 81 pull requests since v1.11.0. A new way to ask a model a closed question, diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 28cebe5da..687351c7a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -5,7 +5,7 @@ Keep changes focused and preserve Colibri's dependency-free default CPU path. ## Branches - **`main`** is the stable branch. It's what users clone, and it stays known-good - (engine always passes the token-exact oracle: `SNAP=./glm_tiny TF=1 ./colibri 64 16 16`, + (engine always passes the oracle: `SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16`, run from `c/`). The `glm_tiny` fixture is generated, not committed -- `python3 c/tools/make_glm_oracle.py` builds it (needs torch). - **`dev`** is the integration branch. **Open your PR against `dev`.** Reviewed PRs @@ -13,9 +13,37 @@ Keep changes focused and preserve Colibri's dependency-free default CPU path. it into `main`. This keeps `main` clean instead of taking every PR one at a time. Every PR — on either branch — is reviewed for a clean build (0 warnings), the oracle -(~30-32/32 TF depending on floating-point near-ties + 20/20 greedy), and its own +(30–32/32 TF depending on floating-point near ties + 20/20 greedy), and its own targeted validation before merge. +After generating the fixture, run both enforced comparisons from `c/`: + +```sh +SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16 +SNAP=./glm_tiny COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16 +python3 tests/test_glm_oracle.py +``` + +`ORACLE_STRICT=1` makes a failed comparison exit with status 1. Strict mode is +token-exact by default. The teacher-forcing command above explicitly allows at +most two mismatches, preserving the 30–32/32 acceptance range for this fixture; +greedy comparison always requires every continuation token to match. Non-finite +output and incomplete generation fail regardless of the mismatch allowance. +Invalid reference arrays/JSON fail in either mode. +Without strict mode (or with `ORACLE_STRICT=0`), a completed comparison remains +report-only for diagnostic/benchmark callers; its exit status is not a correctness +gate. Strict mode rejects `REPLAY`, `CONSIST`, serving, text generation, and other execution +modes that would bypass the comparison. + +`ORACLE_TF_MAX_MISMATCHES` is a nonnegative integer smaller than the number of +TF positions; it is used only for strict teacher-forcing comparisons. Unset or +`0` requires exact TF agreement. Generate the reference on the test host and +inspect near ties with `TF=1 DEBUG_LOGITS=1`; an allowed mismatch is still printed +and does not establish token-exact agreement. The regression script checks the +two-mismatch boundary, optional exact mode, and rejection of a single greedy +mismatch. It validates the original TF fixture within the same allowance. Routine +`make check` covers reference validation without needing torch or a generated model. + ## Local checks Run the lightweight checks locally: diff --git a/README.it.md b/README.it.md index a7de181fe..86dbed613 100644 --- a/README.it.md +++ b/README.it.md @@ -4,7 +4,7 @@

Discord · - English · 简体中文 · 繁體中文 · Italiano + English · 简体中文 · 繁體中文 · Italiano · 日本語

**Motore piccolo, modello immenso.** Esegui **modelli MoE di frontiera — da 744 @@ -34,7 +34,7 @@ ma non ridefinire il modello di nascosto. ``` $ ./coli chat - 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU + 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU ✓ ready in 32s · resident 9.9 GB › ciao! ◆ Ciao! 😊 Come posso aiutarti oggi? diff --git a/README.ja.md b/README.ja.md new file mode 100644 index 000000000..9f4bbdca9 --- /dev/null +++ b/README.ja.md @@ -0,0 +1,707 @@ +

+ colibrì — 小さなエンジン、巨大なモデル +

+ +

+ Website + Latest release +

+ +

+ Website · + Discord · + English · 简体中文 · 繁體中文 · Italiano · 日本語 +

+ +**小さなエンジン、巨大なモデル。** ストレージ・RAM・VRAM を単一の推論階層として扱う +(AI メモリのマルチティア化)ことで、**744B から 2.8T パラメータのフロンティア MoE モデル**を、 +コンシューマー向けや異種混在のハードウェア上で、エンジン依存ゼロの純粋な C で実行します。 + +現在動作するのは 9 つのファミリーです: **GLM-5.2/5.3**(744B)、**GLM-5.3-Flash**(321B、 +ビジョン対応)、**Inkling**(975B)、**Kimi K3**(2.8T)、**DeepSeek V4 Flash**(284B)、**DeepSeek V4.1 Flash**(552B、ビジョン対応)、 +**Qwen3.8-Flash-Next**(125B + 51B n-gram)、**Qwen3.6**(35B-A3B)、そして +**OLMoE**(7B)—— +それぞれが C ファイル 1 つで、同じ `coli chat` / `coli serve` / `coli web` フロントエンドを共有します。 +[全モデル一覧 ↓](#other-supported-models) + +> **Colibrì は今日すぐに動かせる推論エンジンであり、同時にオープンな研究 +> プラットフォームでもあります。** 主な目標は、ソフトウェアとハードウェアの境界全体—— +> モデルフォーマット、メモリ階層、ストレージ I/O、配置、スケジューリング、カーネル、 +> 投機的デコード、CPU/GPU のオーバーラップ——にわたって推論側の性能を追求し、 +> 大規模モデルが希少なハードウェアに依存せず、より低コストで動くようにすることです。 + +Colibrì は VRAM・RAM・ストレージを単一のマルチティア階層として扱い、意図的に +攻めたシステム上のアイデアを試す場となっています。そのため **速度に SLA はありませんが、 +セマンティクスは厳格に保証します**。実験は再現可能なエンドツーエンドの計測によって +採用に値することを示さなければならず、デフォルトのポリシーは **モデルの精度やルーターの +セマンティクスを黙って変更することは決してありません**。高速メモリが不足すると速度は +落ちるかもしれませんが、それによってモデルが密かに別物になることはあってはなりません。 + +``` +$ ./coli chat + 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU + ✓ ready in 32s · resident 9.9 GB + › ciao! + ◆ Ciao! 😊 Come posso aiutarti oggi? +``` + +## 動作の様子 + +

+ colibrì Web ダッシュボード — ライブメトリクス、ハードウェアパネル、エキスパートのティア +

+

Web ダッシュボード(./coli web): 744B モデルが 4 tok/s、TTFT 1.6 秒、ディスク 0 で動作 — +6× RTX 5090 上でエキスパートを完全常駐させ、ライブのトークンメトリクス、ターンごとの時間内訳、 +VRAM/RAM/ディスクのティアバー、隅にはライブのミニ脳を表示しています。

+ +

+ Brain ページ — GLM-5.2 の計測されたエキスパートアトラスを皮質として描画、入れる 10 の領域 +

+

Brain ページの Explore 表示: GLM-5.2 の計測されたエキスパートアトラスを皮質として描画します。 +特性が明らかになった 13,260 個のエキスパートが 10 の領域(Python、SQL、数学、詩、法律、中国語…)に分かれ、位置は学習された埋め込みではなく +計測されたルーティング親和性です。領域を選ぶとその中に入れます。Live routing 表示は実際に動いているモデルに切り替わり、 +エキスパートごとに 1 セル、色はストレージのティア、1 ターンでルーティングされたエキスパートは白く光ります。

+ +

+ Python 領域の内部 — 1,142 個のエキスパート、選択した 1 つとその計測された親和性 +

+

Python 領域の内部: 1,142 個のエキスパートが星座として並び、それぞれにレイヤーと番号のラベルが付きます。パネルはその 1 つ、 +レイヤー 17 のエキスパート 178 を表示しています。エントロピー 3.13 のジェネラリストで、計測された親和性は Python 20.2%、JSON 14.6%、 +会話 14.2%、SQL 13.3% です。

+ +

+ Profiling ページ — エンジンが各ターンで時間を使う場所 +

+

Profiling ページ: エンジンが各ターンで時間を使う場所をフェーズごとに示し、直近 30 ターンを推移として表示します。 +ここでは CPU マシン上の Qwen3.6: プロンプト 36 トークンと生成 55 トークンで壁時計時間 19.0 秒、2.9 tok/s、 +ディスクサービス 11.4 秒は計算と重なっています。

+ +## 研究のミッション + +Colibrì があれば、プライベートなフロンティアモデルへのアクセスが、ハイパースケーラー級ハードウェアの入手可能性に制限されることはありません。 + +マルチティア機能によって、Colibrì は **推論エンジンのパイプラインを積極的に最適化し、 +プロプライエタリなハードウェアへの依存を取り除きます**。 + +私たちの実務上のミッションには、重みの表現方法と移動方法を変えること、何を VRAM・RAM・ +ストレージに置くかを決めること、異種計算資源をオーバーラップさせること、起動と同期の +オーバーヘッドを減らすこと、スパース性と再利用を活用すること、そして新しいデコード +アルゴリズムを試すことが含まれます。慣習的だからという理由だけで守られるものはなく、 +マイクロベンチマークで速く見えるという理由だけで採用されるものもありません。決め手となるのは +実機でのエンドツーエンドの推論結果であり、スループット、レイテンシ、メモリ、コストと +並んで、正しさと品質も計測されます。 + +その実際的な帰結が **アクセシビリティ** です。すでに持っているハードウェアで +744B パラメータのモデルを動かし、すべてのエキスパートが発火する様子をリアルタイムで眺め、 +それを実現しているコードを変更できます。API の向こうにある知能を借りるのではなく、 +それを *手にする* ——調べ、計測し、改善する——のです。エンジンは意図的に小さく保たれており、 +次の有用な最適化は、それを計測しようとする誰からでも生まれ得ます。 + +## コア技術と計測結果 + +- **ティア容量に縛られない単一の階層。** VRAM・RAM・NVMe は同じ重みを置く配置ティアです。 + 高速メモリの制約は速度を変えるだけで、モデルのセマンティクスは変えません。 +- **重みのための JIT。** すべてのエキスパートをロードする代わりに、計測されたルーティングの熱量が + レイヤーごとの LRU、学習されたピン留めホットストア、1 レイヤー先のプリフェッチを駆動します。 + 繰り返しのあるワークロードでは効果がありますが、履歴は過学習し得るし、先読みは一部の + ホストでは逆効果になり得るため、どちらも約束ではなく計測可能なポリシーとして扱われます。 +- **I/O はエンジンの一部。** バッチ化されたエキスパートの和集合、読み込みと計算のオーバーラップ、 + `O_DIRECT`、重み付きデュアル SSD ストライピングによって、ストレージのレイテンシが + タダであるかのように装うのではなく、ストリーミング経路そのものに取り組みます。`O_DIRECT` は + ドライブ依存であり、デュアル SSD はまだより幅広いコミュニティによるエンドツーエンドの A/B を必要としています。 +- **異種混在実行。** CPU、CUDA、Metal、NUMA メモリ、部分的または完全なエキスパート常駐は + 1 つのランタイムを共有し、マシンに応じて組み合わせられます。どの組み合わせが有利かは、 + 計算能力、帯域幅、常駐状況、ワークロードによって決まります。 +- **別のモデルにすることなく状態を圧縮。** トークン単位で完全一致するフォワード検証、 + 57 分の 1 に縮小された MLA KV 状態、永続化されたウォームな会話、忠実な DSA により、 + 最適化を正しさに結び付けています。これらはメモリ・レイテンシ・正しさの特性であり、 + 一律のスループット向上を主張するものではありません。 +- **元が取れる場合にだけ使う投機的デコード。** ネイティブ MTP と文法強制ドラフトは + エンドツーエンドで計測され、受理率が検証コストに見合わない場合は無効化できます。 + +## 未検証の仮説、実験、そして協力の方法 + +Colibrì は、制御されたエンドツーエンドの A/B が示すまで、最適化を仮説として扱います。 +現在の主な問いは次のとおりです: + +| 仮説 | これまでのエビデンス | まだ必要な実験 | +|---|---|---| +| ルーティング履歴は単純な LRU よりもうまくエキスパートを配置できる | 学習されたピンは繰り返しのワークロードを改善するが、プロンプトに過学習し得る | コーディング、チャット、多言語、長コンテキストのワークロードにわたる、ホールドアウトかつセッション横断の A/B | +| 複数の SSD は独立した帯域幅をデコード速度に変えられる | 重み付きミラー/分割ルーティングは実装・検証済みで、帯域幅モデルは妥当 | 実際に独立したコントローラ上での、コールドキャッシュ状態の 1 ドライブ対 2 ドライブによる GLM-5.2 実行 | +| ハードウェアを考慮したプランナーは、各マシンの最適構成に自動で近づける | RAM/VRAM の予算といくつかのバックエンドは現在すでに検出される | 生成されたプランを、ラップトップ、ワークステーション、NUMA ホスト、マルチ GPU システムにわたる制御されたパラメータスイープと比較する | +| ロスレスまたは品質上限付きの表現で、重みの移動を意味のあるほど減らせる | 正しさ/品質ゲート付きのフォーマットと量子化のアブレーションが存在する | 圧縮率だけでなく、品質、移動バイト数、レイテンシ、有用トークンあたりのコストを同時に再現する | +| ルーティングを考慮した投機的デコードは、ほぼ完全常駐に達する前でも元が取れる | MTP と文法ドラフトは動作するが、MTP はエキスパートヒット率約 85% 付近で 32% の損失も計測されている | 受理率、エキスパートヒット率、バッチの和集合、ドラフト深さにわたる損益分岐面をマッピングする | +| CPU/GPU のオーバーラップは、ボトルネックを移すだけでなく転送と同期を隠蔽できる | CUDA と Metal での改善はあるが、高速な CPU や低い常駐率ではそれが打ち消され得る | PCIe、ユニファイドメモリ、完全常駐マシンにわたる、ステージごとのプロファイルと 1 変数ずつの A/B | + +協力したいですか? いずれかの行を選び、ネガティブな結果も公開してください。ハードウェア、 +コミット、モデル/コンテナ、正確なコマンド、プロンプト、キャッシュ状態、スループット、 +TTFT、エキスパートヒット率、読み込みバイト数、品質チェックを記録し、1 つの変数だけを変えて +再実行し、生のログを添付してください。まずは +[CONTRIBUTING.md](CONTRIBUTING.md) から始め、 +[ベンチマークプロトコル](docs/benchmarking.md) と比較したうえで、 +[実験 issue を作成](https://github.com/JustVugg/colibri/issues/new) してください。 +ここでは、説明のつかない速い数値よりも、よく制御された失敗のほうが価値があります。 + +## アイデア + +744B の Mixture-of-Experts モデルは、1 トークンあたり約 40B のパラメータしか活性化せず、 +そのうちトークンごとに入れ替わるのは約 11 GB(ルーティングされるエキスパート)だけです: + +

+ 1 トークンあたり活性化するのはパラメータの約 5.4% のみ +

+ +つまり、モデルは高速メモリに *収まる* 必要はなく、**配置** されればよいのです: + +- **密な部分**(アテンション、共有エキスパート、埋め込み — 約 17B パラメータ)は + **int4 で RAM に常駐** します(約 9.9 GB)。 +- **19,456 個のルーティングエキスパート**(75 MoE レイヤー × 256 + MTP ヘッド、int4 で各約 19 MB)は + **ディスク上** に置かれ(約 370 GB)、レイヤーごとの LRU キャッシュ、学習されたピン留めホットストア、 + オプションの VRAM ティアとともに **オンデマンドでストリーミング** されます。 + +コアアルゴリズムは **重みのための JIT** だと考えてください。コンパイラの JIT は +プログラム全体をコンパイルすることはなく、実際に実行される部分を観察して、ホットパスを +ジャストインタイムでコンパイルします。colibrì は 744B のパラメータ空間に対して同じ賭けをします。 +パラメータは保持すべき常駐状態ではなく、ルーターが必要だと証明したまさにそのときに、 +異種混在のストレージ階層(VRAM / RAM / NVMe)にわたって **ステージングされるデータ** なのです。 +計測されたルーティングの熱量がどのエキスパートをどのティアに置くかを決め、ルーターは +1 レイヤー先を走ってプリフェッチがステージングのレイテンシを隠し、そして JIT と同様に、 +エンジンはあなたのワークロードを学習します。使えば使うほど、適切なエキスパートがホットになっていきます。 +これが機能するのは、ルーティングに計測可能な構造があるからです( +[エキスパートアトラス](https://github.com/JustVugg/colibri/issues/175) を参照)。 +そして構造はキャッシュ可能です。 + +エンジンは単一の C ファイル(`c/colibri.c`)と小さなヘッダ群だけで構成されています。BLAS も、 +実行時の Python も、GPU も必要ありません。 + +### ローカルクラスタモード + +コーディネーターはトークン生成、ルーティング、KV 状態をローカルに保持し、ディスクを +バックエンドとするエキスパートワーカーが、ルーティングされた FFN を他の Mac 上で実行します。 +あるレイヤーでルーティングされたバッチの和集合は 1 つの永続 TCP リクエストとして送られるため、 +1 トークンでエキスパートごとにラウンドトリップが発生することはありません。 + +オプションの登録サービスを起動します: + +```bash +./coli cluster coordinator --host 0.0.0.0 --port 8765 +``` + +各ワーカーで、同じ変換済みモデルをローカルに用意したうえで: + +```bash +./coli cluster worker --model /nvme/glm52_i4 --port 9100 \ + --coordinator http://COORDINATOR:8765 --advertise-host WORKER_IP +``` + +ディスカバリーを使ってコーディネーターを実行するか、静的な構成では `--cluster-workers +HOST:PORT,...` を指定します: + +```bash +./coli serve --model /nvme/glm52_i4 \ + --cluster-coordinator http://127.0.0.1:8765 +``` + +ワーカーが設定されていない限りトランスポートは無効なので、既存の単一マシンでの経路は +変わりません。密レイヤーのシャーディングやブラウザ/WebGPU ワーカーは、別の今後の拡張ポイントです。 + +## 仕組み + +### トークンごとの経路 + +

+ ルーティング → 和集合 → 配置 → オーバーラップ → 学習 +

+ +すべてのトークンのすべてのレイヤーが、同じ 5 つのステップをたどります。設計上の目標は、 +**配置が決めるのは常に速度だけ** ということです。エキスパートが VRAM から応答しようと +ディスクから応答しようと、ルーターの判断も重みの精度も同じです。 + +### 1 つのメモリ要件ではなく、1 つのメモリ階層 + +

+ VRAM / RAM / NVMe の 3 ティアによるエキスパート常駐 +

+ +### デュアル SSD: モデルのコピー 2 つで、読み込み帯域幅を 2 倍に + +ほとんどのマシンでデコードはディスク律速であり、エキスパートの読み込みは読み取り専用です。そこで **2 台目の SSD** があるなら、そこにモデルの完全なコピーを置き、エンジンに両方のドライブから同時にストリーミングさせましょう: + +```bash +COLI_MODEL=/fast/glm52_i4 COLI_MODEL_MIRROR=/second/glm52_i4 ./coli chat +COLI_DISK_WEIGHTS=9,3 ... # オプション: プライマリ,ミラーの帯域幅比(未指定なら起動時に計測) +``` + +各エキスパートは、2 台のドライブの計測された(または宣言された)帯域幅で重み付けされた決定論的ハッシュによって一方のドライブに割り当てられます。そのため readahead/PILOT のプリフェッチと要求時の読み込みは常に同じドライブに当たり、二重にキャッシュされることはありません。合計帯域幅は両ドライブの和になります — 9 GB/s + 3 GB/s の組み合わせでは、高速なドライブ単体よりもエキスパートの読み込みが約 33% 速くなり、OMP 並列のピン留め/ウォームアップのロードも両方からストリーミングされます。知っておくべき詳細: + +- ミラーは **起動時に検証** されます(ファイルごとのサイズと safetensors ヘッダがプライマリとバイト単位で一致する必要があります)。一致しない、または欠けているファイルは黙ってプライマリのままになるので、**部分的なミラーでも問題ありません** — 一部のシャードしか置けない小さめの 2 台目 SSD でも効果があります。 +- ミラーには **一切書き込まれません**: `.coli_usage`、`.coli_kv` およびすべてのサイドカーファイルはプライマリに残ります。 +- ミラーでの読み込みエラーはプライマリにフォールバックします(警告 1 回、クラッシュなし)。そのため実行中に 2 台目のドライブを抜いても、サーバーが落ちるのではなく性能が低下するだけです。 +- ルーティングがトークンを変えることはありません — 両コピーはバイト単位で同一であり、実行ごとの `MIRROR:` 統計行にはドライブごとに提供した GB 数が表示されます。 + +同じエンジンがあらゆる規模をカバーします。25 GB のラップトップではすべてがディスクから +ストリーミングされ(遅いが正しい)、大きなホストではエキスパート全体が常駐し +(`CUDA_EXPERT_GB=auto PIN_GB=all`)、ディスクはデコード経路から完全に外れます。 +ティアの間には **学習キャッシュ** があります。エンジンは *あなたの* ワークロードがどの +エキスパートにルーティングされるかを記録し(`.coli_usage`、ターンごとに更新)、最もホットな +ものを自動でピン留めします — colibrì は文字どおり、使えば使うほど速くなります。マルチソケットの +ホストでは、`COLI_NUMA=1` によって常駐する重みをメモリコントローラ間でインターリーブします +([#82](https://github.com/JustVugg/colibri/issues/82))。 + +モデル全体を置けない 2 台目のドライブ向けに、Colibri はすでに学習しているエキスパート履歴から +部分ミラーの優先順位を付けられます。まずいくつかの代表的なプロンプトを実行して `.coli_usage` に +ワークロードを反映させてから、ミラーを計画・ステージング・検証します: + +```bash +./c/coli mirror plan --model /fast/glm52_i4 --mirror /second/glm52_i4 \ + --budget-gib 200 --reserve-gib 20 +./c/coli mirror stage --model /fast/glm52_i4 --mirror /second/glm52_i4 \ + --budget-gib 200 --reserve-gib 20 +./c/coli mirror verify --model /fast/glm52_i4 --mirror /second/glm52_i4 +``` + +プランナーは safetensors ヘッダを直接読み、`COLI_MODEL_DIRS` から分割モデルのディレクトリをたどり、 +最もホットなルーティングエキスパートを提供できるシャードを優先します。ステージングがプライマリの +モデルを変更することはありません。一時ファイル経由でコピーし、指定された空き容量の予備を確保し、 +すべてのシャードを SHA-256 で検証し、既存のミラーシャードを削除することはなく、選択された +ミラーの準備が整って初めてレシートをアトミックに公開します。 + +### ディスクを二度待たない + +ミスはコストが高いため、エンジンは工夫の大部分をミスの回避とオーバーラップに費やします。 +各エキスパートの 3 つの行列は隣接して格納され、1 回の `pread` で読み込まれます。上限付きの +非同期 I/O プール(`PIPE=1`、デフォルト)は、常駐しているエキスパートが計算している間に +欠けているエキスパートをロードします。バッチ化された位置では一意なエキスパートを 1 回だけ読み込み +(**batch-union**)、ルーター先読みスレッド(`PILOT=1`)が次のレイヤーのエキスパートを +プリフェッチします — ルーティングは **1 レイヤー先を 71.6% の精度で予測可能** であることが計測されています。 +GPU では、常駐パイプライン(`COLI_CUDA_PIPE=2`)が残差ストリームをレイヤーをまたいでデバイス上に +保持するため、CPU のエキスパートループは中断されずに実行されます。Apple Silicon では実験的な +[Metal バックエンド](docs/metal.md) がユニファイドメモリ GPU 上でバッチ化されたエキスパート演算を行い、 +[Vulkan バックエンド](docs/vulkan.md) はエキスパートティア、密な射影、MLA アテンションのコアを、 +Vulkan 1.2 ドライバを持つあらゆる GPU にもたらします — Mesa/RADV 経由の AMD カードも含まれます +(RX 580 のようにベンダーのスタックがサポートを終了したカードでは唯一のバックエンドであり、 +RDNA4 では ROCm と互角です — [ベンチマークに関するメモ](docs/vulkan.md) を参照)。 + +> **実際の NVMe では `DIRECT=1` を計測してください。** O_DIRECT はページキャッシュをバイパスし、 +> DRAM キャッシュと帯域幅に余裕のあるドライブでは大きな改善になることが多いです(Blackwell/Windows +> マシンで `PIPE=1` と併用してデコード +34% を計測、GB10 の iobench で 4.25→9.69 GB/s)。 +> ただしドライブ依存であり、QLC/DRAM レスや仮想化されたディスクでは効果なし〜逆効果になり得ます。 +> まず試して、ハードウェアが報いてくれる設定を残してください。 + +### 忠実なモデル、圧縮された状態 + +フォワードパスは `transformers` のオラクルに対して検証されています(teacher-forcing で +通常 30〜32/32。小さなオラクルの 2 つの位置は浮動小数点上のほぼ同値で、ツールチェーンに依存します)。 +MLA アテンションは圧縮された KV 状態を保存します — 1 トークンあたり 32,768 個ではなく 576 個の +浮動小数点数(**57 分の 1**)— そしてそれを再起動をまたいで永続化します(`.coli_kv`)。 +会話は再プリフィルなしでウォームな状態から再開され、中断されなかったセッションとバイト単位で同一です。 +DSA スパースアテンション(GLM-5.2 の lightning indexer)は忠実に実装されており、全キーを強制的に +選択させたときに密なアテンションを正確に再現することで検証されています。 + +### 投機的デコードを、誠実に + +GLM-5.2 のネイティブ MTP ヘッドがトークンをドラフトし、メインモデルが 1 回のバッチ化された +フォワードでそれを検証します — 効果がある場合はフォワードあたり 2.2〜2.8 トークン。苦労して +得た 2 つのルールがデフォルトとして組み込まれています。MTP ヘッドは **int8** でなければならないこと +(int4 のヘッドは受理率が 0〜4% に崩壊します、[#8](https://github.com/JustVugg/colibri/issues/8))、 +そしてドラフトと検証は **同じ関数** を計算しなければならないこと — `SPEC_PIN=1` は両者を +1 つのカーネルファミリーに固定します(詳しい調査の経緯は [#163](https://github.com/JustVugg/colibri/issues/163) にあります)。 +文法強制ドラフト([`GRAMMAR=file.gbnf`](docs/grammar-draft.md))は、制約付き JSON 出力で +ほぼタダで受理率を上げます。投機的デコードが正味でプラスになるかはキャッシュの温まり具合に +依存します — 計測し、効果がなければ `DRAFT=0` を使ってください。 + +## 何を実現しているか + +

+ ハードウェアクラス別の計測デコード速度 +

+ +同じエンジン、同じ int4 コンテナ — ハードウェアが変えるのはエキスパートの置き場所だけです。 +[完全なベンチマーク表](docs/benchmarks.md) からのハイライト: + +- **6× RTX 5090、完全常駐:** デコード 5.8〜6.8 tok/s、TTFT 約 13 秒 + ([実験ログ](docs/experiments/glm52-6x5090-2026-07-12.md)) +- **128 GB の CPU のみのデスクトップ:** ウォーム時 約 1.8 tok/s([#200](https://github.com/JustVugg/colibri/issues/200)) +- **RTX 5070 Ti 1 枚のラップトップ級マシン:** GPU 常駐パイプラインで 1.07 tok/s + ([#273](https://github.com/JustVugg/colibri/issues/273)) +- **25 GB の開発マシン:** コールド時 0.05〜0.1 tok/s — このプロジェクトが始まった実証済みの下限であり、 + 今も誠実なベースラインです。 + +品質は仮定ではなく計測されています。int4 コンテナの量子化コストと、スケール粒度/回転の +アブレーションは [docs/benchmarks.md](docs/benchmarks.md#quality-benchmark) と +[#108](https://github.com/JustVugg/colibri/issues/108)/[#81](https://github.com/JustVugg/colibri/issues/81) にあります。 + +## はじめに + +必要なものは 2 つです: **プログラム**(数百 KB)と **モデル**(372 GB)。 +全プラットフォーム向けの手順は [クイックスタートガイド](docs/quickstart.md) にあります。 + +### 1. colibri を入手する + +**ビルド済みリリースをダウンロード** — Linux、macOS、Windows に対応し、コンパイラは不要です。 +[Releases](https://github.com/JustVugg/colibri/releases) から自分のプラットフォーム用の +アーカイブを取得して展開します: + +```bash +mkdir colibri && tar xzf colibri-v1.8.0-linux-x86_64.tar.gz -C colibri && cd colibri +python3 coli info # engine ready ✓ +``` + +中にはエンジン(`colibri`、Windows では `colibri.exe`)、`coli` ランチャー、その Python +ヘルパーが入っています。名前の変更や設定は不要です — `coli` は自分の隣にあるエンジンを見つけます。 +必要なのは [Python 3](https://www.python.org/downloads/) のインストールだけです。ランチャーと +API ゲートウェイは Python スクリプトですが、エンジン自体は依存ゼロの純粋な C です。 + +**またはソースからビルド** — OpenMP 対応の `gcc`(または clang)が必要です: + +```bash +git clone https://github.com/JustVugg/colibri && cd colibri/c +./setup.sh # gcc/OpenMP を確認し、ビルドとセルフテストを実行 +``` + +`coli` を PATH に置きたい場合は、チェックアウトから `pip install -e .` を実行すると登録されます +(エンジンは引き続き `c/` にあります — wheel ではなく、クローンからの editable インストールです)。 + +### 2. モデルを入手する + +変換済みの **GLM-5.2 int4** コンテナが Hugging Face にあります — **int8 MTP ヘッド** 付きの +**グループスケール(gs64)** ビルドを使ってください。サイズは約 **372 GB** なので、 +十分な容量のある、できれば高速なディスクに置いてください: + +**https://huggingface.co/mastouri/GLM-5.2-colibri-int4-g64-with-int8-mtp** + +> ⚠️ 古い行単位 int4 のミラー(`mateogrgic/…`、`jlnsrk/…`)ではなく、上記の **gs64** コンテナを +> 使ってください。それらは品質が約 9 ポイント劣ることが計測されており、 +> [#455](https://github.com/JustVugg/colibri/issues/455) で報告された当初の思考モードのループや +> 終わらない生成の根本原因でした。gs64 コンテナは制御された行単位の A/B で見られたそれらの問題を +> 解消しましたが、繰り返しや EOS 欠乏に対する汎用的なガードではありません。MTP ヘッドも +> **int4 ではなく int8** である必要があります +> (int4 → ドラフト受理率 0%、[#8](https://github.com/JustVugg/colibri/issues/8)): +> `ls -l /out-mtp-*` — int8(正しい)なら 3 ファイルで `3527131672 / 5366238584 / 1065950496`、 +> または単一の `out-mtp-00000.safetensors` で `9959321520` バイトです +> (推奨コンテナの現在のアップロードは 1 ファイルで配布されています: 中身は同じ +> int8 テンソルで、1 要素 1 バイトのものが 777 個です)。 + +あるいは FP8 のソースから自分で変換することもできます — 756 GB 全体を一度にディスクに置く必要のない、 +再開可能なコマンド 1 つで行えます: + +```bash +./coli convert --model /nvme/glm52_i4 # シャードごとにダウンロード+変換(python、初回のみ) +``` + + +#### その他の対応モデル + +GLM-5.2 がリファレンスモデルですが、同じストリーミング手法でさらに 6 つのファミリーが動作します。 +それぞれが **兄弟エンジン** です — C ファイル 1 つで独自のアーキテクチャを持ち、同じ +`coli chat` / `coli serve` / `coli web` フロントエンドを使います(ランチャーはモデルの +`config.json` からバイナリを選びます): + +> **それぞれに必要なもの。** これらは大きく異なり、2 つを並べて読んだ人が要件が矛盾していると +> 誤解したこともあります([#191](https://github.com/JustVugg/colibri/issues/191))。矛盾してはいません — +> 別々のモデルなのです。**どれも GPU は必要ありません。** +> +> | モデル | 重み用のディスク | RAM | GPU | +> |---|---|---|---| +> | **OLMoE** | 約 7 GB(int8 コンテナ) | 8 GB | 不要 | +> | **GLM-5.2/5.3** | 約 372 GB | 最低 16 GB、快適には 24 GB | 不要 | +> | **GLM-5.3-Flash** | 変換後 約 195 GB | 25 GB(int4 の重み 12 GB + エキスパートキャッシュ) | 不要 | +> | **Inkling** | 約 469 GB | int4 密コンテナ使用時 25 GB、未使用時 約 120 GB | 不要 | +> | **Kimi K3** | 約 1.6 TB | 32 GB 以上 | 不要 | +> | **DeepSeek V4 Flash** | 約 167 GB(REAP 150B: 約 85 GB) | 最低 16 GB、快適には 32 GB | オプション。GTX 10 シリーズ以降の任意の NVIDIA カード(Pascal/Turing は `CUDA_ARCH=portable-pre-ampere NO_TC=1` で、RTX 50 で最良)により、プリフィルが 5〜10 倍、デコードが約 2.5 倍高速化 | +> | **Qwen3.8-Flash-Next** | 約 185.5 GB(公式 FP8 チェックポイント) | 最低 16 GB、デフォルトのコンテキストで快適には 24 GB | 非対応。CPU のみ | +> | **Qwen3.6-35B-A3B** | 約 20 GB(int4-gs64 コンテナ) | 24 GB(RAM への完全常駐が必要) | オプション。CUDA VRAM エキスパートティアは 8 GB カード 2 枚で **1.44 -> 10.05 tok/s(7.0 倍)** を計測、出力は CPU とビット単位で同一 | +> +> GPU はあくまで速くするだけです。エキスパートはディスクからストリーミングされるため、速度は +> ディスクで決まります — 遅いドライブでは 1 秒あたり 1 トークン未満、高速なドライブでキャッシュが +> 温まっていれば 1 秒あたり数トークンを想定してください。 + +| ファミリー | 総数 / アクティブ | 重み | ビルド | ドキュメント | +|---|---|---|---|---| +| **GLM-5.2/5.3** | 744B / 40B | [`mastouri/…-int4-g64-with-int8-mtp`](https://huggingface.co/mastouri/GLM-5.2-colibri-int4-g64-with-int8-mtp)(372 GB) | `make -C c glm` | このページ | +| **Inkling**(Thinking Machines) | 975B / 41B | [`nbeerbower/Inkling-colibri-int4`](https://huggingface.co/nbeerbower/Inkling-colibri-int4)(469 GB) | `make -C c inkling` | [inkling.md](docs/inkling.md) | +| **GLM-5.3-Flash**(Z.ai) | 321B / 40B | [`zai-org/GLM-5.3-Flash`](https://huggingface.co/zai-org/GLM-5.3-Flash) — ルーティングエキスパートを **int4-gs64** に変換、密部分は BF16 のままで精度はロード時に選択。ビジョン対応 | `make -C c glm53` | [glm53-flash.md](docs/glm53-flash.md) | +| **Kimi K3**(Moonshot) | 2.8T / 104B | [`moonshotai/Kimi-K3`](https://huggingface.co/moonshotai/Kimi-K3) — オリジナルのチェックポイント、ルーティングエキスパートは **ネイティブ MXFP4** のまま | `make -C c kimi_k3` | [kimi_k3.md](docs/kimi_k3.md) | +| **DeepSeek V4 Flash** | 284B / 13B | 公式のシャード化チェックポイント — ルーティングエキスパートは **ネイティブ fp4**、密部分は fp8-e4m3 のまま。**REAP で枝刈りした 150B**([`puwaer/DeepSeek-V4-Flash-0731-reap-150b`](https://huggingface.co/puwaer/DeepSeek-V4-Flash-0731-reap-150b)、85 GB、256 個中 132 個のエキスパート)も同じエンジンで変換なしにロード可能 | `make -C c deepseek-v4` | [deepseek-v4.md](docs/deepseek-v4.md) | +| **DeepSeek V4.1 Flash** | 552B / 16B | 公式チェックポイント、**変換不要**: エキスパートはすでに fp4、密部分は fp8-e4m3。そのうち 203 GB は一度に数百バイトずつディスクから読まれる n-gram メモリで、ルーティングエキスパートのコストは GLM-5.2 の 12.7 GB に対して **1 トークンあたり 4.5 GB**。ビジョン、ツール呼び出し、DSpark ドラフトヘッドはすべて有効 | `make -C c deepseek_v41` | [deepseek-v41.md](docs/deepseek-v41.md) | +| **Qwen3.8-Flash-Next**(Alibaba) | 125B + 51B n-gram / 6B | [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) — オリジナルのチェックポイント。PLE はページング可能なまま、エキスパートは **ネイティブのブロック FP8** のまま | `make -C c qwen38`(CPU のみ) | [qwen38.md](docs/qwen38.md) | +| **Qwen3.6**(Alibaba) | 35B / 3B | [`Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64`](https://huggingface.co/Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64)(約 20 GB、**推奨**)— Gated Attention + Gated DeltaNet のハイブリッド | `make -C c qwen36`(VRAM エキスパートティアには `CUDA=1`) | [qwen36.md](docs/qwen36.md) | +| **OLMoE**(AI2) | 7B / 1B | `c/tools/convert_olmoe_merged.py` で変換 — **int8** コンテナ、約 7 GB | `make -C c olmoe` | — | + +Qwen3.6 には変換済みコンテナが 3 つあります: **int4-gs64**(推奨 — int8 のアンカーに対するコサイン類似度は +行単位と比べて 0.98777 → 0.99313、KL は 0.109 → 0.080 と計測されており、量子化誤差が約 44% 少ない)、 +A/B のベースラインとしての [int4 行単位](https://huggingface.co/Kreuzzelg/qwen36-35b-a3b-colibri-i4)、 +そして [KAT-Coder v2.5](https://huggingface.co/Kreuzzelg/kat-coder-v2.5-dev-colibri-i4-gs64) です。 +KAT-Coder は同じエンジンでそのまま動きます — アーキテクチャが同一のチェックポイントであれば、 +専用のコードパスなしで動作します。`CUDA=1` を使うと、VRAM エキスパートティアは +**8 GB カード 2 枚で 1.44 → 10.05 tok/s(7.0 倍)** を計測し、出力は CPU の経路とビット単位で同一でした。 + +Kimi K3 は変換不要です。QAT で学習された MXFP4 エキスパートはオリジナルの Hugging Face シャードから +直接ストリーミングされ、bf16 の密な重みセットはロード時に量子化されます。長いエージェントセッションでは、 +リカレント状態のチェックポイントをオプトインできます(RAM 上に `COLI_K3_CKPT=N` スロット、または +`COLI_K3_CKPT_DIR` でディスクに退避)。編集されたプロンプトやフォローアッププロンプトは、残っている +最も深いチェックポイントを復元して末尾だけを再プリフィルするため、会話全体を SSM レイヤーで +再生し直す必要がありません。Vulkan ホストでは `K3_VK_UP=auto` が、計測された帯域幅から +エキスパートティアのアップロード量を決めます。エンジンの KDA と MLA の経路は、CI でベンダー実装に +対してトークン単位で完全一致することが検証されています。 + +Inkling は int4 のエキスパートと **bf16 の密な重み**(常駐 49.4 GB)で提供されています。それを +保持できないホスト向けに、[inkling.md](docs/inkling.md) には密な重みセットを 15.3 GB にする +ワンパスのツールがあり、975B を 25 GB のマシンで動かせます — トレードオフも誠実に書かれています。 + +### 3. 実行する + +```bash +COLI_MODEL=/nvme/glm52_i4 ./coli chat # RAM 予算、キャッシュ、MTP を自動検出 +COLI_MODEL=/nvme/glm52_i4 ./coli plan # 計画された VRAM/RAM/ディスクの配置を確認 +COLI_MODEL=/nvme/glm52_i4 ./coli doctor # 読み取り専用の準備状況チェック +COLI_MODEL=/nvme/glm52_i4 ./coli doctor --deep # テンソル/シャード/インデックス/ミラーの厳密な事前チェック +COLI_MODEL=/nvme/glm52_i4 ./coli tune # このマシンで最速かつ安全な実行プロファイルを計測して保存 +./coli web --model /nvme/glm52_i4 # API + ダッシュボード、ブラウザを開く +./coli serve --model /nvme/glm52_i4 # API + ダッシュボード、ブラウザなし(ヘッドレス) +``` + +Windows ではリリースアーカイブに `coli.cmd` が同梱されています。ダブルクリックでクイックスタート、 +または cmd や PowerShell から `coli.cmd chat --model D:\glm52_i4` を実行してください。 +ソースのチェックアウトからは、同じコマンドを `python coli chat --model +D:\glm52_i4` として実行できます。`.exe` ファイルはエンジンでありランチャーではありません。 +単体で起動するとロードするモデルがないため、すぐに終了します。 +実行時のエンジンは純粋な C です — Python は初回のみの変換ツールとオプションの +API ゲートウェイでしか使われません。 + +#### 同じコマンドでどのモデルも動く + +`coli` はモデルの `config.json` を読み、対応するエンジンバイナリを選び、そのファミリーの +チャットテンプレートを適用します — そのため **モデルが変わってもコマンドラインは何も変わりません**。 +使いたいエンジンを一度ビルドしたら、あとは `COLI_MODEL` を適切なディレクトリに向けるだけです: + +```bash +make -C c glm # GLM-5.2 +make -C c inkling # Inkling +make -C c kimi_k3 # Kimi K3 + +COLI_MODEL=/nvme/glm52_i4 ./coli chat # TUI +COLI_MODEL=/nvme/inkling_i4 ./coli chat +COLI_MODEL=/nvme/kimi_k3 ./coli chat + +./coli web --model /nvme/inkling_i4 # API + ダッシュボード、ブラウザを開く +./coli web --model /nvme/kimi_k3 +./coli serve --model /nvme/inkling_i4 # API + ダッシュボード、ブラウザなし +``` + +GLM 以外のエンジンでは、`coli chat` がローカルでゲートウェイを起動して TUI をそれに接続します。 +そのため TUI、API、ダッシュボードはすべて同じアーキテクチャ対応のチャットテンプレートを通り、 +テンプレートを自分で指定する必要はありません。 + +モデルごとに異なる点が 2 つあり、どちらも各モデルのページに記載されています: + +- **RAM に余裕のないホストでの Inkling** には、int4 の密コンテナと小さなエキスパートキャッシュが + 必要です: `./coli chat --model /nvme/inkling_i4 --cap 2` + ([inkling.md](docs/inkling.md) を参照 — デフォルトの `--cap 8` では、常駐セットに加えて + 約 14 GB のキャッシュが必要です)。 +- **Kimi K3** は MXFP4 エキスパートをオリジナルのチェックポイントからストリーミングするため、 + 変換するものはありません — ただしスナップショットは約 1.6 TB です + ([kimi_k3.md](docs/kimi_k3.md) を参照)。 + +### 4. さらに詳しく + +| トピック | ドキュメント | +|---|---| +| ベンチマーク、コミュニティのデータポイント、品質計測 | [docs/benchmarks.md](docs/benchmarks.md) | +| 再現可能なベンチマークプロトコルと最低限のレポート | [docs/benchmarking.md](docs/benchmarking.md) | +| チューニング項目、ポリシー、学習キャッシュ、プリフェッチ | [docs/tuning.md](docs/tuning.md) | +| Windows 11 ネイティブビルド(+ CUDA DLL) | [docs/windows.md](docs/windows.md) | +| CUDA バックエンド、VRAM エキスパートティア、完全常駐 | [docs/cuda.md](docs/cuda.md) | +| Vulkan バックエンド(任意の GPU: RADV 経由の AMD、ROCm がサポートを終了したカードを含む) | [docs/vulkan.md](docs/vulkan.md) | +| Apple Silicon Metal バックエンド | [docs/metal.md](docs/metal.md) | +| OpenAI 互換 API、KV スロット、Web ダッシュボード | [docs/api.md](docs/api.md) | +| 実験的なレイヤーセグメント埋め込み ABI | [docs/segment-runtime.md](docs/segment-runtime.md) | +| 実験的なトークナイザ/埋め込み/ヘッドの Edge ABI | [docs/edge-runtime.md](docs/edge-runtime.md) | +| 文法強制ドラフト(構造化出力) | [docs/grammar-draft.md](docs/grammar-draft.md) | +| 環境変数一覧 | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) | + +## DeepSeek V4 + +**DeepSeek V4 Flash** は公式チェックポイントを変換なしでストリーミングします。ルーティング +エキスパートは **ネイティブ fp4** のまま、密な重みセットは UE8M0 ブロックスケール付きの +**fp8-e4m3** のままです。MLA + DSA スパースアテンション、43 レイヤー、256 個のルーティング +エキスパートと 1 個の共有エキスパート、top-6。x86-64/aarch64 の Linux と Windows/MSYS2(CPU)に +対応し、オプションの CUDA ティア(Windows ではランタイム DLL、Linux では `CUDA=1` による直接リンク、 +WSL2 で検証済み)は、すべてのステージを CPU 基準に保ちつつステージごとにフォールバックします。 + +```bash +cd c +make deepseek-v4 +python ./coli chat --model /path/to/DeepSeek-V4-Flash --ram 32 +# coli run / coli serve / coli web も可 +# Windows CUDA ティア: make cuda-dsv4-dll CUDA_ARCH=portable (RTX 50 では + make cuda-dsv4-dg-dll) +``` + +新しく追加された 2 つのオプトイン GPU レバーがあり、コミュニティによる計測値を求めています。 +どちらもデフォルトでオフで、未設定時はバイト単位で同一です。`DSV4_HYBRID=1` は、実行時に計測した +帯域幅を使って、VRAM ティアのミスを GPU のフィル分岐と CPU の分岐に振り分けます。 +`COLI_CUDA_MOE_DOUBLE=1`(`COLI_CUDA_MOE_BATCH=1` と併用)は、現在のレイヤーを計算している間に +次のレイヤーのエキスパートセット全体を 2 つ目の VRAM バンクにプリフェッチし、VRAM が足りなければ +単一バンクにフォールバックします。 +CUDA ティアは Pascal と Turing のカード(GTX 10 / RTX 20 シリーズ)でも動作するようになりました: +`CUDA_ARCH=portable-pre-ampere NO_TC=1` でビルドしてください。 + +グリーディデコードで KV スロットは 1 つです。ツール呼び出しは、V4 ネイティブのプロンプトと DSML の +呼び出しブロックで HTTP ゲートウェイを通じて接続されています。文法はサポートされていません。 +[エンジンごとの API マトリクス](docs/api.md#tool-calling-support) を参照してください。プレフィックス +チェックポイント(メモリ上 + ディスク上)により、システムプロンプトの初回プリフィル後は、 +エージェントセッションやフォローアップのターンが数秒で始まります。RTX 5080 + NVMe 2 台での計測: +3324 トークンのプリフィルが 90 秒、8.3k トークンの初回ターンが初回のみ約 4 分、以降の +セッション/ターンは 6〜9 秒、3k コンテキストでデコード約 1.6 tok/s — 詳細は +[docs/deepseek-v4.md](docs/deepseek-v4.md) を参照してください。 + +**RAM を与えてください。** 43 × 256 個のルーティングエキスパートはディスク上で約 137 GiB あり、 +1 トークンがそのうち 301 個に触れるため、エキスパートキャッシュのヒット率が tok/s を決めます — +`--ram` は最も価値のある単一の設定項目であり、変わるのは速度だけで、出力は決して変わりません。 + +**投機的ドラフトは存在しますが、オフです。** DSpark のマルコフドラフターと完全な MTP はどちらも +実装・検証済みです。ドラフトはフォワードパスを節約できますが、トークンを変えることは決してありません。 +受理されたトークンはすべてターゲット自身の argmax だからです。実際のマルチターンチャットで計測したところ、 +受理率は 15 個中 1 個と 24 個中 10 個で、このエンジンのリカレントなアテンション状態について +棄却されたサフィックスを再生するコストがドラフトによる節約を上回りました — 14 トークンの回答 1 つに +495 秒かかりました。そのため `V4_DRAFT` と `V4_MTP` はデフォルトで `0` とし、より高速なストレージで +再挑戦する人のために、コードは計測値とともに残してあります。 + +CUDA ティア(ビルド、DLL の選択、GPU の対応範囲)、環境変数リファレンス、性能値、 +チェックポイントの検証、生成された小さな独立オラクルについては +[docs/deepseek-v4.md](docs/deepseek-v4.md) を参照してください。 + +## 今後の展望 + +- **推論システムの研究こそがプロダクトです。** 現在の階層は LRU + 学習されたピンセットです。 + 進行中の作業は、モデルフォーマット、圧縮、配置、スケジューリング、I/O、CPU/GPU カーネル、 + 異種混在のオーバーラップ、KV 状態、ルーティングを考慮した投機的デコードにわたります。 + 目的はハードウェア要件と有用トークンあたりのコストを下げることです。すべてはこのプロジェクトの + やり方で取り込まれます: エンドツーエンドで計測され、レビューされ、オープンに開発されます。 +- **より多くのオープンモデル。** ティアリングアルゴリズムはモデルに依存しません。ルーティング + エキスパートを持つ MoE であれば、どれも同じ方法でステージングできます。現在 9 つのファミリーが + 動作しています(GLM-5.2、GLM-5.3-Flash、Inkling、Kimi K3、DeepSeek V4 Flash、DeepSeek V4.1 Flash、 + Qwen3.8-Flash-Next、Qwen3.6、OLMoE)。さらなるオープンウェイトのファミリー — 候補には + **MiniMax** も含まれます — は、最初の 8 つと同じ方法でエンジンを獲得します: + 誰かがエンドツーエンドで計測したときにです。 + +## プロジェクトへの支援 + +colibrì は、RAM 25 GB の 12 コアのラップトップ上で、1 人のプロジェクトとして始まりました。 +今ではその数値は、実機を持つコミュニティから集まっています。役に立ったと感じたら: + +- ⭐ リポジトリにスターを付けて共有してください。 +- 🐛 あなたのハードウェアでのベンチマーク数値を添えて issue を作成してください — データポイントは + 何よりもこのプロジェクトを前進させます。 +- 💬 [Discord コミュニティ](https://discord.gg/RXV83nSZdk) に参加して、実験、ハードウェアでの結果、 + 研究の方向性について議論してください。 +- 💬 開発のスポンサーやハードウェアの寄贈については、GitHub の issue からご連絡ください。 + +## リポジトリ構成 + +``` +Makefile ルートのビルド/チェックのエントリポイント +c/ +├── colibri.c GLM-5.2 エンジン (make glm) +├── inkling.c Inkling エンジン (make inkling) +├── kimi_k3.c Kimi K3 エンジン (make kimi_k3) +├── deepseek_v4.c DeepSeek V4 Flash エンジン (make deepseek-v4) +├── qwen38.c Qwen3.8-Flash-Next テキストエンジン (make qwen38) +├── qwen36.c Qwen3.6 エンジン (make qwen36) +├── olmoe.c OLMoE エンジン (make olmoe) +│ +├── st.h safetensors のインデックスと範囲読み込み +├── quant.h 正規のコンテナデコーダ +├── expert_ffn.h MoE エンジン共通のルーテッドエキスパート FFN カーネル(planar int4、レイヤーランナー) +├── tok.h, json.h トークナイザと JSON パーサ +├── compat.h Windows/macOS 用シム(POSIX 名を一か所に) +├── expert_store.h ストリーミングエキスパートキャッシュ +├── route_trace.h ルーティングのテレメトリと .coli_usage(エンジン非依存) +├── kv_prefix.h ターンをまたいだ KV プレフィックスの再利用 +│ +├── backend_cuda.* オプションの CUDA ティア (CUDA=1) +├── backend_metal.* オプションの Metal ティア (METAL=1) +├── backend_vulkan.* オプションの Vulkan ティア (VULKAN=1) +│ +├── Makefile ビルドとローカルチェック +├── coli ユーザー向け CLI +├── openai_server.py OpenAI 互換 HTTP ゲートウェイ +├── resource_plan.py `coli plan` と `coli doctor` の背後にある RAM/VRAM プランナー +├── tools/ オフライン変換、フィクスチャ、ベンチマーク +├── scripts/ 長時間実行の変換ヘルパー +└── tests/ 依存関係のない C と Python のテスト +web/ ブラウザ UI(純粋な OpenAI API クライアント) +desktop/ Web UI をラップする Tauri v2 デスクトップシェル +docker/ コンテナイメージ +docs/ リファレンスドキュメント、実験、メディア +``` + +**モデルファミリーごとに `.c` を 1 つ、共有の単一ヘッダの上に。** エンジンは自身のアーキテクチャだけを +持ち、それ以外は持ちません。2 つのエンジンが共に必要とするもの — safetensors リーダー、コンテナ +デコーダ、トークナイザ、エキスパートキャッシュ — は両者がインクルードするヘッダに置かれるため、 +修正は一度にすべてのエンジンに届きます。このルールは飾りではありません。ここで繰り返し発生する +不具合は、ある仕組みが 1 つのエンジンにだけ入り、兄弟エンジンに届かなかったケースなのです。 + +リポジトリのルートからは、`make`、`make check`、`make clean` がエンジンの Makefile に委譲されます。 + +## なぜ「colibrì」なのか + +ハチドリ(イタリア語で colibrì)は体重わずか数グラムで、空中にとどまり、1 日に千もの花を訪れます。 +このエンジンは、7,440 億パラメータの巨人をハチドリの食事量で生かし続けます: RAM 25 GB、 +CPU 12 コア、そしてディスクへのたっぷりの忍耐です。 + +## 謝辞 + +colibrì はエンジンにすぎず、それが動かす知性は贈り物です。フロンティア級の重みをオープンに +公開しているチーム — **Z.ai**(GLM)、**Moonshot AI**(Kimi)、**Alibaba Qwen**、**MiniMax**、 +**Allen AI**(OLMoE)— そして、ベンチマークを取り、バイセクトし、アトラスの実行を再現し、 +パッチを送ってくれたすべてのコントリビューターに感謝します。 +このプロジェクトは、オープンウェイトが何を可能にするかの証明です。 + +このプロジェクトのエキスパート配置、圧縮、ルーティングの実験は、以下のオープンな研究と +システムのアイデアやエビデンスにも基づいています: + +- 出力を考慮した、ドメイン固有のエキスパート重要度については + [REAP](https://github.com/CerebrasResearch/reap) と + [EASY-EP](https://github.com/RUCAIBox/EASYEP)。 +- 類似度に基づくエキスパートの再ルーティングについては [SERE](https://github.com/JL-Cheng/SERE)、 + キャッシュ局所性を考慮したルーターのファインチューニングについては + [ReMoE](https://github.com/BUAA-OSCAR/ReMoE)。 +- ルーティングに導かれたエキスパートのマージと圧縮については + [MC-SMoE](https://github.com/UNITES-Lab/MC-SMoE)。 +- 共有エキスパート基底と低ランクのエキスパート差分については + [MoBE](https://github.com/inclusionAI/MoBE) と + [D²-MoE](https://github.com/lliai/D2MoE)。 +- CPU/GPU ハイブリッドのエキスパートスケジューリングについては + [HybriMoE](https://github.com/PKU-SEC-Lab/HybriMoE)、エキスパート通信と計算のオーバーラップについては + [ScMoE](https://arxiv.org/abs/2404.05019)、分散オンデマンドのエキスパートロードについては + [OD-MoE](https://arxiv.org/abs/2512.03927)。 +- 比較を再現可能にしているオープンな推論システムとエキスパートオフロードの取り組みについては + [vLLM](https://github.com/vllm-project/vllm)、 + [llama.cpp](https://github.com/ggml-org/llama.cpp)、 + [kTransformers](https://github.com/kvcache-ai/ktransformers)。 + +エンジンはアイデアだけでなく、具体的なエンジニアリングの成果の上にも成り立っています。以下はいずれも +現在ツリー内で使われているか、再実装されています: + +- [safetensors](https://github.com/huggingface/safetensors) — すべてのエンジンが読むコンテナ + (`c/st.h`)。fp8 と I64 の dtype を含みます。 +- [tiktoken](https://github.com/openai/tiktoken) — `c/tok.h` はその `byte_pair_encode` を + 正確に再実装しており、連結した結果の語彙 ID が最も小さい隣接ペアをマージするため、 + tiktoken 由来の語彙にはマージリストが不要です。 +- [llama.cpp](https://github.com/ggml-org/llama.cpp) — `c/grammar.h` の GBNF 文法サブセットは + その構文とスタック集合による PDA に従っており、Metal の経路はその + `newBufferWithBytesNoCopy` による常駐テクニックを借用しています。 +- [vLLM](https://github.com/vllm-project/vllm) — エンジンが位置ごとに一致させている出力 + セマンティクスのリファレンス(例: 最終ノルムが LM ヘッドに対してどこに入るか)。 +- [transformers](https://github.com/huggingface/transformers) — オラクル: + CI はランダム初期化モデルをこれに対してトークン単位で再現します。 +- [DietGPU](https://github.com/facebookresearch/dietgpu) — 実験的な圧縮エキスパートティア + (`COLI_ANS`)の背後にある GPU ANS コーデック。 +- [rocWMMA](https://github.com/ROCm/rocWMMA) — HIP バックエンドは CUDA の + `nvcuda::wmma` の fragment/mma_sync API をこれにマッピングしており(`c/backend_gpu_compat.h`)、 + それによって 1 つの .cu ソースを両ベンダー向けにコンパイルできます。 + +## ライセンス + +Apache 2.0。GLM-5.2 の重みは Z.ai により MIT ライセンスで公開されています。 diff --git a/README.md b/README.md index f23129938..3a0ad7d9e 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@

Website · Discord · - English · 简体中文 · 繁體中文 · Italiano + English · 简体中文 · 繁體中文 · Italiano · 日本語

**Tiny engine, immense model.** Run **frontier MoE models — 744B to 2.8T @@ -40,7 +40,7 @@ may reduce speed; it must not quietly redefine the model. ``` $ ./coli chat - 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU + 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU ✓ ready in 32s · resident 9.9 GB › ciao! ◆ Ciao! 😊 Come posso aiutarti oggi? @@ -135,7 +135,7 @@ shows otherwise. These are the main questions now: | hypothesis | evidence so far | experiment still needed | |---|---|---| | Routing history can place experts better than plain LRU | learned pins improve repeated workloads, but can overfit a prompt | held-out, cross-session A/Bs across coding, chat, multilingual, and long-context workloads | -| Multiple SSDs can turn independent bandwidth into decode speed | weighted mirror/split routing is implemented and validated; the bandwidth model is sound | cold-cache one-drive vs two-drive GLM-5.2 runs on real, independent controllers | +| Multiple SSDs can turn independent bandwidth into decode speed | two independent NVMe drives measured +37.5% decode; a slower third drive was neutral after weighted striping ([measurements](docs/multidisk.md#what-has-been-measured)) | reproduce across drive speeds, controller layouts, and cache states | | A hardware-aware planner can approach each machine's best configuration automatically | RAM/VRAM budgets and several backends are detected today | compare the generated plan with a controlled parameter sweep across laptops, workstations, NUMA hosts, and multi-GPU systems | | Lossless or quality-bounded representations can reduce weight movement enough to matter | format and quantization ablations exist, with correctness/quality gates | reproduce quality, bytes moved, latency, and cost per useful token together — not compression ratio alone | | Routing-aware speculation can pay before near-full residency | MTP and grammar drafts work, but MTP has also measured a 32% loss around 85% expert hit | map the break-even surface across acceptance, expert hit rate, batch union, and draft depth | @@ -232,21 +232,30 @@ precision are the same whether an expert answered from VRAM or from disk. VRAM / RAM / NVMe three-tier expert residency

-### Dual-SSD: two copies of the model, twice the read bandwidth + -Decode is disk-bound on most machines, and expert reads are read-only — so if you have a **second SSD**, put a full copy of the model on it and let the engine stream from both drives at once: +### Multiple SSDs: stream model copies from more than one drive + +When decode is disk-bound, a **second SSD** can help: put a copy of the model on +it and let the engine read from both drives. For GLM-5.2, from `c/` in a source +checkout (or from an unpacked release): ```bash -COLI_MODEL=/fast/glm52_i4 COLI_MODEL_MIRROR=/second/glm52_i4 ./coli chat -COLI_DISK_WEIGHTS=9,3 ... # optional: primary,mirror bandwidth ratio (else measured at startup) +COLI_MODEL_MIRROR=/second/glm52_i4 python3 ./coli chat --model /fast/glm52_i4 ``` -Each expert is routed to one drive by a deterministic hash, weighted by the two drives' measured (or declared) bandwidth, so readahead/PILOT prefetch and the demand read always hit the same drive and nothing is cached twice. The aggregate bandwidth is the sum of both drives — a 9 GB/s + 3 GB/s pair reads experts ~33% faster than the fast drive alone, and the OMP-parallel pin/warmup load streams from both. Details worth knowing: +The engine measures the drives at startup to weight the read split. Buffered +reads use deterministic expert routing; eligible direct reads can stripe one +expert across replicas. Independent drives provide bandwidth headroom, not a +guaranteed token-rate multiplier: shared controllers, cache hits, and compute +can limit the gain. See the [multi-disk guide](docs/multidisk.md) for Bash and +PowerShell examples, measured gains and limits, and a single-drive comparison. +Details worth knowing: -- the mirror is **validated at startup** (per-file size + safetensors header must be byte-identical to the primary); divergent or missing files silently stay on the primary, so a **partial mirror is fine** — a smaller second SSD holding only some shards still helps; +- the mirror is **validated at startup** (per-file size + safetensors header must be byte-identical to the primary); divergent or missing files stay on the primary, so a **partial mirror is fine** — a smaller second SSD can serve the shards it holds; - the mirror is **never written**: `.coli_usage`, `.coli_kv` and all sidecars stay on the primary; - a read error on the mirror falls back to the primary (one warning, no crash), so unplugging the second drive mid-run degrades instead of killing the server; -- routing never changes tokens — both copies are byte-identical, and the per-run `MIRROR:` stats line shows GB served per drive. +- routing never changes tokens — both copies are byte-identical; enable `PROF=1` for the `MIRROR:` profile counters showing GB served per drive. The same engine spans the whole range: on a 25 GB laptop everything streams from disk (slow but correct); on a large host the entire expert set becomes resident @@ -325,6 +334,15 @@ full forensic story). Grammar-forced drafts constrained JSON output. Whether speculation is a net win depends on your cache temperature — measure, and use `DRAFT=0` when it doesn't pay. +Verify batches can also opt into an **exact attention core** with +`COLI_EXACT_VERIFY=1` ([#689](https://github.com/JustVugg/colibri/issues/689)): +the CPU MLA-absorb score and context dots accumulate integer products and round +once, so a near-tie in a verify row resolves the same way on every host, at +roughly 0.6x tok/s on a tiny oracle (the dot itself is ~5–7x the float loop). +Two limits to know: with a quantised KV cache (`tq1`, TQ or int8 KV) the +context dot keeps the float path, so exactness there is not provided; and a +real near-tie flip has only been argued, not yet caught on GLM-5.2 at n=64. + ## What it achieves

@@ -434,7 +452,7 @@ the model's `config.json`): > | **Inkling** | ~469 GB | 25 GB with the int4 dense container, ~120 GB without | not needed | > | **Kimi K3** | ~1.6 TB | 32 GB+ | not needed | > | **DeepSeek V4 Flash** | ~167 GB (REAP 150B: ~85 GB) | 16 GB min, 32 GB comfortable | optional; any NVIDIA card from the GTX 10 series up (Pascal/Turing via `CUDA_ARCH=portable-pre-ampere NO_TC=1`, best on RTX 50) makes prefill 5-10x and decode ~2.5x faster | -> | **Qwen3.8-Flash-Next** | ~185.5 GB (official FP8 checkpoint) | 16 GB min, 24 GB comfortable at the default context | not supported; CPU only | +> | **Qwen3.8-Flash-Next** | ~185.5 GB (official FP8 checkpoint) | 16 GB min, 24 GB comfortable at the default context | optional; the CUDA VRAM expert tier with dense trunk quantized to int8 in VRAM | > | **Qwen3.6-35B-A3B** | ~20 GB (int4-gs64 container) | 24 GB (needs full RAM residency) | optional; the CUDA VRAM expert tier measured **1.44 -> 10.05 tok/s (7.0x)** on two 8 GB cards, output bit-identical to CPU | > > A GPU only ever makes it faster. Speed is set by your disk, because the experts @@ -449,7 +467,7 @@ the model's `config.json`): | **Kimi K3** (Moonshot) | 2.8T / 104B | [`moonshotai/Kimi-K3`](https://huggingface.co/moonshotai/Kimi-K3) — original checkpoint, routed experts stay **native MXFP4** | `make -C c kimi_k3` | [kimi_k3.md](docs/kimi_k3.md) | | **DeepSeek V4 Flash** | 284B / 13B | official sharded checkpoint — routed experts stay **native fp4**, dense stays fp8-e4m3; the **REAP-pruned 150B** ([`puwaer/DeepSeek-V4-Flash-0731-reap-150b`](https://huggingface.co/puwaer/DeepSeek-V4-Flash-0731-reap-150b), 85 GB, 132 of 256 experts) loads with the same engine and no conversion | `make -C c deepseek-v4` | [deepseek-v4.md](docs/deepseek-v4.md) | | **DeepSeek V4.1 Flash** | 552B / 16B | official checkpoint, **no conversion**: experts are already fp4, dense is fp8-e4m3. 203 GB of it is an n-gram memory read from disk a few hundred bytes at a time, and the routed experts cost **4.5 GB per token** against GLM-5.2's 12.7. Vision, tool calling and the DSpark draft head are all on | `make -C c deepseek_v41` | [deepseek-v41.md](docs/deepseek-v41.md) | -| **Qwen3.8-Flash-Next** (Alibaba) | 125B + 51B n-gram / 6B | [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) — original checkpoint; PLE stays pageable and experts stay **native block-FP8** | `make -C c qwen38` (CPU only) | [qwen38.md](docs/qwen38.md) | +| **Qwen3.8-Flash-Next** (Alibaba) | 125B + 51B n-gram / 6B | [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) — original checkpoint; PLE stays pageable and experts stay **native block-FP8** | `make -C c qwen38` (`CUDA=1` for the VRAM expert tier) | [qwen38.md](docs/qwen38.md) | | **Qwen3.6** (Alibaba) | 35B / 3B | [`Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64`](https://huggingface.co/Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64) (~20 GB, **recommended**) — hybrid Gated Attention + Gated DeltaNet | `make -C c qwen36` (`CUDA=1` for the VRAM expert tier) | [qwen36.md](docs/qwen36.md) | | **OLMoE** (AI2) | 7B / 1B | converted with `c/tools/convert_olmoe_merged.py` — **int8** container, ~7 GB | `make -C c olmoe` | — | diff --git a/README.zh-CN.md b/README.zh-CN.md index c15a7ea3d..7f9a9420d 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -4,7 +4,7 @@

Discord · - English · 简体中文 · 繁體中文 · Italiano + English · 简体中文 · 繁體中文 · Italiano · 日本語

**小巧引擎,庞大模型。**在消费级与异构硬件上运行**前沿 MoE 模型——从 744B 到 @@ -25,7 +25,7 @@ Colibrì 刻意用于验证激进的系统思路——因此**对速度不作 SL ``` $ ./coli chat - 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU + 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU ✓ ready in 32s · resident 9.9 GB › ciao! ◆ Ciao! 😊 Come posso aiutarti oggi? diff --git a/README.zh-TW.md b/README.zh-TW.md index adfeb43ae..c208ac413 100644 --- a/README.zh-TW.md +++ b/README.zh-TW.md @@ -4,7 +4,7 @@

Discord · - English · 简体中文 · 繁體中文 · Italiano + English · 简体中文 · 繁體中文 · Italiano · 日本語

**小巧引擎,龐大模型。**在消費級與異質硬體上執行**前沿 MoE 模型——從 744B 到 @@ -25,7 +25,7 @@ Colibrì 刻意用於驗證激進的系統構想——因此**對速度不作 SL ``` $ ./coli chat - 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU + 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU ✓ ready in 32s · resident 9.9 GB › ciao! ◆ Ciao! 😊 Come posso aiutarti oggi? diff --git a/c/.gitignore b/c/.gitignore index 36d9678ca..ef2a31d61 100644 --- a/c/.gitignore +++ b/c/.gitignore @@ -21,6 +21,7 @@ tests/fuzz_rans tests/bench_omp_grain tests/mxfp4_ref.o mxfp4_cuda_test +mxfp4_expert_cuda_test absorb_determinism_test cuda_fmt_trap_test tests/*.dSYM/ diff --git a/c/Makefile b/c/Makefile index 2fe13efd5..58a1d00a5 100644 --- a/c/Makefile +++ b/c/Makefile @@ -491,16 +491,36 @@ endif TEST_RULES := $(shell sed -n 's|^tests/\(test_[a-z0-9_]*\)\$$(EXE):.*|\1|p' $(firstword $(MAKEFILE_LIST))) # test_uring is Linux-only. V4 engine tests are appended below only on supported # x86-64 Linux/Windows and aarch64 Linux hosts; the V4 infrastructure tests have -# unconditional rules and therefore remain portable gates. +# unconditional rules and therefore remain portable gates. The six forced +# SSE4.1 tests use -msse4.1/-mno-avx2/-mno-fma, which only exist as gcc/clang +# flags on x86 -- unconditionally in TEST_BINS they take down `check` on +# arm64 (e.g. macOS): "unsupported option '-msse4.1' for target arm64-...". +# Appended back below only on x86-64 hosts, alongside the other conditional +# platform tests already handled the same way. +# test_e8x4g64_loader takes a minted container directory on argv and prints its +# usage (exit 2) without one, so it is a harness rather than a gate: it has a +# rule here so it compiles under the suite's own $(CFLAGS) and its warnings are +# caught -- reachable via `make e8x4g64-loader-check`, kept out of `check` itself +# since a bare invocation exits 2 -- and tests/test_e8x4g64_mint_load.py is what +# actually drives it end to end. # test_qwen38_tier_engine drives qwen38.c over the generated FP8 fixture # (qwen38_tiny_fp8, gitignored); it runs from qwen38-tier-engine-check. TEST_EXCLUDE = test_uring test_deepseek_v4 test_v4_ownership test_v4_serve_framing \ test_segment_adapters_registration test_segment_adapters_real \ test_edge_adapters_registration test_edge_adapters_real \ - test_qwen38_tier_engine + test_gsgemv_sse41 test_qgemv_sse41 test_olmoe_dot_i8_16_sse41 test_st_f16_bf16_simd_sse41 test_qwen38_tier_engine \ + test_e8x4g64_loader \ + test_i4_grouped_sse41 test_i4_grouped_sse41_o1 test_i4_grouped_sse41_no_contract TEST_BINS = $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(TEST_EXCLUDE),$(TEST_RULES)))) ifneq (,$(LINUX)) TEST_BINS += tests/test_uring$(EXE) +TEST_BINS += tests/test_exact_dot$(EXE) +endif +ifneq (,$(X86_64)) +TEST_BINS += tests/test_gsgemv_sse41$(EXE) tests/test_qgemv_sse41$(EXE) \ + tests/test_olmoe_dot_i8_16_sse41$(EXE) tests/test_st_f16_bf16_simd_sse41$(EXE) \ + tests/test_i4_grouped_sse41$(EXE) \ + tests/test_i4_grouped_sse41_o1$(EXE) tests/test_i4_grouped_sse41_no_contract$(EXE) endif ifeq ($(COLI_V4_SUPPORTED),1) TEST_BINS += tests/test_v4_hybrid_policy$(EXE) tests/test_k3_fill_budget$(EXE) tests/test_v4_bank_pair$(EXE) @@ -700,6 +720,9 @@ deepseek-v4-tiny-check: deepseek-v4-tiny-generate $(PYTHON) tests/test_deepseek_v4_prefix.py \ --binary $(CURDIR)/$(if $(IS_WIN),deepseek_v4.exe,deepseek_v4) \ --fixture $(CURDIR)/deepseek_v4_tiny + $(PYTHON) tests/test_deepseek_v4_brio.py \ + --binary $(CURDIR)/$(if $(IS_WIN),deepseek_v4.exe,deepseek_v4) \ + --fixture $(CURDIR)/deepseek_v4_tiny deepseek-v4-oracle: deepseek-v4 @test -n "$(MODEL)" || { echo "usage: make deepseek-v4-oracle MODEL=/path/to/checkpoint" >&2; exit 2; } @@ -749,11 +772,20 @@ ifneq "$(BUILD_CONFIG)" "$(BUILD_CONFIG_OLD)" # cmd.exe, so `make glm.exe` from the VS Native Tools prompt or with scoop MinGW # (neither ships sh.exe) failed with "'printf' is not recognized" and left the # stamp stale. $(file ...) works regardless of the shell make falls back to. (#478) +# +# GNU Make 3.81 -- /usr/bin/make on macOS -- has no $(file ...): it expands to +# nothing, the stamp is never written, and every build relinks (#1732). Make +# 3.x is found there and not under cmd.exe (MSYS2 and MinGW ship 4.x), and +# macOS always has a POSIX printf, so that one case keeps the shell write. +ifneq ($(filter 3.%,$(MAKE_VERSION)),) +$(shell printf '%s\n' '$(subst ','\'',$(BUILD_CONFIG))' > .build-config) +else $(file >.build-config,$(BUILD_CONFIG)) endif +endif .build-config: ; -colibri$(EXE): colibri.c pin_pool.h cli_args.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h sample.h kv_persist.h telemetry.h route_trace.h omp_tune.h kv_fp8.h kv_tq.h abl.h backend_cuda.h backend_metal.h backend_vulkan.h decode_batch.h edge_adapters.h edge_runtime.h edge_tok_internal.h schema_gbnf.h segment_adapter_internal.h segment_adapters.h segment_runtime.h tier.h $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) .build-config +colibri$(EXE): colibri.c sse41_kernels.h exact_dot.h oracle.h pin_pool.h cli_args.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h omp_tune.h kv_fp8.h kv_tq.h abl.h backend_cuda.h backend_metal.h backend_vulkan.h decode_batch.h edge_adapters.h edge_runtime.h edge_tok_internal.h schema_gbnf.h segment_adapter_internal.h segment_adapters.h segment_runtime.h tier.h $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) .build-config $(CC) $(CFLAGS) colibri.c $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) -o colibri$(EXE) $(LDFLAGS) # Vulkan backend object (plain C + vulkan headers) and its SPIR-V shaders. @@ -772,7 +804,7 @@ backend_loader.o: backend_loader.c backend_cuda.h compat.h .build-config # shell that has the MSVC environment set (e.g. after vcvars64.bat, or from a # "x64 Native Tools Command Prompt"). COLI_CUDA_BUILDING_DLL enables # __declspec(dllexport) so the 15 API symbols are exported. -cuda-dll: backend_cuda.cu backend_cuda.h +cuda-dll: backend_cuda.cu backend_cuda.h fp8_format.h @command -v "$(NVCC)" >/dev/null 2>&1 || { echo "nvcc not found: set CUDA_HOME or NVCC" >&2; exit 1; } @command -v cl >/dev/null 2>&1 || { echo "cl.exe (MSVC) not in PATH — run vcvars64.bat first" >&2; exit 1; } # The banner is localized ("for x64" / "per x64" / "pour x64"): match the arch token only (#1531). @@ -802,7 +834,7 @@ cuda-dll: backend_cuda.cu backend_cuda.h # survives. No path is hardcoded — see the HIP SDK contract above. # # make hip-dll HIP_DLL=1 HIP_SDK_ROOT= HIP_ARCH=gfx1151 -hip-dll: backend_cuda.cu backend_cuda.h backend_gpu_compat.h +hip-dll: backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h @test -x "$(HIPCC)" || command -v "$(HIPCC)" >/dev/null 2>&1 || { echo "hipcc not found at \"$(HIPCC)\": set HIP_SDK_ROOT=, HIP_BIN_DIR= or HIPCC=" >&2; exit 1; } @test -d "$(HIP_INCLUDE_DIR)" || { echo "HIP include dir not found: \"$(HIP_INCLUDE_DIR)\" — set HIP_INCLUDE_DIR=" >&2; exit 1; } @test -f "$(HIP_INCLUDE_DIR)/hip/hip_runtime.h" || { echo "hip/hip_runtime.h missing under \"$(HIP_INCLUDE_DIR)\" — set HIP_INCLUDE_DIR=" >&2; exit 1; } @@ -821,7 +853,7 @@ backend_cuda_ink.o: backend_cuda_ink.cu backend_cuda_ink.h .build-config @command -v "$(NVCC)" >/dev/null 2>&1 || { echo "nvcc not found: set CUDA_HOME or NVCC" >&2; exit 1; } "$(NVCC)" $(NVCCFLAGS) -c backend_cuda_ink.cu -o $@ -backend_cuda.o: backend_cuda.cu backend_cuda.h backend_gpu_compat.h .build-config +backend_cuda.o: backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h .build-config @command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; } "$(GPUCC)" $(GPUFLAGS) -c backend_cuda.cu -o $@ @@ -944,10 +976,12 @@ rans: $(RANSLIB) $(RANSLIB): tools/rans_ctypes.c rans.h $(CC) $(CFLAGS) -fPIC -shared $< -o $@ $(LDFLAGS) -cuda-test: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu +cuda-test: tests/test_mxfp4_expert_cuda.cu backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu tests/test_cuda_init_failure.cu @command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; } "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_backend_cuda.cu -o backend_cuda_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS) ./backend_cuda_test$(EXE) + "$(GPUCC)" $(GPUFLAGS) tests/test_cuda_init_failure.cu -o cuda_init_failure_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS) + ./cuda_init_failure_test$(EXE) "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_ragged_attention.cu -o ragged_attention_test$(EXE) $(GPU_TEST_LIBS) ./ragged_attention_test$(EXE) # weight_at's device-side refusal (the __trap()) on real silicon. The @@ -1003,9 +1037,16 @@ cuda-test: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/test_backen # Getting that wrong would leave OpenMP symbols in the object while the link # below no longer pulls the runtime, and CI could not catch it because CI # never runs cuda-test. -Xclang appears nowhere else in CFLAGS. - $(CC) $(filter-out -Xclang -fopenmp,$(CFLAGS)) -fno-openmp -c tests/mxfp4_ref.c -o tests/mxfp4_ref.o - "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.o -o mxfp4_cuda_test$(EXE) $(GPU_TEST_LIBS) + $(CC) $(filter-out -Xclang -fopenmp,$(CFLAGS)) -fno-openmp -fPIC -c tests/mxfp4_ref.c -o tests/mxfp4_ref.o + # -x none only under HIP: hip-test is `$(MAKE) cuda-test HIP=1`, so this + # line also runs under nvcc for CUDA users, and nvcc accepts only c, c++ + # and cu for -x -- it fails the invocation on anything else. CI never + # executes cuda-test, so that breakage would surface only on the next + # CUDA box to run the suite. + "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_mxfp4_cuda.cu $(if $(filter 1,$(HIP)),-x none) tests/mxfp4_ref.o -o mxfp4_cuda_test$(EXE) $(GPU_TEST_LIBS) ./mxfp4_cuda_test$(EXE) + "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_mxfp4_expert_cuda.cu $(if $(filter 1,$(HIP)),-x none) tests/mxfp4_ref.o -o mxfp4_expert_cuda_test$(EXE) $(GPU_TEST_LIBS) + ./mxfp4_expert_cuda_test$(EXE) # Allocator footprint (#687): what a cudaMalloc really takes off the card, # which is what the expert tier has to be charged rather than the logical # byte count. Runs LAST on purpose - it is the only test here that @@ -1046,7 +1087,7 @@ gpu-compile: backend_cuda.o "$(GPUCC)" $(GPUFLAGS) tests/test_weights_owned_cuda.cu -o weights_owned_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS) "$(GPUCC)" $(GPUFLAGS) tests/test_cuda_fmt_trap_cuda.cu -o cuda_fmt_trap_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS) -cuda-bench: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/bench_tensor_core.cu +cuda-bench: backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/bench_tensor_core.cu @command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; } "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/bench_tensor_core.cu -o backend_cuda_bench$(EXE) $(ANS_NVCC_LIBS) ./backend_cuda_bench$(EXE) @@ -1054,19 +1095,23 @@ cuda-bench: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/bench_tens # fmt=8 kernel bench: old vs COLI_CUDA_F8_WARP kernels per decode path, census # expert shapes, JSON on stdout (kernel time only). The build is a separate # file target so tools/run_f8_bench.sh can keep stdout pure JSON. -fp8_bench$(EXE): backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/bench_fp8_cuda.cu +fp8_bench$(EXE): backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/bench_fp8_cuda.cu @command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; } "$(GPUCC)" $(GPUFLAGS) tests/bench_fp8_cuda.cu -o fp8_bench$(EXE) $(ANS_NVCC_LIBS) fp8-bench: fp8_bench$(EXE) ./fp8_bench$(EXE) +# Matched resident int8 GPU projection timings; run explicitly, not in CI. +tests/bench_cuda_resident_batch$(EXE): tests/bench_cuda_resident_batch.cu backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h + "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/bench_cuda_resident_batch.cu -o $@ $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS) + # NOCUDA_*: olmoe.c has no COLI_CUDA code at all, so CUDA=1 would otherwise # hand it a define matching nothing and link a runtime it never calls -- a # build whose compile line, libraries and exit status all claim "CUDA" while # the GPU sits idle. Same guard #783 put on kimi_k3, which since gaining an # MXFP4 expert path no longer needs it. tests/test_makefile_cuda_scope.py # asserts this shape. -olmoe$(EXE): olmoe.c cli_args.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h kv_prefix.h pin_pool.h route_trace.h serve_codec.h edge_adapters.h edge_runtime.h edge_tok_internal.h fused_simd.h segment_adapter_internal.h segment_adapters.h segment_runtime.h +olmoe$(EXE): olmoe.c cli_args.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h kv_prefix.h pin_pool.h route_trace.h serve_codec.h serve_budget.h edge_adapters.h edge_runtime.h edge_tok_internal.h fused_simd.h segment_adapter_internal.h segment_adapters.h segment_runtime.h sse41_kernels.h $(CC) $(NOCUDA_CFLAGS) olmoe.c -o olmoe$(EXE) $(NOCUDA_LDFLAGS) # Qwen3.6-35B-A3B engine (hybrid Gated Attention + Gated DeltaNet + streaming @@ -1092,13 +1137,21 @@ QWEN36_TIER_SRC = QWEN36_CFLAGS = $(NOCUDA_CFLAGS) QWEN36_LDFLAGS = $(NOCUDA_LDFLAGS) endif -qwen36$(EXE): qwen36.c decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +# On Windows, EXE=.exe. Keep a bare qwen36 target so GNU make does not +# fall through to its implicit %: %.c rule and compile qwen36.c alone. +ifneq ($(EXE),) +.PHONY: qwen36 +qwen36: qwen36$(EXE) +endif +# Rebuild when CUDA_DLL changes; otherwise an existing CPU-only executable can +# be reported as up to date despite selecting the Windows CUDA DLL tier. +qwen36$(EXE): qwen36.c sse41_kernels.h decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) .build-config $(CC) $(QWEN36_CFLAGS) qwen36.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen36$(EXE) $(QWEN36_LDFLAGS) # DeepSeek V4.1 Flash: one file, like every other portable engine. The fp4 experts # stream from the official checkpoint and quant.h's mxfp4 kernel reads them as they # are, so there is no conversion target here to go with it. -deepseek_v41$(EXE): deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h \ +deepseek_v41$(EXE): deepseek_v41.c sse41_kernels.h cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h fp8_format.h idot.h \ sparse_attn.h hyper_connections.h serve_codec.h serve_poll.h .build-config $(CC) $(CFLAGS) deepseek_v41.c -o deepseek_v41$(EXE) $(LDFLAGS) @@ -1106,7 +1159,7 @@ deepseek_v41$(EXE): deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h co # loader and SERVE=1 protocol. With CUDA=1 it links the same expert tier as # qwen36 (fp8 streaming mode: hot experts get VRAM copies, the RAM LRU stays); # without it the tier header's inline stubs keep the build toolkit-free. -qwen38$(EXE): qwen38.c pin_pool.h cli_args.h qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h omp_tune.h quant.h route_trace.h tok.h tok_unicode.h tok_unicode_o200k.h serve_codec.h edge_adapter_internal.h edge_adapters.h edge_runtime.h qwen38_vision.h segment_adapter_internal.h segment_adapters.h segment_runtime.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +qwen38$(EXE): qwen38.c sse41_kernels.h pin_pool.h cli_args.h qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h omp_tune.h quant.h fp8_format.h idot.h route_trace.h tok.h tok_unicode.h tok_unicode_o200k.h serve_codec.h edge_adapter_internal.h edge_adapters.h edge_runtime.h qwen38_vision.h segment_adapter_internal.h segment_adapters.h segment_runtime.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) qwen38.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen38$(EXE) $(QWEN36_LDFLAGS) .PHONY: qwen38-tiny-generate qwen38-tiny-check @@ -1127,9 +1180,13 @@ qwen38-ple-prefetch-check: qwen38-tiny-generate qwen38$(EXE) done; done; done; \ [ $$fail -eq 0 ] && echo "PLE prefetch: identical tokens on and off, 16 configurations" || exit 1 +# Q38_TRUNK_MIN_KB=0 puts every dense matrix of the fixture (all under the +# default 1 MiB threshold) on the int8 trunk with integer dot products, the +# path the released checkpoint takes; 1024 leaves them BF16. qwen38-tiny-check: qwen38-tiny-generate qwen38$(EXE) - @for batch in 0 1; do for bf16 in 0 1; do for cap in 1 4; do \ - Q38_PREFILL_BATCH=$$batch Q38_NATIVE_BF16=$$bf16 OMP_NUM_THREADS=2 SNAP=./qwen38_tiny ./qwen38$(EXE) $$cap 8 ./qwen38_tiny/ref.json || exit $$?; \ + @for trunk in 1024 0; do for batch in 0 1; do for bf16 in 0 1; do for cap in 1 4; do \ + Q38_TRUNK_MIN_KB=$$trunk Q38_PREFILL_BATCH=$$batch Q38_NATIVE_BF16=$$bf16 OMP_NUM_THREADS=2 SNAP=./qwen38_tiny ./qwen38$(EXE) $$cap 8 ./qwen38_tiny/ref.json || exit $$?; \ + done; \ done; \ done; \ done @@ -1143,8 +1200,9 @@ qwen38-tiny-fp8-generate: $(PYTHON) tools/make_qwen38_tiny.py --out ./qwen38_tiny_fp8 --fp8-experts qwen38-tiny-fp8-check: qwen38-tiny-fp8-generate qwen38$(EXE) - @for native in 1 0; do for cap in 1 2; do \ - Q38_NATIVE_FP8=$$native OMP_NUM_THREADS=2 SNAP=./qwen38_tiny_fp8 ./qwen38$(EXE) $$cap 8 ./qwen38_tiny_fp8/ref.json || exit $$?; \ + @for trunk in 1024 0; do for native in 1 0; do for cap in 1 2; do \ + Q38_TRUNK_MIN_KB=$$trunk Q38_NATIVE_FP8=$$native OMP_NUM_THREADS=2 SNAP=./qwen38_tiny_fp8 ./qwen38$(EXE) $$cap 8 ./qwen38_tiny_fp8/ref.json || exit $$?; \ + done; \ done; \ done @@ -1156,30 +1214,51 @@ qwen38-tier-engine-check: qwen38-tiny-fp8-generate tests/test_qwen38_tier_engine # Same tier sources as the engine: the test includes qwen36.c, so with CUDA=1 # it needs qwen36_tier.c and the backend object too (without CUDA, the header's # inline stubs cover it and both are empty). -tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) + $(CC) $(CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(LDFLAGS) + +# CACHE_ROUTE: route_select() against a residency table, no model needed. +tests/test_qwen36_cache_route$(EXE): tests/test_qwen36_cache_route.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(LDFLAGS) -tests/test_qwen36_dense_batch$(EXE): tests/test_qwen36_dense_batch.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/test_qwen36_dense_batch$(EXE): tests/test_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) -tests/test_qwen36_json_escape$(EXE): tests/test_qwen36_json_escape.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/test_qwen36_json_escape$(EXE): tests/test_qwen36_json_escape.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # Both tokenizer.json merge spellings ("a b" strings and ["a","b"] pairs) # must index the same merge table; the pair form is what Qwen3.6 ships. -tests/test_qwen36_tok_merges$(EXE): tests/test_qwen36_tok_merges.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/test_qwen36_tok_merges$(EXE): tests/test_qwen36_tok_merges.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # A byte-counted serving payload may end mid-character; the pre-tokenizer must # not read past it. qwen38 already gates this (tests/test_qwen38_tokenizer.c). -tests/test_qwen36_tok_truncated$(EXE): tests/test_qwen36_tok_truncated.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/test_qwen36_tok_truncated$(EXE): tests/test_qwen36_tok_truncated.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) + +# push_id / bpe_piece must refuse loudly, not crash, on a failed realloc during growth. +tests/test_qwen36_encode_oom$(EXE): tests/test_qwen36_encode_oom.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) + +# slot_ensure_int8's wiring onto #1271's unpack_int4_to_int8: regression check +# that its output matches the old separate scalar nibble-unpack loop it replaced. +tests/test_qwen36_slot_int8$(EXE): tests/test_qwen36_slot_int8.c qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) + +# the dense trunk's integer path: quantizer contract, int4 planar packing, dispatch. +tests/test_qwen36_dense_idot$(EXE): tests/test_qwen36_dense_idot.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) + $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) + +# #1653: added tokens are split out before the regex pre-tokenizer, HF-style. +tests/test_qwen36_tokenizer$(EXE): tests/test_qwen36_tokenizer.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) # Reproducible local timing evidence; intentionally not a noisy CI perf gate. -tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) -inkling$(EXE): inkling.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda_ink.h backend_metal.h edge_adapters.h edge_runtime.h edge_tok_internal.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(INK_CUDA_OBJ) $(METAL_OBJ) +inkling$(EXE): inkling.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h serve_budget.h backend_cuda_ink.h backend_metal.h edge_adapters.h edge_runtime.h edge_tok_internal.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) -o inkling$(EXE) $(LDFLAGS) # ENGINES WITHOUT A CUDA BACKEND (#783). @@ -1196,10 +1275,10 @@ NOCUDA_LDFLAGS = $(filter-out -lcudart -lstdc++ -lcuda -L$(CUDA_HOME)/lib64 \ # GLM-5.3-Flash: routed experts stream from the int4-gs64 container. # METAL=1 accelerates resident matrices and routed MoE; CPU remains fallback. -glm53$(EXE): glm53.c decode_batch.h pin_pool.h cli_args.h st.h json.h stop_ids.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h serve_poll.h route_trace.h quant.h hyper_connections.h delta_attention.h sparse_index.h vision_tower.h backend_metal.h backend_vulkan.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) +glm53$(EXE): glm53.c sse41_kernels.h decode_batch.h pin_pool.h cli_args.h st.h json.h stop_ids.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h serve_poll.h route_trace.h quant.h fp8_format.h idot.h hyper_connections.h delta_attention.h sparse_index.h vision_tower.h backend_metal.h backend_vulkan.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) $(CC) $(CFLAGS) glm53.c $(METAL_OBJ) $(VK_OBJ) -o glm53$(EXE) $(LDFLAGS) -kimi_k3$(EXE): kimi_k3.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda.h backend_metal.h backend_vulkan.h edge_adapters.h edge_runtime.h edge_tok_internal.h hybrid_split.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(CUDA_OBJ) $(VK_OBJ) $(VK_SPV) $(METAL_OBJ) +kimi_k3$(EXE): kimi_k3.c sse41_kernels.h serve_budget.h cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda.h backend_metal.h backend_vulkan.h edge_adapters.h edge_runtime.h edge_tok_internal.h hybrid_split.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(CUDA_OBJ) $(VK_OBJ) $(VK_SPV) $(METAL_OBJ) $(CC) $(CFLAGS) kimi_k3.c $(CUDA_OBJ) $(VK_OBJ) $(METAL_OBJ) -o kimi_k3$(EXE) $(LDFLAGS) # Use a baseline that matches the compiler target. macOS already targets a @@ -1227,8 +1306,8 @@ iobench$(EXE): iobench.c compat.h tests/test_serve_sentinel$(EXE): tests/test_serve_sentinel.c compat.h serve_codec.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_cluster_protocol$(EXE): tests/test_cluster_protocol.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_cluster_protocol$(EXE): tests/test_cluster_protocol.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) tests/test_ue8m0$(EXE): tests/test_ue8m0.c st.h json.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1239,10 +1318,13 @@ tests/test_json$(EXE): tests/test_json.c json.h tests/test_tok_o200k$(EXE): tests/test_tok_o200k.c tok.h tok_unicode.h tok_unicode_o200k.h json.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_k3_ram_budget$(EXE): tests/test_k3_ram_budget.c kimi_k3.c st.h tok.h quant.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h +tests/test_tok_gpt2$(EXE): tests/test_tok_gpt2.c tok.h tok_unicode.h tok_unicode_o200k.h json.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_k3_mmap$(EXE): tests/test_k3_mmap.c kimi_k3.c st.h tok.h quant.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h +tests/test_k3_ram_budget$(EXE): tests/test_k3_ram_budget.c sse41_kernels.h kimi_k3.c st.h tok.h quant.h fp8_format.h idot.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) + +tests/test_k3_mmap$(EXE): tests/test_k3_mmap.c sse41_kernels.h kimi_k3.c st.h tok.h quant.h fp8_format.h idot.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) tests/test_tok_kimi_tiny$(EXE): tests/test_tok_kimi_tiny.c tok.h tok_unicode.h tok_unicode_o200k.h json.h @@ -1253,13 +1335,19 @@ tests/test_st_pread$(EXE): tests/test_st_pread.c st.h json.h compat.h tests/test_st_slice$(EXE): tests/test_st_slice.c st.h json.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h edge_runtime.c edge_runtime.h edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h route_trace.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h edge_runtime.c edge_runtime.h edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< edge_runtime.c -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h + $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) + +tests/test_dsv41_serve_budget$(EXE): tests/test_dsv41_serve_budget.c sse41_kernels.h deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h fp8_format.h idot.h sparse_attn.h hyper_connections.h serve_codec.h serve_poll.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_serve_framing$(EXE): tests/test_qwen38_serve_framing.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_idot$(EXE): tests/test_qwen38_idot.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h + $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) + +tests/test_qwen38_serve_framing$(EXE): tests/test_qwen38_serve_framing.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) tests/test_qwen38_vision$(EXE): tests/test_qwen38_vision.c qwen38_vision.h st.h json.h compat.h @@ -1281,13 +1369,13 @@ qwen38-vision-serve-check: qwen38$(EXE) $(PYTHON) tools/make_edge_tiny_tokenizer.py --vocab-size 64 ./qwen38_mm_tiny $(PYTHON) tests/test_qwen38_vision_serve.py --binary ./qwen38$(EXE) --fixture ./qwen38_mm_tiny -tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h segment_runtime.c segment_runtime.h segment_adapters.h segment_adapter_internal.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h +tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h segment_runtime.c segment_runtime.h segment_adapters.h segment_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h $(CC) $(NOCUDA_CFLAGS) $< segment_runtime.c -o $@ $(NOCUDA_LDFLAGS) tests/test_st_map$(EXE): tests/test_st_map.c st.h json.h compat.h @@ -1299,6 +1387,28 @@ tests/test_dup_name_refusal$(EXE): tests/test_dup_name_refusal.c st.h json.h com tests/test_st_shape$(EXE): tests/test_st_shape.c st.h json.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +# Exhaustive bit-exact gate for st.h's bf16_to_f32_bulk/f16_to_f32_bulk AVX2 tier +# (all 65536 patterns per format -- see the file for why no tolerance applies). +tests/test_st_f16_bf16_simd$(EXE): tests/test_st_f16_bf16_simd.c st.h json.h compat.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march +# so the SSE4.1 body in st.h is actually exercised even on a devbox that +# would otherwise always pick the AVX2 tier. +tests/test_st_f16_bf16_simd_sse41$(EXE): tests/test_st_f16_bf16_simd.c st.h json.h compat.h + $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS) + +# bf16_to_f32_bulk/f16_to_f32_bulk vs the scalar per-element reference, NOT a +# test gate. Build on demand: make tests/bench_st_f16_bf16_simd ARCH=native +tests/bench_st_f16_bf16_simd$(EXE): tests/bench_st_f16_bf16_simd.c st.h json.h compat.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# Synthetic peak-RSS proxy for qwen36.c's dense-int8-during-load change +# (load_tq vs the old post-hoc qdw_register pass), NOT a test gate. Build on +# demand: make tests/bench_load_tq_peak_rss +tests/bench_load_tq_peak_rss$(EXE): tests/bench_load_tq_peak_rss.c + $(CC) -O3 -march=native $< -o $@ -lm + tests/test_st$(EXE): tests/test_st.c st.h json.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1333,25 +1443,30 @@ tests/test_grammar$(EXE): tests/test_grammar.c grammar.h # equivalence under the shipping flags is covered by the differential dump # (old vs new rope_interleave are byte-identical); this test guards the # source-level formula equivalence and must not depend on contraction luck. -tests/test_rope_invfreq$(EXE): tests/test_rope_invfreq.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) -ffp-contract=off $< -o $@ $(LDFLAGS) +tests/test_rope_invfreq$(EXE): tests/test_rope_invfreq.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) -ffp-contract=off $< $(VK_OBJ) -o $@ $(LDFLAGS) # schema->GBNF compile cache (#7): grammar_reset must equal a fresh setup, and # the GrDraft.src ownership must not leak/double-free. Includes colibri.c. -tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # Standalone: drives a faithful miniature of moe()'s routing+accumulate and links # the SAME abl.h the engine links -- no model/weights needed (the ablation-logic gate). tests/test_ablate$(EXE): tests/test_ablate.c abl.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h json.h +# Standalone: exercises the DEGRADE_ZERO miss-slot zero-fill logic extracted from +# moe() -- no model/weights needed (issue #865). +tests/test_degrade_zero$(EXE): tests/test_degrade_zero.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_spec_decode_state$(EXE): tests/test_spec_decode_state.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h json.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_spec_decode_state$(EXE): tests/test_spec_decode_state.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) + tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1361,13 +1476,16 @@ tests/test_pin_pool$(EXE): tests/test_pin_pool.c pin_pool.h kv_prefix.h tests/test_serve_codec$(EXE): tests/test_serve_codec.c serve_codec.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_serve_budget$(EXE): tests/test_serve_budget.c serve_budget.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + tests/test_segment_runtime$(EXE): tests/test_segment_runtime.c segment_runtime.c segment_runtime.h $(CC) $(CFLAGS) tests/test_segment_runtime.c segment_runtime.c -o $@ $(LDFLAGS) tests/test_segment_conformance$(EXE): tests/test_segment_conformance.c tests/segment_conformance_fixtures.c tests/segment_conformance_fixtures.h segment_runtime.c segment_runtime.h $(CC) $(CFLAGS) tests/test_segment_conformance.c tests/segment_conformance_fixtures.c segment_runtime.c -o $@ $(LDFLAGS) -tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) +tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h serve_budget.h $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) tests/test_inkling_shared_batch$(EXE): tests/test_inkling_shared_batch.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) @@ -1381,58 +1499,68 @@ tests/bench_inkling_shared_batch$(EXE): tests/bench_inkling_shared_batch.c inkli tests/test_inkling_cache_index$(EXE): tests/test_inkling_cache_index.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ) $(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS) -tests/test_kimi_serve_framing$(EXE): tests/test_kimi_serve_framing.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h +tests/test_kimi_serve_framing$(EXE): tests/test_kimi_serve_framing.c sse41_kernels.h serve_budget.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) -tests/test_kimi_cache_index$(EXE): tests/test_kimi_cache_index.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h +tests/test_kimi_cache_index$(EXE): tests/test_kimi_cache_index.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) -tests/test_k3_chat_tools$(EXE): tests/test_k3_chat_tools.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h +tests/test_k3_chat_tools$(EXE): tests/test_k3_chat_tools.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) # olmoe's matmul_q, not colibri's: compares the IDOT path against FP32 # ACTIVATIONS rather than an integer reference. NOCUDA_* for the same reason # the olmoe target uses it -- olmoe.c contains no COLI_CUDA code. -tests/test_olmoe_matmul_q$(EXE): tests/test_olmoe_matmul_q.c olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_matmul_q$(EXE): tests/test_olmoe_matmul_q.c sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_olmoe_serve_framing$(EXE): tests/test_olmoe_serve_framing.c olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_serve_framing$(EXE): tests/test_olmoe_serve_framing.c serve_budget.h sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_olmoe_cache_index$(EXE): tests/test_olmoe_cache_index.c olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h +tests/test_olmoe_cache_index$(EXE): tests/test_olmoe_cache_index.c sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) -tests/test_qwen36_cache_index$(EXE): tests/test_qwen36_cache_index.c qwen36.c expert_ffn.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) +tests/test_qwen36_cache_index$(EXE): tests/test_qwen36_cache_index.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS) -tests/test_idot$(EXE): tests/test_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_idot$(EXE): tests/test_idot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) -DCOLI_HAVE_GROUPED_PAIR $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_stops$(EXE): tests/test_stops.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +# Force the grouped-int4 oracle onto the SSE4.1 tier even on an AVX2 host. +tests/test_i4_grouped_sse41$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) -O3 -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_cfg_topk$(EXE): tests/test_cfg_topk.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_i4_grouped_sse41_o1$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) -O1 -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_topp$(EXE): tests/test_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_i4_grouped_sse41_no_contract$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) -O3 -ffp-contract=off -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS) + +tests/test_stops$(EXE): tests/test_stops.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) + +tests/test_cfg_topk$(EXE): tests/test_cfg_topk.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) + +tests/test_topp$(EXE): tests/test_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test # gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_topp -tests/bench_topp$(EXE): tests/bench_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_topp$(EXE): tests/bench_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_sample_nan$(EXE): tests/test_sample_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_sample_nan$(EXE): tests/test_sample_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_temp_env$(EXE): tests/test_temp_env.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_temp_env$(EXE): tests/test_temp_env.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c colibri.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c sse41_kernels.h colibri.c oracle.h st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) tests/test_kv_fp8$(EXE): tests/test_kv_fp8.c kv_fp8.h decode_batch.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1440,15 +1568,15 @@ tests/test_kv_fp8$(EXE): tests/test_kv_fp8.c kv_fp8.h decode_batch.h tests/test_kv_tq$(EXE): tests/test_kv_tq.c kv_tq.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_kv_disk$(EXE): tests/test_kv_disk.c colibri.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_kv_disk$(EXE): tests/test_kv_disk.c sse41_kernels.h colibri.c oracle.h st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # fmt=6 kernel oracle: needs the generated grid table, and its fixture comes from # the reference codec (tools/make_e8_fixture.py) — regenerate if the layout moves. -tests/test_e8_kernel$(EXE): tests/test_e8_kernel.c quant.h +tests/test_e8_kernel$(EXE): tests/test_e8_kernel.c sse41_kernels.h quant.h fp8_format.h idot.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_e4m3_vector$(EXE): tests/test_e4m3_vector.c quant.h +tests/test_e4m3_vector$(EXE): tests/test_e4m3_vector.c sse41_kernels.h quant.h fp8_format.h idot.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) tests/test_stop_ids$(EXE): tests/test_stop_ids.c stop_ids.h json.h @@ -1475,13 +1603,13 @@ fuzz-rans: tests/fuzz_rans.c rans.h tests/fuzz_rans.c -o tests/fuzz_rans -lm ./tests/fuzz_rans -tests/test_int3$(EXE): tests/test_int3.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_int3$(EXE): tests/test_int3.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_int3_load$(EXE): tests/test_int3_load.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_int3_load$(EXE): tests/test_int3_load.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c quant.h +tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c sse41_kernels.h quant.h fp8_format.h idot.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # Host-only: backend_cuda.h's format predicate is plain C, so its truth table is @@ -1490,17 +1618,24 @@ tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c quant.h tests/test_cuda_fmt_guard$(EXE): tests/test_cuda_fmt_guard.c backend_cuda.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_fp8_load$(EXE): tests/test_fp8_load.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h +# The fmt=8 LUT-gate state machine. Like test_cuda_fmt_guard it links no CUDA +# object and needs no toolchain: the two decisions live as pure predicates in +# backend_cuda.h and backend_cuda.cu calls them, so this pins the engine's own +# logic on a plain CPU build. +tests/test_cuda_lut_gate$(EXE): tests/test_cuda_lut_gate.c backend_cuda.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_qt_addrow$(EXE): tests/test_qt_addrow.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_fp8_load$(EXE): tests/test_fp8_load.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_logit_nan$(EXE): tests/test_logit_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_qt_addrow$(EXE): tests/test_qt_addrow.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_router_nan$(EXE): tests/test_router_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_logit_nan$(EXE): tests/test_logit_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) + +tests/test_router_nan$(EXE): tests/test_router_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1511,7 +1646,7 @@ tests/test_compat_direct$(EXE): tests/test_compat_direct.c compat.h tests/test_expert_store_ops$(EXE): tests/test_expert_store_ops.c expert_store.h tensor.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_native_quant$(EXE): tests/test_native_quant.c deepseek_v4.c native_quant.h tensor.h quant.h +tests/test_native_quant$(EXE): tests/test_native_quant.c sse41_kernels.h deepseek_v4.c native_quant.h tensor.h quant.h fp8_format.h idot.h $(CC) $(CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT deepseek_v4.c $< -o $@ $(LDFLAGS) tests/test_edge_runtime$(EXE): tests/test_edge_runtime.c edge_runtime.c edge_runtime.h @@ -1549,14 +1684,14 @@ SEGMENT_RUNTIME_LIB = $(SEGMENT_BUILD_DIR)/libcolibri_segment_edge.a $(SEGMENT_BUILD_DIR): mkdir -p $@ -$(SEGMENT_BUILD_DIR)/glm.o: colibri.c segment_runtime.h edge_runtime.h \ +$(SEGMENT_BUILD_DIR)/glm.o: colibri.c sse41_kernels.h oracle.h segment_runtime.h edge_runtime.h \ segment_adapters.h edge_adapters.h segment_adapter_internal.h \ edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DCOLIBRI_NO_MAIN -c colibri.c -o $@ -$(SEGMENT_BUILD_DIR)/glm53.o: glm53.c segment_runtime.h edge_runtime.h \ +$(SEGMENT_BUILD_DIR)/glm53.o: glm53.c sse41_kernels.h segment_runtime.h edge_runtime.h \ segment_adapters.h edge_adapters.h segment_adapter_internal.h \ - edge_adapter_internal.h st.h quant.h tok.h hyper_connections.h \ + edge_adapter_internal.h st.h quant.h fp8_format.h tok.h hyper_connections.h \ delta_attention.h sparse_index.h vision_tower.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DGLM53_NO_MAIN -c glm53.c -o $@ @@ -1565,29 +1700,29 @@ $(SEGMENT_BUILD_DIR)/inkling.o: inkling.c segment_runtime.h edge_runtime.h \ edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DINKLING_NO_MAIN -c inkling.c -o $@ -$(SEGMENT_BUILD_DIR)/kimi.o: kimi_k3.c segment_runtime.h edge_runtime.h \ +$(SEGMENT_BUILD_DIR)/kimi.o: kimi_k3.c sse41_kernels.h segment_runtime.h edge_runtime.h \ segment_adapters.h edge_adapters.h segment_adapter_internal.h \ edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DKIMI_K3_NO_MAIN -c kimi_k3.c -o $@ -$(SEGMENT_BUILD_DIR)/olmoe.o: olmoe.c segment_runtime.h edge_runtime.h \ +$(SEGMENT_BUILD_DIR)/olmoe.o: olmoe.c sse41_kernels.h segment_runtime.h edge_runtime.h \ segment_adapters.h edge_adapters.h segment_adapter_internal.h \ edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DOLMOE_NO_MAIN -c olmoe.c -o $@ -$(SEGMENT_BUILD_DIR)/qwen36.o: qwen36.c segment_runtime.h edge_runtime.h \ +$(SEGMENT_BUILD_DIR)/qwen36.o: qwen36.c gsgemv.h qgemv.h sse41_kernels.h segment_runtime.h edge_runtime.h \ segment_adapters.h edge_adapters.h segment_adapter_internal.h \ edge_adapter_internal.h st.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DQWEN36_NO_MAIN -c qwen36.c -o $@ -$(SEGMENT_BUILD_DIR)/qwen38.o: qwen38.c qwen38_core.h segment_runtime.h \ +$(SEGMENT_BUILD_DIR)/qwen38.o: qwen38.c sse41_kernels.h qwen38_core.h kv_prefix.h segment_runtime.h \ segment_adapters.h segment_adapter_internal.h edge_runtime.h \ - edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h \ + edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h \ qwen38_nfc.h qwen38_nfc_tables.h route_trace.h tok_unicode.h \ tok_unicode_o200k.h | $(SEGMENT_BUILD_DIR) $(CC) $(SEGMENT_CPU_CFLAGS) -DQWEN38_NO_MAIN -c qwen38.c -o $@ -$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: deepseek_v4.c deepseek_v4.h \ +$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h \ deepseek_v4_internal.h segment_runtime.h segment_adapters.h \ edge_runtime.h edge_adapters.h segment_adapter_internal.h \ edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR) @@ -1714,7 +1849,7 @@ V4_TEST_LINK_OBJS += COLI_V4_UNIT_GPU.o backend_loader_dsv4.o V4_TEST_EXTRA_TARGETS = COLI_V4_UNIT_GPU.o backend_loader_dsv4.o endif -tests/test_deepseek_v4$(EXE): tests/test_deepseek_v4.c deepseek_v4.c deepseek_v4.h compat.h \ +tests/test_deepseek_v4$(EXE): tests/test_deepseek_v4.c sse41_kernels.h deepseek_v4.c deepseek_v4.h compat.h \ Makefile.deepseek-v4.units Makefile.deepseek-v4 $(MAKE) -f Makefile.deepseek-v4 ARCH=$(ARCH) deepseek-v4-test-objs \ COLI_V4_UNIT_ST.o COLI_V4_UNIT_CONFIG.o COLI_V4_UNIT_MATH.o COLI_V4_UNIT_SPARSE_ATTENTION.o \ @@ -1739,16 +1874,16 @@ V4_OWNERSHIP_OBJS = \ $(V4_OWN_DIR): mkdir -p $(V4_OWN_DIR) -$(V4_OWN_DIR)/COLI_V4_UNIT_RUNTIME.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_RUNTIME.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_RUNTIME -c deepseek_v4.c -o $@ -$(V4_OWN_DIR)/COLI_V4_UNIT_CONFIG.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_CONFIG.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_CONFIG -c deepseek_v4.c -o $@ -$(V4_OWN_DIR)/COLI_V4_UNIT_ST.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h st.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_ST.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h st.h | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_ST -c deepseek_v4.c -o $@ -$(V4_OWN_DIR)/COLI_V4_UNIT_NATIVE_QUANT.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h quant.h | $(V4_OWN_DIR) +$(V4_OWN_DIR)/COLI_V4_UNIT_NATIVE_QUANT.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h quant.h fp8_format.h idot.h | $(V4_OWN_DIR) $(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT -c deepseek_v4.c -o $@ # The engine (RUNTIME unit) opens its expert store through the pluggable @@ -1760,7 +1895,7 @@ $(V4_OWN_DIR)/expert_store_registry.o: expert_store_registry.c expert_store_regi tests/test_v4_ownership$(EXE): tests/test_v4_ownership.c $(V4_OWNERSHIP_OBJS) $(CC) $(V4_OWN_CFLAGS) $< $(V4_OWNERSHIP_OBJS) -o $@ -pthread $(LDFLAGS) -tests/test_v4_serve_framing$(EXE): tests/test_v4_serve_framing.c deepseek_v4.c \ +tests/test_v4_serve_framing$(EXE): tests/test_v4_serve_framing.c sse41_kernels.h deepseek_v4.c \ deepseek_v4.h deepseek_v4_internal.h serve_codec.h Makefile.deepseek-v4 Makefile.deepseek-v4.units $(MAKE) -f Makefile.deepseek-v4 ARCH=$(ARCH) $@ @@ -1772,6 +1907,9 @@ tests/test_route_trace$(EXE): tests/test_route_trace.c route_trace.h compat.h tests/test_cli_args$(EXE): tests/test_cli_args.c cli_args.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_oracle$(EXE): tests/test_oracle.c oracle.h json.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + # Il tier CUDA con un backend finto: gira SENZA GPU perche' il test definisce i # coli_cuda_* e registra cosa riceve. E' il solo modo di provare in CI che un # esperto arriva davvero in VRAM e nel formato giusto (#1331) -- un test che si @@ -1785,12 +1923,20 @@ tests/test_qwen36_tier_int8$(EXE): tests/test_qwen36_tier_int8.c tests/qwen36_fa tests/test_qwen36_tier_multidev$(EXE): tests/test_qwen36_tier_multidev.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) +# Budget separate gate/up/down scale allocations at their own size classes. +tests/test_qwen36_tier_scale_budget$(EXE): tests/test_qwen36_tier_scale_budget.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + # Same fake backend, single device: proves qt_shutdown returns (under an # alarm(10) watchdog) instead of hanging forever when a group is still open # and an LFRU swap is parked waiting for cv_take (#1340). tests/test_qwen36_tier_shutdown$(EXE): tests/test_qwen36_tier_shutdown.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) +# Release owned projections and experts before CUDA teardown, then reopen safely. +tests/test_qwen36_tier_release$(EXE): tests/test_qwen36_tier_release.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + # Same fake backend, but driving the ENGINE: the warmstart in qwen36.c hands the # tier raw pointers into the RAM expert slots, so only a test that includes # qwen36.c can prove the weights behind them are still there afterwards -- on an @@ -1806,6 +1952,10 @@ tests/test_qwen36_tier_shutdown$(EXE): tests/test_qwen36_tier_shutdown.c tests/q tests/test_qwen36_tier_invariants$(EXE): tests/test_qwen36_tier_invariants.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) +# Async collection errors must drain all devices and reject partial output. +tests/test_qwen36_tier_take_error$(EXE): tests/test_qwen36_tier_take_error.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + # Same fake backend, with uploads that take time: qt_fill_wait must not return # until the last enqueued expert is RESIDENT, not merely dequeued -- the engine # frees the RAM int8 copies right after it (#1360 saw the gap one run in @@ -1821,6 +1971,10 @@ tests/test_qwen36_tier_fill_wait$(EXE): tests/test_qwen36_tier_fill_wait.c tests tests/test_qwen36_tier_autoplace$(EXE): tests/test_qwen36_tier_autoplace.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) +# Failed tier startup must release host storage and initialized synchronization. +tests/test_qwen36_tier_init_failure$(EXE): tests/test_qwen36_tier_init_failure.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + # Same fake backend: the fp8 streaming mode a model whose experts do not fit # in RAM needs (Qwen3.8) -- cap < n_experts accepted, fmt=8 uploads with # 128x128 block scales, bytes staged unchanged, no pointer kept into the @@ -1828,21 +1982,37 @@ tests/test_qwen36_tier_autoplace$(EXE): tests/test_qwen36_tier_autoplace.c tests tests/test_qwen36_tier_fp8$(EXE): tests/test_qwen36_tier_fp8.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) +# Failed gate/up/down uploads must release every unpublished CUDA tensor. +tests/test_qwen36_tier_rollback$(EXE): tests/test_qwen36_tier_rollback.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + # Generic resident dense matrices (the Qwen3.8 trunk) and per-offer placement. tests/test_qwen36_tier_dense$(EXE): tests/test_qwen36_tier_dense.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) -tests/test_qwen36_tier_int8_engine$(EXE): tests/test_qwen36_tier_int8_engine.c tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h +# DeltaNet input projection batching, state parity and block failure fallback. +tests/test_qwen36_dnproj_batch$(EXE): tests/test_qwen36_dnproj_batch.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h quant.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + +# the automatic trunk placement can be withdrawn after the startup probe (fake backend). +tests/test_qwen36_tier_withdraw$(EXE): tests/test_qwen36_tier_withdraw.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + +tests/test_qwen36_tier_int8_engine$(EXE): tests/test_qwen36_tier_int8_engine.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h + $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) + +# dnout / attnproj / shexp offered, placed and served from VRAM (fake backend). +tests/test_qwen36_trunk_dense$(EXE): tests/test_qwen36_trunk_dense.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # #1391: il decode path deve offrire gli esperti int8 al tier, non solo il warmstart -tests/test_qwen36_tier_int8_decode$(EXE): tests/test_qwen36_tier_int8_decode.c tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h +tests/test_qwen36_tier_int8_decode$(EXE): tests/test_qwen36_tier_int8_decode.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) # The qwen38 engine through its own main() on the fake backend: tier start # from the FP8 fixture, reduced CPU list, qt_note on recycled slots, oracle # tokens unchanged. Needs the fixture (qwen38-tiny-fp8-generate). -tests/test_qwen38_tier_engine$(EXE): tests/test_qwen38_tier_engine.c tests/qwen36_fake_cuda.h qwen38.c qwen38_core.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h quant.h route_trace.h +tests/test_qwen38_tier_engine$(EXE): tests/test_qwen38_tier_engine.c sse41_kernels.h tests/qwen36_fake_cuda.h qwen38.c qwen38_core.h kv_prefix.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS) tests/test_serve_poll$(EXE): tests/test_serve_poll.c serve_poll.h @@ -1858,19 +2028,22 @@ tests/test_rss_anon$(EXE): tests/test_rss_anon.c tests/test_mem_available$(EXE): tests/test_mem_available.c compat.h $(CC) $(CFLAGS) tests/test_mem_available.c -o $@ $(LDFLAGS) +tests/test_exact_dot$(EXE): tests/test_exact_dot.c exact_dot.h + $(CC) $(CFLAGS) tests/test_exact_dot.c -o tests/test_exact_dot$(EXE) $(LDFLAGS) + tests/test_798_guards$(EXE): tests/test_798_guards.c st.h json.h compat.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_dsa_select$(EXE): tests/test_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_dsa_select$(EXE): tests/test_dsa_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_corpus_draft$(EXE): tests/test_corpus_draft.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_corpus_draft$(EXE): tests/test_corpus_draft.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_cap_precedence$(EXE): tests/test_cap_precedence.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_mirror_stripe_split$(EXE): tests/test_mirror_stripe_split.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_cap_precedence$(EXE): tests/test_cap_precedence.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) +tests/test_mirror_stripe_split$(EXE): tests/test_mirror_stripe_split.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) tests/test_v4_hybrid_policy$(EXE): tests/test_v4_hybrid_policy.c deepseek_v4_hybrid.h hybrid_split.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1880,29 +2053,29 @@ tests/test_k3_fill_budget$(EXE): tests/test_k3_fill_budget.c hybrid_split.h tests/test_v4_bank_pair$(EXE): tests/test_v4_bank_pair.c deepseek_v4_bank_pair.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_ram_clamp$(EXE): tests/test_ram_clamp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_ram_clamp$(EXE): tests/test_ram_clamp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_cap_mixed_width$(EXE): tests/test_cap_mixed_width.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_cap_mixed_width$(EXE): tests/test_cap_mixed_width.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_eslot_inflight$(EXE): tests/test_eslot_inflight.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_eslot_inflight$(EXE): tests/test_eslot_inflight.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_glm_cache_index$(EXE): tests/test_glm_cache_index.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_glm_cache_index$(EXE): tests/test_glm_cache_index.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) # bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356), # NOT a test gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_dsa_select -tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_router_select is a microbenchmark (duplicate-prefix scan vs marked-score scan), # not a test gate. Build on demand: make tests/bench_router_select. -tests/bench_router_select$(EXE): tests/bench_router_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_router_select$(EXE): tests/bench_router_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_indexer_allocations: microbenchmark (DeepSeek V4 indexer malloc vs persistent arena scratch), NOT a test gate. @@ -1912,30 +2085,35 @@ tests/bench_indexer_allocations$(EXE): tests/bench_indexer_allocations.c # bench_idot: microbenchmark (single-acc vs independent-acc AVX-VNNI idot), NOT a test gate. # Build on demand on an AVX-VNNI CPU: make tests/bench_idot ARCH=native -tests/bench_idot$(EXE): tests/bench_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_idot$(EXE): tests/bench_idot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# bench_i4p_gidot: microbenchmark (per-row vs multi-row/AMX K1b grouped planar IDOT), NOT a test gate. +# Build on demand: make tests/bench_i4p_gidot ARCH=native +tests/bench_i4p_gidot$(EXE): tests/bench_i4p_gidot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_gemv_stream: microbenchmark (decode-regime GEMV bandwidth vs the read ceiling; # frozen-baseline + deinterleaved-x candidate A/B), NOT a test gate. # Build on demand: make tests/bench_gemv_stream ARCH=native -tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) # bench_mla_simd: microbenchmark (scalar vs AVX2/NEON MLA-absorb reductions, #442), # NOT a test gate. Build on demand: make tests/bench_mla_simd -tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h +tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_uring$(EXE): tests/test_uring.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_uring$(EXE): tests/test_uring.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_pipe_block$(EXE): tests/test_pipe_block.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_pipe_block$(EXE): tests/test_pipe_block.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) -tests/test_pilot_ring$(EXE): tests/test_pilot_ring.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_moe_gs_guard$(EXE): tests/test_moe_gs_guard.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h - $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +tests/test_pilot_ring$(EXE): tests/test_pilot_ring.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) +tests/test_moe_gs_guard$(EXE): tests/test_moe_gs_guard.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) tests/test_omp_tune$(EXE): tests/test_omp_tune.c omp_tune.h compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) @@ -1943,9 +2121,68 @@ tests/test_omp_tune$(EXE): tests/test_omp_tune.c omp_tune.h compat.h tests/test_compat_env$(EXE): tests/test_compat_env.c compat.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) -tests/test_kvb_notice$(EXE): tests/test_kvb_notice.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h +tests/test_kvb_notice$(EXE): tests/test_kvb_notice.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS) + +# Standalone: proves the group-scaled int8 GEMV keeps every row's float +# operations in their original order -- the assumption qwen36's byte-identical +# output rests on. Links the SAME gsgemv.h the engine links. +tests/test_gsgemv$(EXE): tests/test_gsgemv.c gsgemv.h sse41_kernels.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march +# so the SSE4.1 body in gsgemv.h is actually exercised even on a devbox that +# would otherwise always pick the AVX2 tier. +tests/test_gsgemv_sse41$(EXE): tests/test_gsgemv.c gsgemv.h sse41_kernels.h + $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS) + +# Standalone: proves the plain (non-group-scaled) int8 GEMV keeps its exact +# sequence of float operations -- the assumption qwen36's byte-identical +# output rests on. Links the SAME qgemv.h the engine links. +tests/test_qgemv$(EXE): tests/test_qgemv.c qgemv.h sse41_kernels.h $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) +# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march +# so the SSE4.1 body in qgemv.h is actually exercised even on a devbox that +# would otherwise always pick the AVX2 tier. +tests/test_qgemv_sse41$(EXE): tests/test_qgemv.c qgemv.h sse41_kernels.h + $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS) + +# olmoe's dot_i8_16 (ARM NEON/AVX2/SSE4.1 variants): must be bit-exact against +# a scalar int8 reference -- pure integer arithmetic, no tolerance needed. +# NOCUDA_* for the same reason the other olmoe.c-including test rules use it. +tests/test_olmoe_dot_i8_16$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c sse41_kernels.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h + $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS) + +# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march +# so the SSE4.1 body in olmoe.c's dot_i8_16 is actually exercised even on a +# devbox that would otherwise always pick the AVX2 tier. +tests/test_olmoe_dot_i8_16_sse41$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c sse41_kernels.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h + $(CC) $(NOCUDA_CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(NOCUDA_LDFLAGS) + +# layer_cuda_shard_kvb's format-allowlist refusal (see the test's file header). The +# function only exists under -DCOLI_CUDA, so on a default (CPU) build the binary is a +# loud SKIP; built with CUDA=1 it links backend_cuda.o ($(CUDA_OBJ)) and exercises the +# real guard -- no GPU needed, every probed path returns before any device context. +tests/test_shard_kvb_refuse$(EXE): tests/test_shard_kvb_refuse.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h sample.h kv_persist.h telemetry.h $(CUDA_OBJ) $(VK_OBJ) + $(CC) $(CFLAGS) $< $(VK_OBJ) $(CUDA_OBJ) -o $@ $(LDFLAGS) + +# The e8x4g64 container loader harness. It takes a minted directory on argv and +# is driven by tests/test_e8x4g64_mint_load.py, which builds it too -- this rule +# exists so it also compiles under the suite's own $(CFLAGS) rather than only +# under the driver's hand-copied flag list, and so a warning regression in it +# fails the normal build. +tests/test_e8x4g64_loader$(EXE): tests/test_e8x4g64_loader.c st.h quant.h fp8_format.h compat.h + $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS) + +# Reachable build-only entry point for the harness above: `make check` never +# calls this (it stays out of TEST_BINS/test-c, per TEST_EXCLUDE), so this adds +# no time to `check`. It exists so the suite's own $(CFLAGS) and warnings are +# actually exercised on demand, the way qwen38-tier-engine-check does for +# test_qwen38_tier_engine below. +.PHONY: e8x4g64-loader-check +e8x4g64-loader-check: tests/test_e8x4g64_loader$(EXE) + test-c: $(TEST_BINS) $(PYTHON) tools/run_tests.py $(TEST_BINS) @@ -2045,7 +2282,7 @@ install: colibri$(EXE) glm53$(EXE) inkling$(EXE) kimi_k3$(EXE) olmoe$(EXE) qwen3 $(INSTALL) -m 755 deepseek_v4$(EXE) $(DESTDIR)$(LIBEXECDIR)/deepseek_v4$(EXE); \ fi $(INSTALL) -m 644 family_registry.py resource_plan.py doctor.py autotune.py \ - openai_server.py cluster.py v4_dsml.py version.py $(DESTDIR)$(LIBEXECDIR)/ + openai_server.py cluster.py v4_dsml.py v41_dsml.py version.py $(DESTDIR)$(LIBEXECDIR)/ $(INSTALL) -m 644 tools/*.py $(DESTDIR)$(LIBEXECDIR)/tools/ @# The dashboard is an optional build artifact (cd web && npm run build), so install @# it only when it exists. It goes NEXT TO openai_server.py, which probes ./web/dist. @@ -2080,5 +2317,9 @@ bench: iobench$(EXE) tests/test_kv_prefix$(EXE): tests/test_kv_prefix.c kv_prefix.h $(CC) $(CFLAGS) tests/test_kv_prefix.c -o tests/test_kv_prefix$(EXE) $(LDFLAGS) -tests/test_kimi_request_state$(EXE): tests/test_kimi_request_state.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h +tests/test_kimi_request_state$(EXE): tests/test_kimi_request_state.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) $(CC) $(NOCUDA_CFLAGS) tests/test_kimi_request_state.c $(VK_OBJ) -o tests/test_kimi_request_state$(EXE) $(NOCUDA_LDFLAGS) + +# Kimi CUDA dispatch/fallback with a fake backend; no CUDA toolkit required. +tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c sse41_kernels.h kimi_k3.c backend_cuda.h kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ) + $(CC) $(NOCUDA_CFLAGS) -DCOLI_CUDA $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS) diff --git a/c/Makefile.deepseek-v4 b/c/Makefile.deepseek-v4 index a7858363f..fccd0bdae 100644 --- a/c/Makefile.deepseek-v4 +++ b/c/Makefile.deepseek-v4 @@ -223,8 +223,48 @@ REGISTRY_OBJ = expert_store_registry.o V4_SERVE_TEST := tests/test_v4_serve_framing$(if $(IS_WIN),.exe,) V4_SERVE_TEST_OBJS := $(filter-out COLI_V4_UNIT_GENERATE_STATS.o,$(V4_OBJS)) +# The objects are named after the unit, not after the flags they were built +# with, and three builds share them: this engine (CUDA=1 adds +# -DCOLI_V4_GPU_TIER), the parent's test-c (no GPU tier) and its test-asan +# (EXTRA_CFLAGS). Timestamps cannot tell those apart, so make would report +# "is up to date" and link one build's objects into another (#1702). The +# compiler line is recorded here and rewritten only when it changes; every +# object compiled with $(CFLAGS) depends on it. Same scheme as git's GIT-CFLAGS. +V4_FLAGS_STAMP = deepseek_v4.cflags +V4_TRACK_FLAGS = $(subst ','\'',$(CC) $(CFLAGS)) +$(V4_FLAGS_STAMP): FORCE + @flags='$(V4_TRACK_FLAGS)'; \ + if test x"$$flags" != x"`cat $@ 2>/dev/null`"; then \ + test -f $@ && echo "deepseek-v4: build flags changed, rebuilding the units" >&2; \ + echo "$$flags" > $@; \ + fi +FORCE: + +# The same defect one level down. backend_cuda_dsv4.o is named after its source, +# but what it contains comes from $(NVCC) $(V4_NVCCFLAGS): -arch=$(CUDA_ARCH) +# (or the -gencode preset), the -DCOLI_DSV4_NO_TC guard and the DeepGEMM defines +# all arrive through it. $(CFLAGS) above does not cover it, so without this: +# +# make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 CUDA_ARCH=sm_86 +# make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 CUDA_ARCH=sm_80 +# -> no output at all, exit 0 +# +# and the engine that asked for sm_80 links the sm_86 object, in silence. The +# parent Makefile keeps the same class of state for the same reason, CUDA_ARCH +# included, in .build-config (#306). Same scheme as $(V4_FLAGS_STAMP) above, so +# a dry run (make -n) or a clean writes nothing: the file is only touched by the +# recipe, and only when the command really changed. +V4_CUDA_FLAGS_STAMP = deepseek_v4.cudaflags +V4_TRACK_CUDA_FLAGS = $(subst ','\'',$(NVCC) $(V4_NVCCFLAGS)) +$(V4_CUDA_FLAGS_STAMP): FORCE + @flags='$(V4_TRACK_CUDA_FLAGS)'; \ + if test x"$$flags" != x"`cat $@ 2>/dev/null`"; then \ + test -f $@ && echo "deepseek-v4: CUDA build flags changed, rebuilding backend_cuda_dsv4.o" >&2; \ + echo "$$flags" > $@; \ + fi + .PHONY: deepseek-v4 deepseek-v4-objs deepseek-v4-clean deepseek-v4-test-objs \ - deepseek-v4-test-registry print-v4-objs + deepseek-v4-test-registry print-v4-objs FORCE deepseek-v4: $(V4_BINARY) # Build just the amalgamation unit + registry objects (no link). External @@ -242,24 +282,25 @@ $(V4_BINARY): $(V4_OBJS) # Pluggable expert-store backend registry (standalone; not a deepseek_v4.c unit). # Linked into the binary so the engine can dispatch COLI_EXPERT_STORE backends. -$(REGISTRY_OBJ): expert_store_registry.c expert_store_registry.h expert_store.h +$(REGISTRY_OBJ): expert_store_registry.c expert_store_registry.h expert_store.h \ + $(V4_FLAGS_STAMP) $(CC) $(CFLAGS) -c expert_store_registry.c -o $@ $(TARGET_OBJS) $(TEST_UNIT_OBJS): %.o: deepseek_v4.c deepseek_v4.h \ - deepseek_v4_internal.h deepseek_v4_dspark.inc st.h json.h compat.h tensor.h quant.h \ + deepseek_v4_internal.h deepseek_v4_dspark.inc st.h json.h compat.h tensor.h quant.h fp8_format.h sse41_kernels.h \ route_trace.h \ native_quant.h native_quant_batch.h native_quant_dual.h \ - native_quant_fp4_rows16.h expert_store_registry.h + native_quant_fp4_rows16.h expert_store_registry.h $(V4_FLAGS_STAMP) $(CC) $(CFLAGS) -D$* -c deepseek_v4.c -o $@ $(V4_HOT_TEST_OBJ): deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h \ - st.h json.h compat.h tensor.h quant.h route_trace.h \ - native_quant.h native_quant_fp4_rows16.h expert_store_registry.h + st.h json.h compat.h tensor.h quant.h fp8_format.h route_trace.h sse41_kernels.h \ + native_quant.h native_quant_fp4_rows16.h expert_store_registry.h $(V4_FLAGS_STAMP) $(CC) $(CFLAGS) -DCOLI_V4_TEST_HOOKS \ -DCOLI_V4_UNIT_EXPERT_STORE_HOT_ROWS16 -c deepseek_v4.c -o $@ $(V4_BATCH_TEST_OBJ): deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h \ - tensor.h quant.h native_quant.h native_quant_batch.h + tensor.h quant.h fp8_format.h native_quant.h native_quant_batch.h $(V4_FLAGS_STAMP) sse41_kernels.h $(CC) $(CFLAGS) -DCOLI_V4_TEST_HOOKS \ -DCOLI_V4_UNIT_NATIVE_QUANT_BATCH -c deepseek_v4.c -o $@ @@ -272,11 +313,12 @@ $(V4_SERVE_TEST): tests/test_v4_serve_framing.c deepseek_v4.c deepseek_v4.h \ # The CUDA tier loader object. Compiled only on Windows, where it is appended # to V4_OBJS and resolves the MSVC-built coli_cuda_dsv4.dll at runtime. On # Linux it is empty so the engine links without it. -backend_loader_dsv4.o: backend_loader_dsv4.c backend_cuda_dsv4.h +backend_loader_dsv4.o: backend_loader_dsv4.c backend_cuda_dsv4.h $(V4_FLAGS_STAMP) $(CC) $(CFLAGS) -c backend_loader_dsv4.c -o $@ # Linux/macOS CUDA=1: the tier compiled straight into the engine. -backend_cuda_dsv4.o: backend_cuda_dsv4.cu backend_cuda_dsv4.h $(V4_DEEPGEMM_DEP) +backend_cuda_dsv4.o: backend_cuda_dsv4.cu backend_cuda_dsv4.h $(V4_DEEPGEMM_DEP) \ + $(V4_CUDA_FLAGS_STAMP) "$(NVCC)" $(V4_NVCCFLAGS) -c backend_cuda_dsv4.cu -o $@ ifneq ($(DEEPGEMM_STAMP),) @@ -297,4 +339,5 @@ test_expert_store_registry: test_expert_store_registry.c expert_store_registry.c deepseek-v4-clean: rm -f $(V4_BINARY) $(V4_OBJS) $(TEST_UNIT_OBJS) $(V4_HOT_TEST_OBJ) \ $(V4_BATCH_TEST_OBJ) \ - $(V4_SERVE_TEST) test_expert_store_registry + $(V4_SERVE_TEST) test_expert_store_registry $(V4_FLAGS_STAMP) \ + $(V4_CUDA_FLAGS_STAMP) diff --git a/c/autotune.py b/c/autotune.py index b8f64e8ba..31d853f71 100644 --- a/c/autotune.py +++ b/c/autotune.py @@ -466,6 +466,16 @@ def run_once(name, overlay, launch_cap): if proc.returncode: raise RuntimeError(f"{name} failed ({proc.returncode})\n{output[-2000:]}") return parse_replay(output) + + def measure(name, overlay, launch_cap): + samples = [] + for repeat in range(repeats): + progress(f"{name} cap={cap} ({repeat + 1}/{repeats})") + sample = run_once(name, overlay, launch_cap) + sample["ttft_s"] = None + samples.append(sample) + recorded_cap = launch_cap if arch in CAP_ARCHES else None + return _summarize_measurement(name, overlay, samples, recorded_cap) else: if engine_cls is None: from openai_server import Engine as engine_cls @@ -533,17 +543,6 @@ def collect(piece): recorded_cap = launch_cap if arch in CAP_ARCHES else None return _summarize_measurement(name, overlay, samples, recorded_cap) - if arch == "glm": - def measure(name, overlay, launch_cap): - samples = [] - for repeat in range(repeats): - progress(f"{name} cap={cap} ({repeat + 1}/{repeats})") - sample = run_once(name, overlay, launch_cap) - sample["ttft_s"] = None - samples.append(sample) - recorded_cap = launch_cap if arch in CAP_ARCHES else None - return _summarize_measurement(name, overlay, samples, recorded_cap) - baseline = measure("baseline", {}, cap) winner = baseline accumulated = {} diff --git a/c/backend_cuda.cu b/c/backend_cuda.cu index 07d5dfc42..b7930a0f7 100644 --- a/c/backend_cuda.cu +++ b/c/backend_cuda.cu @@ -1,7 +1,12 @@ #include "backend_cuda.h" +#include "fp8_format.h" /* FP8_BLOCK: the shared fmt=8 scale-block edge (see that header) */ #include "backend_gpu_compat.h" +static_assert(FP8_BLOCK == 128, "fmt=8 on-disk containers carry ceil(dim/128)-edged scale " + "grids (mint tool, docs/FORMATS.md); FP8_BLOCK is container format, not a " + "tunable -- an edit here is a format change"); + /* Optional fmt=8 decode candidate (COLI_CUDA_F8_WARP=2): cuda_fp8.h maps * __nv_cvt_fp8_to_halfraw to an sm_89+ cvt instruction, with a bit-manip * fallback below 890. CUDA-only; the HIP build keeps the LUT decode. */ @@ -63,7 +68,7 @@ struct ColiCudaTensor { int fmt, I, O, device; int gs; /* quant group size; 0 = per-row scales (#334) */ int ng; /* number of scale groups per row = ceil(I/gs) for fmt=4 */ - size_t scale_count; /* floats in `scales`: O per-row, O*ng grouped */ + size_t scale_count; /* scale elements: ue8m0 bytes for fmt=7, floats otherwise */ int tracked; int weights_owned; #ifdef COLI_ANS @@ -74,6 +79,11 @@ struct ColiCudaTensor { int ragged_count; }; +static size_t tensor_scale_bytes(const ColiCudaTensor *t) { + if (!t->fmt || t->fmt == 6) return 0; + return t->scale_count * (t->fmt == 7 ? sizeof(uint8_t) : sizeof(float)); +} + #ifdef COLI_ANS struct AnsArenaChunk { uint8_t *p; size_t used,cap; }; #endif @@ -82,6 +92,11 @@ typedef struct { int compute_major,compute_minor; float *x, *y, *gate, *up; size_t x_cap, y_cap, gate_cap, up_cap; + /* Streaming MXFP4 weights are refreshed on every call; only storage is reused. */ + void *mxfp4_weights, *mxfp4_scales; + size_t mxfp4_weights_cap, mxfp4_scales_cap; + void *mxfp4_expert_weights, *mxfp4_expert_scales; + size_t mxfp4_expert_weights_cap, mxfp4_expert_scales_cap; /* Staging of the resident dense matvec (coli_cuda_matmul), apart from * x/y: the expert group (coli_cuda_expert_group_issue) runs on ctx->stream * asynchronously while the engine's thread keeps computing -- qwen38's @@ -277,13 +292,14 @@ __device__ static inline float mx4_weight_at(const uint8_t *q, int i) { * branch and the fall-through is a refusal. * * It used to be the other way round: int2 was the fall-through, so every format - * this function does not decode -- fmt=5 (int3-g64), fmt=6 (E8/IQ3), fmt=8 - * (fp8-e4m3), and anything added later -- was read as 2-bit values and returned - * numbers. Meanwhile the CPU functions doing the same job on the same tensor, - * qt_addrow and qt_matvec_rows (colibri.c), both exit(1) naming the function and - * the fmt. Two backends, identical unsupported input, one refusing and one - * fabricating: that asymmetry is the defect, independent of any particular - * format's arrival. + * this function does not decode -- fmt=5 (int3-g64), fmt=6 (E8/IQ3), and + * anything added later -- was read as 2-bit values and returned numbers. + * (fmt=8 was in that misread set too, then refused, until it gained its own + * explicit branch below for the absorb path.) Meanwhile the CPU functions + * doing the same job on the same tensor, qt_addrow and qt_matvec_rows + * (colibri.c), both exit(1) naming the function and the fmt. Two backends, + * identical unsupported input, one refusing and one fabricating: that + * asymmetry is the defect, independent of any particular format's arrival. * * WHY __trap() AND NOT A DIAGNOSTIC. This is device code inside a running * kernel; there is no stderr to name the tensor on and no way to unwind. __trap @@ -312,6 +328,15 @@ __device__ static float weight_at(const void *weights, int fmt, size_t row, int const uint8_t *base = static_cast(weights) + row; if (fmt == 0) return reinterpret_cast(base)[i]; if (fmt == 1) return static_cast(reinterpret_cast(base)[i]); + /* fmt=8 (fp8-e4m3): raw byte, same layout as fmt=1 (row_bytes(8,I)==I), decoded + * through the shared c_e4m3 LUT (same table quant_matmul's fmt==8 branch reads, + * uploaded once by coli_cuda_fp8_set_lut). Callers gate on the LUT being live + * before a fmt=8 tensor ever reaches this function (coli_cuda_tensor_upload + * refuses the upload otherwise), so the table is always populated here. Returns + * the decoded WEIGHT only, unscaled -- absorb_scale below applies the + * per-128x128-block scale, exactly like every other quantized fmt returns + * unscaled through this function. */ + if (fmt == 8) return c_e4m3[base[i]]; const uint8_t *q = base; if (fmt == 2 || fmt == 4) { /* fmt=4: same nibble layout */ uint8_t v = q[i >> 1]; @@ -325,13 +350,34 @@ __device__ static float weight_at(const void *weights, int fmt, size_t row, int return 0.0f; /* not reached: __trap() does not return */ } -/* Scale for output `row`, input element `k`. fmt=4 (grouped int4) stores ng - * scales per row at scales[row*ng + k/gs]; every other quantized format has - * one scale per row at scales[row]. Mirrors quant_matmul's fmt==4 branch so the +/* Scale for output `row`, input element `k`. Three layouts reach this: fmt=4 + * (grouped int4) stores ng scales per row at scales[row*ng + k/gs]; fmt=8 + * (fp8-e4m3) stores one scale per 128x128 BLOCK, block-row-major, and is + * handled by its own branch below; every OTHER quantized format has one scale + * per row at scales[row]. Mirrors quant_matmul's fmt==4 branch so the * attention absorb kernels apply per-group scales instead of the per-row * (fmt=2) semantic that crashed #298's g64 kv_b. */ __device__ static float absorb_scale(const float *wscale, int fmt, int gs, int ng, int row, int k) { if (!fmt) return 1.f; + if (fmt == 8) { + /* fp8-e4m3: one f32 scale per 128x128 BLOCK, block-row-major + * ([ceil(O/128), ceil(I/128)]), exactly quant_matmul's fmt==8 indexing + * (scl[i >> 7] on a scale row selected by o >> 7) and matmul_fp8's CPU + * reference (quant.h). `ng` here is coli_cuda_tensor_upload's t->ng, + * which for fmt=8 is set to ceil(I/128) specifically (not the fmt=4 + * group count) -- see the upload-time assignment there. `gs` is unused + * for fmt=8 (always 0, only fmt=4 sets it), so the block edge is the + * fixed FP8_BLOCK constant (fp8_format.h, shared with the CPU side), + * not a caller-supplied group size. Rounding note: the GEOMETRY here + * matches quant_matmul_f8w/matmul_fp8, but their fp8 accumulation + * convention (f32 partial per block, scale once per partial, double + * across blocks) is NOT carried into the absorb kernels -- they apply + * the scale per element into a float accumulator, matching their own + * fmt=4 arm's long-standing behavior; CPU-vs-CUDA absorb divergence + * is an accepted, documented class (#510). */ + int rowBlk = row / FP8_BLOCK, colBlk = k / FP8_BLOCK; + return wscale[(size_t)rowBlk * ng + colBlk]; + } if (fmt != 4) return wscale[row]; int g = k / gs; if (g >= ng) g = ng - 1; /* tail of the last (partial) group */ return wscale[(size_t)row * ng + g]; @@ -548,9 +594,9 @@ __global__ static void quant_matmul(float *y, const float *x, const void *weight * the ORIGINAL dense path, kept for COLI_CUDA_F8_WARP=0; the default * routes fmt=8 to quant_matmul_f8w instead (quant_matmul_launch). */ const uint8_t *wrow = static_cast(weights) + row; - const float *scl = scales + (size_t)(o >> 7) * (size_t)((I + 127) >> 7); + const float *scl = scales + (size_t)(o / FP8_BLOCK) * (size_t)((I + FP8_BLOCK - 1) / FP8_BLOCK); for (int i = threadIdx.x; i < I; i += blockDim.x) - sum += xs[i] * c_e4m3[wrow[i]] * scl[i >> 7]; + sum += xs[i] * c_e4m3[wrow[i]] * scl[i / FP8_BLOCK]; } else { for (int i = threadIdx.x; i < I; i += blockDim.x) sum += xs[i] * weight_at(weights, fmt, row, i); @@ -634,6 +680,14 @@ __global__ static void silu_mul(float *gate, const float *up, size_t n) { } } +__global__ static void situ_mul(float *gate, const float *up, size_t n, float b1, float b2) { + size_t i = (size_t)blockIdx.x * blockDim.x + threadIdx.x; + if (i < n) { + float g = gate[i], u = up[i]; + gate[i] = b1 * tanhf(g / b1) * (1.f / (1.f + expf(-g))) * b2 * tanhf(u / b2); + } +} + /* Four warps share one A tile and compute 16x64 outputs. This matters for * prefill: the first prototype reloaded/converter A once per 16 output cols. */ __global__ static void w4a16_matmul(float *y,const float *x,const uint8_t *w, @@ -1242,28 +1296,39 @@ extern "C" int coli_cuda_init(const int *devices, int count) { int available = 0; if (!devices || count < 1 || count > COLI_CUDA_MAX_DEVICES) return 0; if (!cuda_ok(cudaGetDeviceCount(&available), "device discovery")) return 0; - g_nctx = 0; + /* Validate the whole list before creating resources or replacing state. */ for (int i = 0; i < count; i++) { - int device = devices[i]; - if (device < 0 || device >= available) { - std::fprintf(stderr, "[CUDA] invalid device %d (available: 0..%d)\n", device, available - 1); - g_nctx = 0; + if (devices[i] < 0 || devices[i] >= available) { + std::fprintf(stderr, "[CUDA] invalid device %d (available: 0..%d)\n", devices[i], available - 1); return 0; } - if (find_ctx(device)) { - std::fprintf(stderr, "[CUDA] duplicate device %d\n", device); - g_nctx = 0; + for (int j = 0; j < i; j++) if (devices[j] == devices[i]) { + std::fprintf(stderr, "[CUDA] duplicate device %d\n", devices[i]); return 0; } + } + if (g_nctx) { + /* Same decision as before, routed through the shared predicate in + * backend_cuda.h so a host-side test can pin it without nvcc; the + * return value is unchanged (1 for the same set, 0 otherwise). */ + int live[COLI_CUDA_MAX_DEVICES]; + for (int i = 0; i < g_nctx; i++) live[i] = g_ctx[i].device; + int d = coli_cuda_init_disposition(g_nctx, count, devices, live); + if (d == COLI_CUDA_INIT_REFUSE) + std::fprintf(stderr, "[CUDA] device list change requires shutdown first\n"); + return d == COLI_CUDA_INIT_ACCEPT; + } + for (int i = 0; i < count; i++) { + int device = devices[i]; DeviceContext *ctx = &g_ctx[g_nctx]; *ctx = {}; ctx->device = device; - if (!select_ctx(ctx)) { g_nctx = 0; return 0; } + if (!select_ctx(ctx)) { coli_cuda_shutdown(); return 0; } cudaDeviceProp prop{}; - if (!cuda_ok(cudaGetDeviceProperties(&prop, device), "device properties")) { g_nctx = 0; return 0; } + if (!cuda_ok(cudaGetDeviceProperties(&prop, device), "device properties")) { coli_cuda_shutdown(); return 0; } ctx->compute_major=prop.major;ctx->compute_minor=prop.minor; if(!cuda_ok(cudaStreamCreateWithFlags(&ctx->stream,cudaStreamNonBlocking),"stream creation")){ - g_nctx=0;return 0; + coli_cuda_shutdown();return 0; } #ifdef COLI_ANS if(std::getenv("CUDA_RAW_EXPERTS")){ @@ -1288,6 +1353,10 @@ extern "C" void coli_cuda_shutdown(void) { for (int i = 0; i < g_nctx; i++) { DeviceContext *ctx = &g_ctx[i]; if (!select_ctx(ctx)) continue; + if (ctx->mxfp4_weights) cudaFree(ctx->mxfp4_weights); + if (ctx->mxfp4_scales) cudaFree(ctx->mxfp4_scales); + if (ctx->mxfp4_expert_weights) cudaFree(ctx->mxfp4_expert_weights); + if (ctx->mxfp4_expert_scales) cudaFree(ctx->mxfp4_expert_scales); if (ctx->x) cudaFree(ctx->x); if (ctx->y) cudaFree(ctx->y); if (ctx->dx) cudaFree(ctx->dx); @@ -1312,6 +1381,10 @@ extern "C" void coli_cuda_shutdown(void) { ctx->ans_scratch=nullptr;ctx->ans_chunks=nullptr;ctx->ans_raw=nullptr;ctx->ans_raw_cap=0; ctx->ans_host=nullptr;ctx->ans_host_cap=0;ctx->ans_copy_pending=0; #endif + ctx->mxfp4_weights = ctx->mxfp4_scales = nullptr; + ctx->mxfp4_weights_cap = ctx->mxfp4_scales_cap = 0; + ctx->mxfp4_expert_weights = ctx->mxfp4_expert_scales = nullptr; + ctx->mxfp4_expert_weights_cap = ctx->mxfp4_expert_scales_cap = 0; ctx->x = ctx->y = ctx->gate = ctx->up = nullptr; ctx->dx = ctx->dy = nullptr; ctx->dx_cap = ctx->dy_cap = 0; ctx->qx=nullptr; ctx->qscale=nullptr; @@ -1324,6 +1397,22 @@ extern "C" void coli_cuda_shutdown(void) { ctx->group_desc=nullptr; ctx->group_desc_cap=0; } g_nctx = 0; + /* g_fp8_lut_ready is PROCESS-WIDE while the e4m3 table (c_e4m3, a + * __constant__ device symbol whose lifetime is the CUDA primary context, + * not this file's host-side DeviceContext structs) is PER-DEVICE. A later + * coli_cuda_init may select a device the previous span never published to; + * without this reset the upload gate (g_fp8_lut_ready, checked in + * coli_cuda_tensor_upload) would still be satisfied from the PREVIOUS + * boot and admit fmt=8 tensors whose kernels there decode against an + * unwritten (zero) table: silent all-zero weights, the exact + * fabricated-numbers failure mode the format gates exist to refuse. + * Reset so every boot must publish its own LUT (coli_cuda_fp8_set_lut) + * before any fmt=8 upload. Shutdown is the ONLY site that needs to clear + * the flag: coli_cuda_init refuses a re-init that names a different + * device set (it returns early while g_nctx is non-zero, leaving the + * existing contexts and their published table untouched), so the device + * set can only WIDEN by passing through here first. */ + g_fp8_lut_ready = 0; #ifdef COLI_ANS if(g_ans_sidecar){std::fclose(g_ans_sidecar);g_ans_sidecar=nullptr;} #if defined(__linux__) @@ -1412,16 +1501,22 @@ extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor, /* fmt=6 keeps its scales inside each 98-byte block, so it is the one * quantized format that legitimately arrives with scales == NULL. */ if (!rb || (fmt && fmt != 6 && !scales)) return 0; - if (fmt == 8 && !g_fp8_lut_ready) return 0; /* kernels would read a zero LUT */ + /* kernels would read a zero LUT; shared predicate, pinned by + * tests/test_cuda_lut_gate.c without a CUDA toolchain */ + if (!coli_cuda_fp8_gate_admits(fmt, g_fp8_lut_ready)) return 0; ColiCudaTensor *t = static_cast(std::calloc(1, sizeof(*t))); if (!t) return 0; t->fmt = fmt; t->I = I; t->O = O; t->device = device; t->weight_bytes = rb * (size_t)O; t->gs = (fmt==4 && g_upload_gs>0) ? g_upload_gs : 0; t->ng = t->gs ? (I + t->gs - 1) / t->gs : 1; t->scale_count = t->gs ? (size_t)O * (size_t)t->ng : (size_t)O; - if (fmt == 8) { /* per-128x128-block scales: [ceil(O/128), ceil(I/128)] */ - t->ng = (I + 127) / 128; - t->scale_count = (size_t)((O + 127) / 128) * (size_t)t->ng; + if (fmt == 7) { + t->ng = (I + 31) / 32; + t->scale_count = (size_t)O * t->ng; + } + if (fmt == 8) { /* per-block scales: [ceil(O/FP8_BLOCK), ceil(I/FP8_BLOCK)] (fp8_format.h) */ + t->ng = (int)fp8_nblk(I); + t->scale_count = (size_t)fp8_nblk(O) * (size_t)t->ng; } if (!cuda_ok(cudaMalloc(&t->weights, t->weight_bytes), "tensor allocation")) { coli_cuda_tensor_free(t); @@ -1439,8 +1534,8 @@ extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor, offset_to_signed_s4<<<(unsigned)((t->weight_bytes+255)/256),256>>>((uint8_t*)t->weights,t->weight_bytes); if(!cuda_ok(cudaGetLastError(),"int4 weight conversion")){coli_cuda_tensor_free(t);return 0;}} if (fmt && fmt != 6) { - if (!cuda_ok(cudaMalloc(&t->scales, t->scale_count * sizeof(float)), "scale allocation") || - !cuda_ok(cudaMemcpy(t->scales, scales, t->scale_count * sizeof(float), cudaMemcpyHostToDevice), "scale upload")) { + if (!cuda_ok(cudaMalloc(&t->scales, tensor_scale_bytes(t)), "scale allocation") || + !cuda_ok(cudaMemcpy(t->scales, scales, tensor_scale_bytes(t), cudaMemcpyHostToDevice), "scale upload")) { coli_cuda_tensor_free(t); return 0; } @@ -1448,7 +1543,7 @@ extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor, if (fmt == 6) t->scale_count = 0; /* in-block scales: nothing separate to track */ t->tracked = 1; ctx->tensor_count++; - ctx->tensor_bytes += t->weight_bytes + ((fmt && fmt != 6) ? t->scale_count * sizeof(float) : 0); + ctx->tensor_bytes += t->weight_bytes + tensor_scale_bytes(t); *tensor = t; return 1; } @@ -1634,10 +1729,9 @@ extern "C" int coli_cuda_tensor_update(ColiCudaTensor *tensor, (uint8_t*)tensor->weights,tensor->weight_bytes); if(!cuda_ok(cudaGetLastError(),"int4 weight refresh")) return 0; } - /* fmt=6 has no scale buffer at all (scales live in-block, scale_count 0), and - * the fallback below would otherwise copy O floats out of a NULL host pointer. */ + /* fmt=6 stores scales in-block; fmt=7 stores byte exponents separately. */ return !tensor->fmt || tensor->fmt==6 || cuda_ok(cudaMemcpy(tensor->scales,scales, - (tensor->scale_count?tensor->scale_count:(size_t)tensor->O)*sizeof(float), + tensor_scale_bytes(tensor), cudaMemcpyHostToDevice),"scale refresh"); } @@ -1728,9 +1822,10 @@ extern "C" int coli_cuda_matmul_mxfp4(float *y, const float *x, size_t wb = (size_t)O * rb, sb = (size_t)O * ng; size_t xb = (size_t)S * I * sizeof(float), yb = (size_t)S * O * sizeof(float); - uint8_t *dw = nullptr, *ds = nullptr; - if (!cuda_ok(cudaMalloc(&dw, wb), "mxfp4 weight alloc")) return 0; - if (!cuda_ok(cudaMalloc(&ds, sb), "mxfp4 scale alloc")) { cudaFree(dw); return 0; } + if (!reserve_bytes(&ctx->mxfp4_weights, &ctx->mxfp4_weights_cap, wb) || + !reserve_bytes(&ctx->mxfp4_scales, &ctx->mxfp4_scales_cap, sb)) return 0; + uint8_t *dw = static_cast(ctx->mxfp4_weights); + uint8_t *ds = static_cast(ctx->mxfp4_scales); int ok = reserve(&ctx->x, &ctx->x_cap, xb) && reserve(&ctx->y, &ctx->y_cap, yb) && cuda_ok(cudaMemcpy(dw, q4, wb, cudaMemcpyHostToDevice), "mxfp4 weight upload") && @@ -1743,8 +1838,52 @@ extern "C" int coli_cuda_matmul_mxfp4(float *y, const float *x, ok = cuda_ok(cudaGetLastError(), "mxfp4 launch") && cuda_ok(cudaMemcpy(y, ctx->y, yb, cudaMemcpyDeviceToHost), "mxfp4 output download"); } - cudaFree(dw); - cudaFree(ds); + return ok; +} + +/* Reuse one weight/scale staging allocation across the three projections. + * Default-stream copies are ordered after the previous projection's reads. */ +static int mxfp4_project(float *y, const float *x, uint8_t *dw, uint8_t *ds, + const uint8_t *w, const uint8_t *sc, int S, int I, int O) { + size_t rb = ((size_t)I + 1) / 2, ng = ((size_t)I + 31) / 32; + if (!cuda_ok(cudaMemcpy(dw, w, (size_t)O * rb, cudaMemcpyHostToDevice), "expert weight upload") || + !cuda_ok(cudaMemcpy(ds, sc, (size_t)O * ng, cudaMemcpyHostToDevice), "expert scale upload")) return 0; + quant_matmul<<>>(y, x, dw, reinterpret_cast(ds), + 7, S, I, O, rb, 32, (int)ng); + return cuda_ok(cudaGetLastError(), "MXFP4 expert projection"); +} + +extern "C" int coli_cuda_expert_mxfp4(float *y, const float *x, + const unsigned char *gate_w, const unsigned char *gate_s, + const unsigned char *up_w, const unsigned char *up_s, + const unsigned char *down_w, const unsigned char *down_s, + int S, int D, int I, float b1, float b2) { + if (fault_injected() || !x || !y || !gate_w || !gate_s || !up_w || !up_s || + !down_w || !down_s || S < 1 || S > 65535 || D < 1 || I < 1 || + !(b1 > 0.f) || !(b2 > 0.f) || !std::isfinite(b1) || !std::isfinite(b2)) return 0; + DeviceContext *ctx = find_ctx(0); + if (!select_ctx(ctx)) return 0; + size_t xb = (size_t)S * D * sizeof(float), ib = (size_t)S * I * sizeof(float); + if (!reserve(&ctx->x, &ctx->x_cap, xb) || !reserve(&ctx->y, &ctx->y_cap, xb) || + !reserve(&ctx->gate, &ctx->gate_cap, ib) || !reserve(&ctx->up, &ctx->up_cap, ib)) return 0; + size_t gw = (size_t)I * (((size_t)D + 1) / 2), dwb = (size_t)D * (((size_t)I + 1) / 2); + size_t gs = (size_t)I * (((size_t)D + 31) / 32), dsb = (size_t)D * (((size_t)I + 31) / 32); + /* Grow to the largest projection seen, then reuse across routed experts. + * Slot identity is irrelevant: every call refreshes all weight bytes. */ + if (!reserve_bytes(&ctx->mxfp4_expert_weights, &ctx->mxfp4_expert_weights_cap, gw > dwb ? gw : dwb) || + !reserve_bytes(&ctx->mxfp4_expert_scales, &ctx->mxfp4_expert_scales_cap, gs > dsb ? gs : dsb)) return 0; + uint8_t *dw = static_cast(ctx->mxfp4_expert_weights); + uint8_t *ds = static_cast(ctx->mxfp4_expert_scales); + int ok = cuda_ok(cudaMemcpy(ctx->x, x, xb, cudaMemcpyHostToDevice), "expert input upload") && + mxfp4_project(ctx->gate, ctx->x, dw, ds, gate_w, gate_s, S, D, I) && + mxfp4_project(ctx->up, ctx->x, dw, ds, up_w, up_s, S, D, I); + if (ok) { + size_t n = (size_t)S * I; + situ_mul<<<(unsigned)((n + 255) / 256), 256>>>(ctx->gate, ctx->up, n, b1, b2); + ok = cuda_ok(cudaGetLastError(), "SiTU-GLU launch") && + mxfp4_project(ctx->y, ctx->gate, dw, ds, down_w, down_s, S, I, D) && + cuda_ok(cudaMemcpy(y, ctx->y, xb, cudaMemcpyDeviceToHost), "expert output download"); + } return ok; } @@ -2198,18 +2337,20 @@ extern "C" const float *coli_cuda_expert_group_take(int device) { /* The absorb kernels decode `w` through weight_at + absorb_scale, which know - * per-row and fmt=4 group scales only. Refuse anything else (fmt=5/6/8) rather - * than mis-decode it — the caller keeps its CPU attention path. (`proj` - * tensors are exempt: they run through quant_matmul, which dispatches every - * format it uploads.) A dedicated block-scale absorb for fmt=8 is follow-up - * work, same shape as routing fmt=4 through the grouped kernels was. + * per-row scales, fmt=4 group scales, and fmt=8 per-128x128-block scales. + * Refuse anything else (fmt=5/6/7) rather than mis-decode it — the caller + * keeps its CPU attention path. (`proj` tensors are exempt: they run through + * quant_matmul, which dispatches every format it uploads.) fmt=8 support + * funnels through this one predicate for all the absorb host wrappers below, + * so none of them needed a separate change. * * The admissible set is weight_at's own, taken from the shared predicate rather - * than restated as `fmt <= 4`: this gate and weight_at's device-side backstop - * must not be able to drift apart, and the old inequality also admitted - * NEGATIVE fmt values, which weight_at would then have fallen through on. Same - * truth table for every fmt a container can actually carry (0..8), so no - * existing container changes behaviour here. */ + * than restated as an inequality: this gate and weight_at's device-side + * backstop must not be able to drift apart, and the old `fmt <= 4` also + * admitted NEGATIVE fmt values, which weight_at would then have fallen through + * on. A fmt=8 tensor implies a live e4m3 LUT (upload refuses it otherwise -- + * see the predicate's caveat note in backend_cuda.h), so no extra gate is + * needed here. */ static int absorb_fmt_ok(const ColiCudaTensor *w){ return w && coli_cuda_weight_at_supported(w->fmt); } @@ -2386,19 +2527,14 @@ extern "C" void coli_cuda_tensor_free(ColiCudaTensor *tensor) { DeviceContext *ctx = find_ctx(tensor->device); if (ctx) select_ctx(ctx); if (tensor->tracked && ctx) { - /* Must mirror the upload's accounting exactly -- literally the same - * expression upload uses to charge (scale_count * sizeof(float), gated - * on fmt=6 never having a separate scale buffer), so the two can no - * longer drift independently. Over-subtracting here trips the >= guard - * below, which silently leaves the tensor's bytes on the device counter - * forever. */ + /* Charge and release the same format-specific scale storage. */ size_t storage_bytes = #ifdef COLI_ANS tensor->compressed ? tensor->archive_bytes : #endif tensor->weight_bytes; size_t bytes = storage_bytes + - ((tensor->fmt && tensor->fmt != 6) ? tensor->scale_count * sizeof(float) : 0); + tensor_scale_bytes(tensor); if (ctx->tensor_count) ctx->tensor_count--; if (ctx->tensor_bytes >= bytes) ctx->tensor_bytes -= bytes; } @@ -2410,19 +2546,14 @@ extern "C" void coli_cuda_tensor_free(ColiCudaTensor *tensor) { extern "C" size_t coli_cuda_tensor_bytes(const ColiCudaTensor *tensor) { if (!tensor) return 0; - /* Must mirror upload's and free's accounting exactly -- literally the same - * expression they use (scale_count * sizeof(float), gated on fmt=6 never - * having a separate scale buffer) -- so all three can no longer drift - * independently. The prior `O * ng` shape over-reported for fmt=8 (real - * footprint is (O+127)/128 * ng block scales, not O * ng) and for fmt=6 - * (which has no separate scale buffer at all). */ + /* Logical size uses the same scale layout as upload and free. */ size_t storage_bytes = #ifdef COLI_ANS tensor->compressed ? tensor->archive_bytes : #endif tensor->weight_bytes; return storage_bytes + - ((tensor->fmt && tensor->fmt != 6) ? tensor->scale_count * sizeof(float) : 0); + tensor_scale_bytes(tensor); } /* What a cudaMalloc of `bytes` actually takes off the card. @@ -2536,7 +2667,7 @@ extern "C" size_t coli_cuda_tensor_vram(const ColiCudaTensor *tensor) { tensor->weight_bytes; size_t total = coli_cuda_alloc_footprint(storage_bytes); if (tensor->fmt && tensor->fmt != 6) - total += coli_cuda_alloc_footprint(tensor->scale_count * sizeof(float)); + total += coli_cuda_alloc_footprint(tensor_scale_bytes(tensor)); return total; } diff --git a/c/backend_cuda.h b/c/backend_cuda.h index 953e8b39f..e58dbf95f 100644 --- a/c/backend_cuda.h +++ b/c/backend_cuda.h @@ -23,7 +23,9 @@ extern "C" { /* Weight formats the generic per-element device decoder (weight_at, * backend_cuda.cu) can actually decode: f32, int8-row, int4 nibbles (fmt=2 and - * the grouped fmt=4, same packing), and int2. Nothing else. + * the grouped fmt=4, same packing), int2, and fmt=8 (fp8-e4m3 raw bytes, + * decoded through the c_e4m3 LUT -- absorb-path support; absorb_scale supplies + * its per-128x128-block scale). Nothing else. * * WHY THIS IS A PREDICATE AND NOT A COMMENT. weight_at used to END in the int2 * decode as an unguarded fall-through, so ANY other format handed to it -- a @@ -42,18 +44,69 @@ extern "C" { * same arrangement colibri.c uses for metal_fused_fmt_ok. * * NOT a statement about which formats the CUDA BACKEND supports: quant_matmul - * has its own explicit branches for fmt=6 (E8/IQ3), fmt=7 (MXFP4) and fmt=8 - * (fp8-e4m3) that never route through weight_at. This predicate is scoped to - * weight_at's own dispatch, which is what the absorb and grouped-expert kernels - * decode through. */ + * has its own explicit branches for fmt=6 (E8/IQ3) and fmt=7 (MXFP4) that + * never route through weight_at (and its own fmt=8 branch for the dense path + * -- weight_at's fmt=8 branch serves the absorb kernels, which share the same + * c_e4m3 LUT). This predicate is scoped to weight_at's own dispatch, which is + * what the absorb and grouped-expert kernels decode through. + * + * fmt=8 CAVEAT, stated because the truth table alone cannot carry it: a fmt=8 + * decode additionally requires the e4m3 LUT to have been published to the + * configured devices (coli_cuda_fp8_set_lut). The exact mechanism, so the + * claim cannot outrun it: coli_cuda_fp8_set_lut copies the table into every + * context live AT CALL TIME and sets a process-wide flag; the flag gates + * fmt=8 uploads (coli_cuda_tensor_upload refuses until it is set). + * coli_cuda_shutdown clears the flag; coli_cuda_init never writes it. That is + * enough because init will not rebuild contexts underneath a live set: a + * re-init naming the same device set returns success and leaves the contexts, + * and the table published to them, untouched, while one naming a different + * device set is refused before any context is rebuilt. So the device set + * cannot widen past what the last publish covered without going through + * shutdown, and no fmt=8 ColiCudaTensor can reach a kernel whose device has + * an unwritten table. This predicate + * deliberately does not restate that gate: it answers "does weight_at have a + * decode branch for this fmt", which is the question the launch-site gates + * and the device-side __trap() backstop share. */ static inline int coli_cuda_weight_at_supported(int fmt) { - return fmt == 0 || fmt == 1 || fmt == 2 || fmt == 3 || fmt == 4; + return fmt == 0 || fmt == 1 || fmt == 2 || fmt == 3 || fmt == 4 || fmt == 8; +} + +/* The two decisions the fmt=8 LUT gate rests on, as pure predicates. They live + * here rather than inline in backend_cuda.cu so a host-side test can pin them + * with no CUDA toolchain and no GPU (tests/test_cuda_lut_gate.c). backend_cuda.cu + * calls BOTH at the real decision sites, so the test pins the engine's own + * logic rather than a second copy that could drift from it -- which is the + * failure this factoring exists to prevent, the gate having no CI reach + * otherwise. */ + +/* Does the upload gate admit this tensor? Only fmt=8 needs the published + * table; every other format decodes without one. */ +static inline int coli_cuda_fp8_gate_admits(int fmt, int lut_ready) { + return fmt != 8 || lut_ready != 0; +} + +/* What coli_cuda_init must do with a request while a device set may be live. + * BUILD: nothing is live, build the contexts. ACCEPT: the same set is already + * live -- return success and touch nothing, so the table published to those + * contexts stays valid. REFUSE: a different set is live -- refuse before + * rebuilding anything, so the set cannot widen past the last publish. */ +enum { COLI_CUDA_INIT_BUILD = 0, COLI_CUDA_INIT_ACCEPT = 1, COLI_CUDA_INIT_REFUSE = -1 }; +static inline int coli_cuda_init_disposition(int nctx, int count, + const int *want, const int *live) { + int i; + if (nctx <= 0) return COLI_CUDA_INIT_BUILD; + if (count != nctx) return COLI_CUDA_INIT_REFUSE; + for (i = 0; i < count; i++) if (want[i] != live[i]) return COLI_CUDA_INIT_REFUSE; + return COLI_CUDA_INIT_ACCEPT; } /* Opaque, persistent device copy of one resident quantized tensor. */ typedef struct ColiCudaTensor ColiCudaTensor; -/* Devices are CUDA ordinals, not positions in the input list. */ +/* Devices are CUDA ordinals, not positions in the input list. + * Repeating the same ordered list preserves active contexts. Changing an + * active list returns 0 without replacing it; release tensors and shut down + * before selecting a different list. Init/shutdown require caller serialization. */ COLI_CUDA_DLLEXPORT int coli_cuda_init(const int *devices, int count); COLI_CUDA_DLLEXPORT void coli_cuda_shutdown(void); /* Number of CUDA devices visible to this process, before a device list is @@ -115,6 +168,15 @@ COLI_CUDA_DLLEXPORT int coli_cuda_matmul_mxfp4(float *y, const float *x, const unsigned char *e8s, int S, int I, int O); +/* Streaming Kimi expert: down(SiTU(gate(x), up(x))). Weights are MXFP4 + * host buffers; intermediate activations remain on device. No weight cache. + * Returns 0 on failure; callers must accumulate y only after success. */ +COLI_CUDA_DLLEXPORT int coli_cuda_expert_mxfp4(float *y, const float *x, + const unsigned char *gate_w, const unsigned char *gate_s, + const unsigned char *up_w, const unsigned char *up_s, + const unsigned char *down_w, const unsigned char *down_s, + int S, int D, int I, float b1, float b2); + COLI_CUDA_DLLEXPORT int coli_cuda_matmul(ColiCudaTensor **tensor, float *y, const float *x, const void *weights, const float *scales, diff --git a/c/backend_gpu_compat.h b/c/backend_gpu_compat.h index dbbd07ff9..f3a3508f3 100644 --- a/c/backend_gpu_compat.h +++ b/c/backend_gpu_compat.h @@ -60,6 +60,7 @@ namespace nvcuda { namespace wmma = ::rocwmma; } #endif #define cudaError_t hipError_t #define cudaSuccess hipSuccess +#define cudaErrorMemoryAllocation hipErrorOutOfMemory #define cudaGetErrorString hipGetErrorString #define cudaGetLastError hipGetLastError #define cudaSetDevice hipSetDevice @@ -67,6 +68,7 @@ namespace nvcuda { namespace wmma = ::rocwmma; } #define cudaDeviceProp hipDeviceProp_t #define cudaGetDeviceProperties hipGetDeviceProperties #define cudaMalloc hipMalloc +#define cudaMallocManaged hipMallocManaged #define cudaFree hipFree #define cudaMemcpy hipMemcpy #define cudaMemcpy2D hipMemcpy2D diff --git a/c/backend_loader.c b/c/backend_loader.c index d877553a8..a3d4db200 100644 --- a/c/backend_loader.c +++ b/c/backend_loader.c @@ -102,6 +102,11 @@ typedef int (*fn_matmul)(ColiCudaTensor **tensor, float *y, const flo int fmt, int S, int I, int O, int device, int gs); typedef int (*fn_matmul_mxfp4)(float *y, const float *x, const unsigned char *q4, const unsigned char *e8s, int S, int I, int O); +typedef int (*fn_expert_mxfp4)(float *y, const float *x, + const unsigned char *gate_w, const unsigned char *gate_s, + const unsigned char *up_w, const unsigned char *up_s, + const unsigned char *down_w, const unsigned char *down_s, + int S, int D, int I, float b1, float b2); typedef void (*fn_tensor_free)(ColiCudaTensor *tensor); typedef size_t (*fn_tensor_bytes)(const ColiCudaTensor *tensor); typedef size_t (*fn_tensor_vram)(const ColiCudaTensor *tensor); @@ -179,6 +184,7 @@ static struct { fn_fp8_set_lut fp8_set_lut; fn_matmul matmul; fn_matmul_mxfp4 matmul_mxfp4; + fn_expert_mxfp4 expert_mxfp4; fn_tensor_free tensor_free; fn_tensor_bytes tensor_bytes; fn_tensor_vram tensor_vram; @@ -1433,6 +1439,7 @@ static int coli_cuda_load(void){ * nothing by this name, and the wrapper's 0 is the engine's own "fall back * to CPU" result, so an older DLL still serves GLM and Qwen3.6 (#1405). */ RESOLVE_OPT(matmul_mxfp4, fn_matmul_mxfp4) + RESOLVE_OPT(expert_mxfp4, fn_expert_mxfp4) RESOLVE_OPT(available_device_count, fn_available_device_count) /* qwen36 tier (#1533); older DLLs fall back to device_count */ RESOLVE(tensor_free, fn_tensor_free) RESOLVE(tensor_bytes, fn_tensor_bytes) @@ -1656,6 +1663,15 @@ int coli_cuda_matmul_mxfp4(float *y, const float *x, const unsigned char *q4, return g_cuda.matmul_mxfp4(y, x, q4, e8s, S, I, O); } +int coli_cuda_expert_mxfp4(float *y, const float *x, + const unsigned char *gate_w, const unsigned char *gate_s, + const unsigned char *up_w, const unsigned char *up_s, + const unsigned char *down_w, const unsigned char *down_s, + int S, int D, int I, float b1, float b2) { + if (!g_cuda.available || !g_cuda.expert_mxfp4) return 0; + return g_cuda.expert_mxfp4(y, x, gate_w, gate_s, up_w, up_s, down_w, down_s, S, D, I, b1, b2); +} + void coli_cuda_tensor_free(ColiCudaTensor *tensor){ if(g_cuda.available && g_cuda.tensor_free) g_cuda.tensor_free(tensor); } diff --git a/c/backend_metal.mm b/c/backend_metal.mm index e3497320c..3b767e2b3 100644 --- a/c/backend_metal.mm +++ b/c/backend_metal.mm @@ -767,7 +767,7 @@ static size_t fmt_bytes(int fmt, int I, int O) { // Grouped-int4 (fmt=4) scale-array size: one f32 per gsz-element group, per row -> O*ceil(I/gsz). // fp8 (fmt=8) scale-array size: one f32 per 128x128 BLOCK -> ceil(O/128)*ceil(I/128) (2D, // not per-row -- quant.h isn't included here, so the ceil-div is inlined rather than sharing -// colibri.c's qt_scale_bytes/quant.h's fp8_nblk). The block is a fixed 128x128, so gs is +// colibri.c's qt_scale_bytes/fp8_format.h's fp8_nblk). The block is a fixed 128x128, so gs is // ignored for fmt==8. f32 is this build's implemented scale // ENCODING for fmt=8 (see quant.h/colibri.c) -- this file has no reason to know that a // UE8M0 encoding exists at all: qt_resolve_fmt refuses it on the CPU read path before any diff --git a/c/coli b/c/coli index bdc1072e2..3f081a627 100755 --- a/c/coli +++ b/c/coli @@ -32,9 +32,9 @@ import os, sys, subprocess, argparse, json, time, signal, shutil, threading, re, # and the console provides its own editing; on FreeBSD Python links libedit, # which this activates the same way. try: - import readline # noqa: F401 — importing it is the activation + import readline except ImportError: - pass + readline = None # The engine mmaps every shard (144+ files); macOS default RLIMIT_NOFILE is 256. if sys.platform != "win32": @@ -53,7 +53,10 @@ if sys.platform == "win32": try: s.reconfigure(encoding="utf-8") except (AttributeError, OSError): pass -HERE = os.path.dirname(os.path.abspath(__file__)) +# realpath, not abspath: on a merged-/usr system /bin and /sbin are symlinks +# to /usr/bin, and an installed launcher invoked as /bin/coli would otherwise +# derive /libexec/colibri instead of /usr/libexec/colibri (#1689). +HERE = os.path.dirname(os.path.realpath(__file__)) sys.path.insert(0, HERE) # version.py sits next to this script in a source checkout, but an installed # layout puts the launcher in $(PREFIX)/bin while the support modules live in @@ -477,9 +480,10 @@ def env_for_engine(a, arch, plan=None): # sets it for GLM and openai_server.py sets it for the gateway, so `coli # chat` and `coli serve` worked while `coli run` handed the sister engines # an environment without it: OLMoE exited with "started without a model" - # (#1501). Set it here, once, for all of them. + # (#1501, #1600). Set it here, once, for all of them. --model is the + # directory, same as env_for(): a leftover SNAP must not load another model. if getattr(a, "model", None): - env.setdefault("SNAP", os.path.abspath(a.model)) + env["SNAP"] = os.path.abspath(a.model) if arch == "olmoe": env["CHAT"] = "1" env["MAX_NEW"] = str(ngen_for(a, family=arch)) @@ -491,7 +495,7 @@ def env_for_engine(a, arch, plan=None): # #855; before that `grep -c RAM_GB c/kimi_k3.c` returned 0, so `coli chat # --ram 242` on Kimi K3 set an environment variable nobody looked at and the # user's session ran itself out of memory with the flag apparently set. - if arch in ("deepseek_v4", "kimi", "glm53", "olmoe"): + if arch in ("deepseek_v4", "deepseek_v41", "kimi", "glm53", "olmoe"): if a.ram: env["RAM_GB"] = str(a.ram) if arch == "glm53": # I densi vanno a int4 di default: sono 9,7 B parametri su 321, e in @@ -606,6 +610,14 @@ def env_for_engine(a, arch, plan=None): gain=100.0*profile["gain"] print(f" {C.dim}[TUNE] applied measured profile · +{gain:.1f}% " f"calibration throughput{C.r}",file=sys.stderr) + # Kimi's CUDA path is gated on K3_CUDA, not COLI_CUDA. --gpu / --vram / + # auto-tier write COLI_CUDA=1 after proving the build; without this copy + # the flag was accepted and the experts still stayed on the CPU. + if arch == "kimi": + if env.get("COLI_CUDA") == "0": + env["K3_CUDA"] = "0" + elif env.get("COLI_CUDA") == "1" and "K3_CUDA" not in explicit_env: + env["K3_CUDA"] = "1" return env def dsv4_cuda_available(model=None): @@ -672,15 +684,20 @@ def cuda_binary(engine=None): for line in linked.stdout.splitlines()) except (OSError,subprocess.SubprocessError): return False if sys.platform == "win32": - # Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads - # coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no - # import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a - # marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require - # coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup). + # Windows CUDA_DLL/HIP_DLL hosts never link the GPU runtime directly: + # the engine LoadLibrary's its backend (backend_loader.c). The host + # compiles exactly one basename -- coli_hip.dll or coli_cuda.dll -- + # and that is the file that must sit next to it. The GLM/Qwen banner + # is not a GPU-build marker: Kimi K3 CUDA_DLL builds link the same + # loader without printing it, and a HIP host that only had coli_hip.dll + # used to be refused as CPU-only. try: - with open(engine,"rb") as f: built=b"[CUDA] mode: routed experts" in f.read() + with open(engine,"rb") as f: image=f.read() except OSError: return False - return built and os.path.exists(os.path.join(os.path.dirname(engine),"coli_cuda.dll")) + from doctor import windows_backend_dll + expected=windows_backend_dll(image) + if not expected: return False + return os.path.exists(os.path.join(os.path.dirname(engine),expected)) return False def resource_request(a, env): @@ -1191,7 +1208,7 @@ def cmd_info(a): try: n_shards=len([x for x in os.listdir(a.model) if x.endswith('.safetensors')]) except OSError: n_shards=0 print(f" {C.yel}config.json is missing{C.r}: coli picks the engine from it, so nothing can run here yet.") - print(f" Copy the checkpoint's config.json (with tokenizer.json and model.safetensors.index.json)") + print(" Copy the checkpoint's config.json (with tokenizer.json and model.safetensors.index.json)") print(f" from the model repo next to the {n_shards} shard(s) found here, then run coli info again.") try: mi=open('/proc/meminfo').read() @@ -1511,9 +1528,7 @@ def chat_commands_help(): def install_chat_completer(): """TAB completa i comandi. Senza readline (Windows) si perde solo il TAB.""" - try: - import readline - except ImportError: + if readline is None: return names = [p + n for n in CHAT_COMMANDS for p in ("/", ":")] def complete(text, state): @@ -1624,7 +1639,10 @@ def chat_attached(a, base, model_id): else: print(f" {C.yel} /brio needs the options: /brio merge | request changes | close{C.r}\n") continue - brio_options=[o.strip() for o in spec.split("|") if o.strip()] + # dict.fromkeys deduplica tenendo l'ordine: il server rifiuta due + # opzioni uguali con un 400, e "merge | merge | close" e' un refuso + # facile da fare a mano, non una domanda a tre opzioni. + brio_options=list(dict.fromkeys(o.strip() for o in spec.split("|") if o.strip())) if len(brio_options)<2: brio_options=[] print(f" {C.yel} at least two options are needed, separated by |{C.r}\n") @@ -2252,9 +2270,21 @@ def cmd_stop(a): print(f" nothing running — no serve on port {a.port}, no SERVE engines"); return for pid,desc in targets.items(): print(f" {'would stop' if a.dry_run else 'stopping'} {pid}: {desc}") if a.dry_run: return - for pid in targets: + # Graceful first: only the launcher's Engine.close drains the engine + # (stdin EOF -> atexit -> HEAT_FILE save). SIGTERM the launchers and + # give them the drain window to exit on their own; engine pids left + # behind (orphaned, no launcher) get the old treatment after. + launchers = [pid for pid, d in targets.items() if d.startswith("coli serve")] + strays = [pid for pid, d in targets.items() if not d.startswith("coli serve")] + deadline = time.time() + 30.0 + for pid in launchers: try: os.kill(pid, signal.SIGTERM) except OSError: pass + while time.time() < deadline and any(_pid_alive(p) for p in launchers): + time.sleep(0.25) + for pid in strays: + try: os.kill(pid, signal.SIGTERM) # normally already gone with the launcher + except OSError: pass time.sleep(2.0) for pid in targets: # signal.SIGKILL does not exist on win32 and AttributeError is not OSError, diff --git a/c/colibri.c b/c/colibri.c index 348d4e1af..678a30f23 100644 --- a/c/colibri.c +++ b/c/colibri.c @@ -60,6 +60,7 @@ #include /* hwinfo_emit: CPU brand string senza /proc */ #endif #include "cli_args.h" +#include "oracle.h" #include "st.h" #ifdef __linux__ #include "uring.h" @@ -285,9 +286,10 @@ static int64_t qt_bytes(const QT *t){ /* byte residenti del tensore */ return (int64_t)t->O*(((int64_t)t->I+255)/256)*98 + 4; if(t->fmt==8){ /* fp8-e4m3 passthrough: O*I raw e4m3 bytes (n, byte-identical layout * to fmt=1's weight bytes) + one f32 scale per 128x128 block - * (FP8_BLOCK in quant.h, included below qt_bytes -- keep the - * arithmetic literal here, same discipline as fmt=5's comment - * above). Missing this branch would fall through to the fmt=2 + * (FP8_BLOCK in fp8_format.h via quant.h, included below + * qt_bytes -- keep the arithmetic literal here, same + * discipline as fmt=5's comment above). Missing this branch would + * fall through to the fmt=2 * default below (packed-nibble formula, ~half the real weight * bytes) and undercount a resident fp8 tensor's byte footprint -- * feeds AUTOPIN/RAM-budget math, so this branch is load-bearing @@ -546,6 +548,19 @@ typedef struct { * than in quant.h: that header is shared by standalone kernel tests and * sibling engines, where translation-unit-local copies are unused and trip * -Wunused-variable. */ +#include "exact_dot.h" +/* COLI_EXACT_VERIFY=1 (opt-in, #689): during draft+verify forwards (g_spec_live) the CPU + * MLA-absorb attention core accumulates its score and context dots EXACTLY (integer products, + * one rounding per dot; exact_dot.h). No summation order, SIMD width or contraction flag can + * change those bits, so a verify row decides near-ties the same way on every host. Off by + * default: it is an integer path (~7x the float loop on the dot itself at -O3). The default paths + * are untouched. */ +static int g_exact_verify=-1; +static int exact_verify_on(void){ + if(g_exact_verify<0){ const char *e=getenv("COLI_EXACT_VERIFY"); g_exact_verify=(e&&atoi(e))?1:0; + if(g_exact_verify) fprintf(stderr,"[EXACT_VERIFY] draft+verify attention core on the exact (order-independent) dot (#689; COLI_EXACT_VERIFY=0 to disable)\n"); } + return g_exact_verify; +} static int g_idot=1; #if defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD) static int g_i4s=1; @@ -902,7 +917,9 @@ static double edisk_s(void){ return atomic_load_explicit(&g_edisk_ns,memory_orde * served that gets labeled cold overstates the cold class, the bucket this line exists to * size). */ static uint32_t g_direct_heat_ticks=0; +#ifndef COLIBRI_NO_MAIN static int g_direct_heat_explicit=0; /* 1 if COLI_DISKCLASS_WINDOW was set (skip the auto-derive) */ +#endif #define DC_COLD 0 #define DC_WARM 1 static _Atomic uint64_t g_dc_n[2]; /* [DC_COLD]/[DC_WARM]: loads classified */ @@ -1012,8 +1029,16 @@ static void matmul_i4_grouped_pair(float *yg, float *yu, const float *x, const uint8_t *qu, const float *su, int S, int I, int O, int gs){ int rb=(I+1)/2; int ng=(I+gs-1)/gs; + int o0=0; +#if defined(__SSE4_1__) && !defined(__AVX2__) + if(!(gs&1)){ + o0=O&~3; + if(o0) matmul_i4_grouped_pair_sse41_rows4(yg,yu,x,qg,sg,qu,su,S,I,O,gs,rb,ng,o0); + if(o0==O) return; + } +#endif #pragma omp parallel for schedule(static) - for(int o=0;oq4,t->O,t->I); t->planar=1; return; } if(t->fmt!=2) return; +#if defined(__AVX512F__)&&defined(__AVX512BW__) + /* fmt=2 stays a coppie qui: il gemello f32 planare replica l'ordine di + * accumulo AVX2, non quello del ramo dot_i4f_avx512 a 512 bit — il claim + * bit-identico della famiglia f32 vale solo dove i gemelli coincidono. + * EN: fmt=2 keeps the pair layout on AVX-512 builds; the f32 planar twin + * mirrors the AVX2 accumulation order, not the 512-bit f32 arm's. */ + return; +#endif planarize_i4(t->q4,t->O,t->I); t->planar=1; if(atomic_fetch_add_explicit(&g_planar_n,1,memory_order_relaxed)==0) fprintf(stderr,"[K1] planar int4 layout active (PLANAR=0 disables)\n"); @@ -1301,6 +1339,13 @@ static int g_expert_budget=0; /* EXPERT_BUDGET=N -> cap distinct experts loaded * (arXiv 2602.16052): top-32 of 64 capture 93% routing weight. */ static int64_t g_budget_dropped=0; /* total experts dropped by EXPERT_BUDGET across all layers */ static int64_t g_budget_rescued=0; /* experts re-kept because a position would have been left with zero */ +static int g_degrade_zero=0; /* DEGRADE_ZERO=1: zero-fill miss slots whose per-position gate weight + * is below DEGRADE_TAU instead of blocking on a demand-load. + * Opt-in only; changes output. Decode-only (S<=4 guard in moe()). */ +static float g_degrade_tau=0.03f; /* DEGRADE_TAU=: gate weight threshold (default 0.03). + * Issue #865: tau=0.03 zeroes 21.8% of slots for +2.9% perplexity. */ +static int64_t g_degrade_dropped=0; /* cumulative miss slots zeroed by DEGRADE_ZERO across all layers */ +static int64_t g_degrade_dropped_by_layer[512]; /* per-layer miss slots zeroed (for footer breakdown) */ /* CACHE_ROUTE (paper 2412.00099 max-rank): opt-in only. Keep true top-J always; * fill remaining slots preferring pin∪LRU experts ranked within top-M (or mass ROUTE_P). */ static int g_cache_route=0; @@ -1642,7 +1687,8 @@ static void rope_interleave(float *v, int pos, const Cfg *c){ * unverified mirrors (see qt_check_fmt threat model); an unbounded ftell->malloc * gave a hostile file a load-time OOM or, on malloc failure, a NULL deref via * b[got]=0. Cap the size, NULL-check the alloc, require a full read. Returns a - * malloc'd NUL-terminated buffer, or NULL on any failure. Mirrors tok.h tk_read_file. */ + * malloc'd NUL-terminated buffer, or NULL on any failure. Embedded NUL bytes are + * invalid JSON and must not hide an unchecked suffix. Mirrors tok.h tk_read_file. */ #define CFG_MAX_BYTES (256ll<<20) /* config/oracle JSON is KB-MB in practice */ static char* cfg_slurp(const char *path){ FILE *f=fopen(path,"rb"); if(!f) return NULL; @@ -1650,7 +1696,7 @@ static char* cfg_slurp(const char *path){ if(n<0 || (long long)n>CFG_MAX_BYTES){ fclose(f); return NULL; } char *b=malloc((size_t)n+1); if(!b){ fclose(f); return NULL; } size_t got=fread(b,1,(size_t)n,f); fclose(f); - if((long)got!=n){ free(b); return NULL; } + if((long)got!=n || memchr(b,'\0',got)){ free(b); return NULL; } b[got]=0; return b; } static jval* cfg_root(const char *snap, char **arena){ @@ -2241,6 +2287,42 @@ static void qt_cuda_colocate(QT *dst,const QT *src){ } static void layer_cuda_shard_kvb(Layer *l,int H,int Q,int V){ if(!g_cuda_enabled||!g_cuda_dense||g_cuda_ndev<2||l->kv_b.fmt==0)return; + /* SHARD FORMAT ALLOWLIST (explicit refusal; this was an ACCIDENTAL fail-safe): the + * rb/weights/scale arithmetic below is written for exactly fmt=1 (int8, per-row + * scale), fmt=2 (int4 per-row), fmt=3 (int2 per-row) and fmt=4 (int4 grouped). + * Any other fmt reaching it computes a wrong row-byte stride, takes l->kv_b.q4 as + * the weight pointer (NULL for fmt=8, whose raw e4m3 bytes live in q8 -- see the + * QT struct comment), and slices l->kv_b.s with per-row/per-group geometry that + * fmt=8's per-128x128-BLOCK scales (and fmt=6's single 4-byte tag) simply do not + * have. fmt=8 only ever "worked" here by accident: q4==NULL made + * coli_cuda_tensor_upload_g's !weights check reject the upload before anything + * dereferenced it -- silent, unnamed, and one refactor away from a misread. + * Refuse BY NAME instead, BEFORE any pointer/stride use, and say what happens + * instead: the un-sharded kv_b stays whole on its layer home device, where fmt=8 + * kv_b decode runs the absorb path (qt_addrow/qt_matvec_rows' fmt=8 branches, or + * the CUDA absorb kernels via absorb_fmt_ok) -- COLI_CUDA_ATTN_SHARD is a no-op + * for it. Same "refuse rather than misread" discipline as qt_addrow/ + * qt_matvec_rows' guards; notice only (no exit): sharding is an opt-in + * optimization and skipping it is the correct, working behavior. Bounded once + * per process per fmt, never per layer (metal_fmt_gate_notice, the precedent + * for bounded notices, is coarser still: one line per tensor KIND, naming only + * the first offending fmt). */ + if(l->kv_b.fmt!=1&&l->kv_b.fmt!=2&&l->kv_b.fmt!=3&&l->kv_b.fmt!=4){ + static int refused_fmt[32]; + if(!refused_fmt[l->kv_b.fmt&31]){ refused_fmt[l->kv_b.fmt&31]=1; + if(l->kv_b.fmt==8) + fprintf(stderr,"layer_cuda_shard_kvb: kv_b fmt=8 (fp8-e4m3, per-128x128-block " + "scales) has no head-shard layout here -- refusing the shard; fmt=8 kv_b " + "runs the absorb path on the layer home device instead, so " + "COLI_CUDA_ATTN_SHARD is a no-op for it (applies to every layer)\n"); + else + fprintf(stderr,"layer_cuda_shard_kvb: unsupported kv_b fmt=%d for the head-shard " + "upload (only fmt 1/2/3/4 match the per-row byte/scale strides computed " + "here) -- refusing the shard; kv_b stays whole on its layer home device " + "(applies to every layer)\n",l->kv_b.fmt); + } + return; + } int rb=l->kv_b.fmt==1?l->kv_b.I: (l->kv_b.fmt==2||l->kv_b.fmt==4)?(l->kv_b.I+1)/2:(l->kv_b.I+3)/4; const uint8_t *weights=l->kv_b.fmt==1?(const uint8_t*)l->kv_b.q8:l->kv_b.q4; @@ -3833,6 +3915,32 @@ static void expert_prefetch(Model *m, int layer, int eid){ /* ---- helper per l'ABSORPTION: accesso per-riga ai QT quantizzati ---- */ /* acc[0..I) += coef * W[row,:] (dequant al volo) */ +/* One fmt=8 block scale, checked before it multiplies anything. + * + * A NaN scale has no safe interpretation: it poisons the whole block and every + * accumulator downstream of it, and this function cannot repair it, so it is + * refused by name with the block that carried it -- the same "refuse rather + * than misread" discipline the format guards below apply. + * + * A ZERO scale is NOT refused: it is valid data. Block scales are amax/448, so + * a genuinely all-zero block (padding, an unused slice) legitimately produces + * zero, and decoding it as zeros is the correct answer. It cannot be confused + * with a decode against an unwritten table the way it can on the GPU, because + * there is no table here to be unwritten -- the CPU decoder reads its e4m3 + * values from a compile-time constant table, which is why the LUT-ready gate + * exists only on the CUDA side. Do not re-add a zero refusal: it would reject + * valid checkpoints. + * + * The check is per BLOCK, not per element: one branch per FP8_BLOCK columns. */ +static float fp8_block_scale(float sc, int64_t blkO, int64_t bi, const char *who){ + if(isnan(sc)){ + fprintf(stderr,"%s: fmt=8 scale block [%lld,%lld] is NaN -- refusing rather than " + "propagate it through the absorb accumulator\n",who,(long long)blkO,(long long)bi); + exit(1); + } + return sc; +} + static void qt_addrow(const QT *t, int row, float coef, float *acc){ int I=t->I; if(t->fmt==0){ const float *w=t->qf+(int64_t)row*I; for(int i=0;i>2]>>((k&3)*2))&3)|(((hi[k>>3]>>(k&7))&1)<<2); acc[base+k]+=cg*(float)((int)u-4); } } return; } + /* fmt=8 (fp8-e4m3-b128, absorb-path support added here): t->s holds ONE f32 scale + * per 128x128 BLOCK (ceil(O/128)*ceil(I/128) entries, block-row-major), not O, and + * t->q4 is NULL for this format -- raw e4m3 bytes live in t->q8 instead, same + * convention as fmt=1 (see the QT struct comment). Mirrors matmul_fp8's (quant.h) + * block-scale indexing exactly: blkO=row/FP8_BLOCK selects the scale row, then one + * scale per FP8_BLOCK-wide slice of I. This is the branch that used to be missing + * -- see the guard below's history note. */ + if(t->fmt==8){ const uint8_t *w=(const uint8_t*)t->q8+(int64_t)row*I; + int64_t nblkI=fp8_nblk(I), blkO=(int64_t)row/FP8_BLOCK; + const float *scl=t->s+blkO*nblkI; + for(int64_t bi=0; bi*FP8_BLOCKI) blen=I-base; + float sc=coef*fp8_block_scale(scl[bi],blkO,bi,"qt_addrow"); + for(int i=base;is[row]) followed by fmt=1 (int8, explicit * branch), fmt=2 (int4 packed, explicit branch), or the tail's own IMPLICIT fmt=3 * (int2 packed, the final unconditional block) -- there was no guard stopping any @@ -3866,20 +3989,16 @@ static void qt_addrow(const QT *t, int row, float coef, float *acc){ * a heap OVERREAD, and the untouched fall-through then misreads t->q4's real E8 * lattice bytes as int2-packed data (same bug SHAPE as #298's CUDA absorb-kernel * fix, and the same one this file's own fmt=4/5 branches above were added to - * dodge -- fmt=6 was simply missed). fmt=8 (fp8-e4m3-b128): t->s holds - * ceil(O/128)*ceil(I/128) per-block floats, not O -- t->s[row] overreads for - * row>=nblk (e.g. a [130,130] tensor has nblk=4, so every row past 3 already reads - * out of bounds), AND t->q4 is NULL for fmt=8 (raw bytes live in t->q8 instead, - * same convention as fmt=1 -- see the QT struct comment), so the fall-through's - * `t->q4+(int64_t)row*((I+3)/4)` dereferences NULL-plus-offset: SIGSEGV, - * reproduced (see the report's proof-of-bite transcript). Refuse loudly instead -- - * this function has no byte-count context of its own to validate against (it only - * ever sees an already-resolved QT), so "unsupported fmt" is the only check - * available, same "refuse rather than misread" discipline qt_resolve_fmt applies - * at load time. */ + * dodge -- fmt=6 was simply missed). fmt=8 previously landed here too (t->s[row] + * overreads past row>=nblk, t->q4 is NULL -> SIGSEGV via the int2 fall-through); + * it now returns above via its own branch and never reaches this guard. Refuse + * loudly instead for anything else -- this function has no byte-count context of + * its own to validate against (it only ever sees an already-resolved QT), so + * "unsupported fmt" is the only check available, same "refuse rather than + * misread" discipline qt_resolve_fmt applies at load time. */ if(t->fmt!=1 && t->fmt!=2 && t->fmt!=3){ fprintf(stderr,"qt_addrow: unsupported fmt=%d for the per-row-scale absorb path " - "(only fmt 1/2/3 reach this point; fmt 0/4/5 are handled above and return " + "(only fmt 1/2/3 reach this point; fmt 0/4/5/8 are handled above and return " "before it) -- refusing rather than misread t->s[row]/t->q4\n", t->fmt); exit(1); } @@ -3929,18 +4048,35 @@ static void qt_matvec_rows(const QT *t, int r0, int n, const float *x, float *y) for(int k=0;k>2]>>((k&3)*2))&3)|(((hi[k>>3]>>(k&7))&1)<<2); acc+=(float)((int)u-4)*x[base+k]; } a+=(double)(acc*sr[g]); } } + /* fmt=8 (fp8-e4m3-b128, absorb-path support added here): per-128x128-BLOCK f32 + * scale, block-row-major (ceil(O/128)*ceil(I/128) entries), raw bytes in t->q8 + * (t->q4 is NULL for this format). Same block-scale indexing as matmul_fp8 + * (quant.h) and qt_addrow's fmt=8 branch above: blkO=row/FP8_BLOCK picks the + * scale row, one scale per FP8_BLOCK-wide slice of I, double-accumulated + * across blocks like matmul_fp8 to avoid unfairly penalizing cross-block + * cancellation (widen-then-multiply, a+=(double)acc*sc -- the same rounding + * as matmul_fp8 and this function's grouped fmt=4 arm; fmt=5's arm rounds + * differently, multiplying in float before widening). */ + else if(t->fmt==8){ const uint8_t *w=(const uint8_t*)t->q8+(int64_t)row*I; + int64_t nblkI=fp8_nblk(I), blkO=(int64_t)row/FP8_BLOCK; + const float *scl=t->s+blkO*nblkI; + for(int64_t bi=0; bi*FP8_BLOCKI) blen=I-base; + float sc=fp8_block_scale(scl[bi],blkO,bi,"qt_matvec_rows"); float acc=0; + for(int i=base;is is a fixed 4-byte tag (t->s[row] overreads for row>0), - * fmt=8's t->s holds per-128x128-block floats (t->s[row] overreads for - * row>=nblk) and t->q4 is NULL for fmt=8 -- both would have silently misread or - * crashed here exactly like qt_addrow did before its own fix; refuse instead. */ + * defect): fmt=6's t->s is a fixed 4-byte tag (t->s[row] overreads for row>0) -- + * it would silently misread or crash here exactly like qt_addrow did before its + * fix; refuse instead. fmt=8 previously fell into this same trap and now has its + * own branch above instead. */ else if(t->fmt==3){ const uint8_t *w=t->q4+(int64_t)row*((I+3)/4); float s=t->s[row]; float acc=0; for(int i=0;i>2]; acc+=((int)((b>>((i&3)*2))&3)-2)*x[i]; } a=acc*s; } else { fprintf(stderr,"qt_matvec_rows: unsupported fmt=%d for the per-row-scale absorb " - "path (only fmt 0/1/2/3/4/5 are handled) -- refusing rather than misread " + "path (only fmt 0/1/2/3/4/5/8 are handled) -- refusing rather than misread " "t->s[row]/t->q4\n", t->fmt); exit(1); } @@ -3948,7 +4084,9 @@ static void qt_matvec_rows(const QT *t, int r0, int n, const float *x, float *y) } } static int g_absorb=-1; +#if defined(COLI_METAL) || !defined(COLIBRI_NO_MAIN) static int g_metal_prefill=0; /* default 0: S>4 prefill attention stays on the CPU (bit-exact). COLI_METAL_PREFILL=1 opts it onto the GPU (~4x, near-tie divergence — see docs/metal.md, #622) */ +#endif /* KV8=1: cache latente Lc/Rc in fp8 e4m3 + scala f32 per riga (~4x meno RAM del f32). * CPU-only in this PR — sui percorsi CUDA/Metal che leggono righe f32 si spegne da * solo (guardie !g_kv8), e forza COLI_CUDA_PIPE=0 (il pipe-prefill legge righe f32). @@ -4776,6 +4914,13 @@ static void attention_rows(Model *m, Layer *l, int layer, float *x, int S, int p } else { const float *Lt=coli_kv_row(ks->Lc[layer],t,kvl); const float *kr=coli_kv_row(ks->Rc[layer],t,c->qk_rope); + if(exact_verify_on()&&g_spec_live){ + /* #689 exact verify: one exact accumulator over BOTH partial dots, rounded once */ + exd_acc ea; exd_init(&ea); + for(int i=0;iqk_rope;d++) exd_add_ff(&ea,qr[d],kr[d]); + a=exd_finish(&ea); + } else { /* MLA-absorb score: dot(qabs, Lt) + dot(qr, kr). #442: the qabs·Lt * reduction is the hot f32 dot at this site (kvl=512 on GLM-5.2, * runs nt times per (s,h), grows with context). SIMD-ify under @@ -4798,10 +4943,21 @@ static void attention_rows(Model *m, Layer *l, int layer, float *x, int S, int p for(;iqk_rope;d++) a+=qr[d]*kr[d]; } + } sc[jj]=a*c->attn_scale; } softmax(sc,nt); float clat[512]; memset(clat,0,kvl*sizeof(float)); + if(exact_verify_on()&&g_spec_live&&!tq1&&!g_tq&&!g_kv8){ + /* #689 exact verify: clat[i] = sum_t sc[t]*Lt[i] as an exact dot over t per column + * (transposed walk: cache-unfriendly, verify rows only). NOT taken on the quantised + * KV paths (tq1 / TQ / kv8): those keep the float context dot, so COLI_EXACT_VERIFY + * does not provide exactness for the context dot with a quantised cache (README). */ + for(int i=0;iLc[layer],t,kvl)[i]); } + clat[i]=exd_finish(&ea); } + } else for(int jj=0;jjpin[layer]; + for(int z=0;znpin[layer];z++) if(P[z].eid==eid){ resident=1; break; } + if(!resident){ ESlot *Sl=m->ecache[layer]; int nn=m->ecn[layer]; + for(int z=0;z= tau. Each position's weight is tested independently — this + * is the gate the +2.9% ppl measurement was taken under. */ + for(int s=0;s=g_degrade_tau){ + int e=idxs[(int64_t)s*K+kk]; + for(int j=0;jbw){ bw=wv; be=idxs[(int64_t)s*K+kk]; } + } + if(be<0) be=idxs[(int64_t)s*K]; + seen[be]=1; + for(int j=0;jdst[e->n++]=t; } +typedef struct { EmitStore tokens; int vocab, finite; } OracleEmit; +static void emit_oracle(int t, const float *lo, void *ud){ + OracleEmit *e=(OracleEmit*)ud; + if(!oracle_logits_finite(lo,e->vocab)){ + fprintf(stderr,"[ORACLE] non-finite logits at generated token %d\n",e->tokens.n); + e->finite=0; + } + emit_store(t,lo,&e->tokens); +} /* emit callback: detokenizza e stampa in streaming (chat/run), con heartbeat */ typedef struct { Tok *T; Model *m; double t0; int count; int quiet; } EmitStream; static void emit_stream(int t, const float *lo, void *ud){ @@ -7827,8 +8065,9 @@ static void dump_top5_logits(int pos, const float *lo, int V, int expected, int for(int k=0;k<5&&idx[k]>=0;k++) fprintf(stderr," %d:%.5f", idx[k], (double)val[k]); fprintf(stderr,"\n"); } -static void forward_all(Model *m, const int *ids, int S, int *pred, const int *ref){ +static int forward_all(Model *m, const int *ids, int S, int *pred, const int *ref){ Cfg *c=&m->c; int D=c->hidden; + int finite=1; int dbg = ref && getenv("DEBUG_LOGITS"); kv_alloc(m,S); float *x=falloc((int64_t)S*D); @@ -7839,11 +8078,16 @@ static void forward_all(Model *m, const int *ids, int S, int *pred, const int *r for(int s=0;sfinal_norm, D, c->eps); /* heap row (#183) */ matmul_qt(lo, row, &m->lm_head, 1); + if(!oracle_logits_finite(lo,c->vocab)){ + fprintf(stderr,"[ORACLE] non-finite logits at teacher-forcing position %d\n",s); + finite=0; pred[s]=-1; continue; + } int best=0; float bv=lo[0]; for(int i=1;ivocab;i++) if(lo[i]>bv){bv=lo[i];best=i;} pred[s]=best; if(dbg && pred[s]!=ref[s]) dump_top5_logits(s, lo, c->vocab, ref[s], pred[s]); } free(x); free(lo); free(row); + return finite; } /* log-prob (log-softmax) del token target dato il vettore di logit; *am=1 se e' l'argmax */ @@ -8007,12 +8251,14 @@ static void run_ablate_score(Model *m, const char *path){ free(ln); free(ids); free(x); free(lo); free(row); fclose(f); } -static void generate(Model *m, const int *prompt, int np, int n_new, int *out){ +static int generate(Model *m, const int *prompt, int np, int n_new, int *out, int *finite){ kv_alloc(m,np+n_new+g_draft+2); for(int i=0;ic.vocab,1}; + int emitted=spec_decode(m,out,np,n_new,-1,logit,emit_oracle,&es,NULL,NULL); + *finite=es.finite; + return emitted; } static void profile_print(Model *m, double elapsed){ @@ -8389,6 +8635,22 @@ static void run_text(Model *m, const char *snap, const char *prompt, int ngen){ printf(" | EXPERT_BUDGET=%d (dropped %lld experts, ~%.1f GB I/O saved)", g_expert_budget, (long long)g_budget_dropped, g_budget_dropped*18.9e6/1e9); if(g_budget_rescued) printf(" [%lld rescued: budget too tight, position would have had 0 routed experts]", (long long)g_budget_rescued); } + if(g_degrade_zero){ + printf(" | DEGRADE_ZERO tau=%.3f (zeroed %lld miss slots", g_degrade_tau, (long long)g_degrade_dropped); + if(g_degrade_dropped>0){ + /* top-3 layers by drop count */ + int top[3]={-1,-1,-1}; int64_t tv[3]={0,0,0}; + for(int i=0;i<512;i++){ + int64_t v=g_degrade_dropped_by_layer[i]; if(!v) continue; + if(v>tv[0]){tv[2]=tv[1];top[2]=top[1];tv[1]=tv[0];top[1]=top[0];tv[0]=v;top[0]=i;} + else if(v>tv[1]){tv[2]=tv[1];top[2]=top[1];tv[1]=v;top[1]=i;} + else if(v>tv[2]){tv[2]=v;top[2]=i;} + } + printf("; top layers:"); + for(int k=0;k<3&&top[k]>=0;k++) printf(" L%d:%lld",top[k],(long long)tv[k]); + } + printf(")"); + } printf("\n"); printf("speculation: %.2f tokens/forward (%llu forwards per %llu tokens) | MTP acceptance %.0f%% (%llu/%llu)\n", m->n_fw?(double)m->n_emit/m->n_fw:1.0, (unsigned long long)m->n_fw, (unsigned long long)m->n_emit, @@ -9501,13 +9763,6 @@ static void run_serve(Model *m, const char *snap){ free(ctx); m->kv=NULL; m->Lc=m->Rc=m->Ic=NULL; m->Lc8=m->Rc8=NULL; m->Lsc=m->Rsc=NULL; m->kv_start=NULL; m->max_t=0; } -static int *read_arr(jval*o,const char*k,int*n){ - jval*a=json_get(o,k); - if(!a){ *n=0; return NULL; } - int*r=malloc(a->len*sizeof(int)); - if(!r){ fprintf(stderr,"OOM read_arr\n"); exit(1); } - for(int i=0;ilen;i++) r[i]=(int)a->kids[i]->num; *n=a->len; return r; } - /* telemetry, stats, usage persistence — moved to telemetry.h */ #ifdef COLI_VULKAN @@ -10908,6 +11163,19 @@ static int coli_env_on(const char *name) #ifndef COLIBRI_NO_MAIN int main(int argc, char **argv){ + int strict=coli_env_on("ORACLE_STRICT"); + if(strict){ + const char *modes[]={"REPLAY","CONSIST","SERVE","SCORE","ABLATE_SCORE","EXPERT_WORKER", + "I4_ACC512_TEST","I3_AVX512_TEST"}; + for(size_t i=0;i16) g_pilot_nw=16; g_pilot_evict_guard = getenv("PILOT_EVICT_GUARD")?atoi(getenv("PILOT_EVICT_GUARD")):1; /* 0 = old LRU eviction (A/B) */ + g_degrade_zero = getenv("DEGRADE_ZERO")?atoi(getenv("DEGRADE_ZERO")):0; + g_degrade_tau = getenv("DEGRADE_TAU") ?atof(getenv("DEGRADE_TAU")) :0.03f; + if(g_degrade_tau<=0.f||g_degrade_tau>1.f) g_degrade_tau=0.03f; /* clamp to sane range */ + if(g_degrade_zero) + fprintf(stderr,"[DEGRADE] zero-fill ON, tau=%.3f (approximate mode: miss slots with per-position gate weight < tau are never loaded)\n",g_degrade_tau); g_disk_split = getenv("DISK_SPLIT")?atoi(getenv("DISK_SPLIT")):0; /* 1 = split dei disk load nelle stats */ g_pipe = getenv("PIPE")?atoi(getenv("PIPE")): #ifdef _WIN32 @@ -11684,13 +11957,23 @@ int main(int argc, char **argv){ } /* altrimenti: validazione contro l'oracolo (ref_glm.json) */ + /* Diagnostic modes take precedence over TF and do not read predictions. */ + int teacher_forcing=getenv("TF")!=NULL && !getenv("REPLAY") && !getenv("CONSIST"); const char *refpath=getenv("REF")?getenv("REF"):"ref_glm.json"; char *b=cfg_slurp(refpath); - if(!b){ fprintf(stderr,"%s: cannot read oracle file (missing, unreadable, short, or > %lld bytes)\n",refpath,(long long)CFG_MAX_BYTES); return 1; } - char *ar=NULL; jval *ref=json_parse(b,&ar); - int np=0,nfull=0; int *prompt=read_arr(ref,"prompt_ids",&np); int *full=read_arr(ref,"full_ids",&nfull); - if(!prompt||!full||np<1||nfull %lld bytes)\n",refpath,(long long)CFG_MAX_BYTES); return 1; } + OracleRef ref; + int valid=oracle_ref_parse(b,m.c.vocab,teacher_forcing,&ref); + free(b); + if(!valid) return 1; + int np=ref.np,nfull=ref.nfull; int *prompt=ref.prompt,*full=ref.full; int n_new=nfull-np; + int tf_allowed=0; + if(strict && teacher_forcing && + !oracle_tf_allowance(getenv("ORACLE_TF_MAX_MISMATCHES"),nfull,&tf_allowed)){ + fprintf(stderr,"[ORACLE] ORACLE_TF_MAX_MISMATCHES must be an integer in [0,%d)\n",nfull); + oracle_ref_free(&ref); return 1; + } /* L'oracolo (ref_glm.json in repo) e' del modello TINY: contro il 744B da' 0/20 * garantito su OGNI piattaforma (prompt-token tiny = spazzatura per il modello vero). * Non e' un bug del motore — vedi #76. */ @@ -11707,25 +11990,29 @@ int main(int argc, char **argv){ " Nessun PROMPT: modo auto-validazione, ma ref_glm.json e' l'oracolo del modello TINY\n" " (token max %d, il tuo vocab e' %d). Usa PROMPT=... per generare davvero (vedi sopra).\n", maxid, m.c.vocab, maxid, m.c.vocab); - return 1; + oracle_ref_free(&ref); return 1; } } if(getenv("REPLAY")){ run_replay(&m,full,nfull,np); if(stats) stats_dump(&m,stats); + oracle_ref_free(&ref); return 0; } if(getenv("CONSIST")){ run_consist(&m,full,nfull,np); if(stats) stats_dump(&m,stats); + oracle_ref_free(&ref); return 0; } - if(getenv("TF")){ - int *tf=read_arr(ref,"tf_pred",&(int){0}); - int *pred=malloc(nfull*sizeof(int)); double tt=now_s(); - forward_all(&m, full, nfull, pred, tf); double tdt=now_s()-tt; + if(teacher_forcing){ + int *tf=ref.tf; + int *pred=malloc((size_t)nfull*sizeof(int)); + if(!pred){ oracle_ref_free(&ref); return 1; } + double tt=now_s(); + int finite=forward_all(&m, full, nfull, pred, tf); double tdt=now_s()-tt; int ok=0; for(int i=0;itf_allowed); } - int *out=malloc((np+n_new)*sizeof(int)); + int *out=malloc((size_t)nfull*sizeof(int)); + if(!out){ oracle_ref_free(&ref); return 1; } ProfBase pb; prof_base(&m,&pb); - double t=now_s(); generate(&m,prompt,np,n_new,out); double dt=now_s()-t; + int finite=1; + double t=now_s(); int emitted=generate(&m,prompt,np,n_new,out,&finite); double dt=now_s()-t; int match=0; printf("\nReference (oracle): "); for(int i=np;i +#include #endif #include -static inline double compat_mem_available_gb(void){ + +/* Total AND available in one call, one pass. + * + * "Available" alone was consolidated here by #1375 because there were two + * copies of it and one was wrong. "Total" is now in the same position: every + * caller that wants it hand-rolls its own #ifdef ladder (telemetry.h uses + * sysctl hw.memsize on macOS, olmoe.c uses sysconf(_SC_PHYS_PAGES), glm53.c + * read /proc/meminfo unconditionally), and those definitions do not agree. + * A budget computed from a total and an available that came from two + * different definitions is not a budget, it is a coincidence. + * + * One pass also matters on Linux specifically: MemTotal and MemAvailable are + * two lines of the same file, and reading it twice to get them is both a + * second open and a second chance to read a file that changed underneath. + * + * 0 means "not measurable" for either field; the caller decides the fallback. */ +static inline void compat_meminfo_gb(double *total_gb, double *avail_gb){ + double total = 0, avail = 0; #ifdef __APPLE__ + uint64_t memsize = 0; size_t len = sizeof memsize; + if(sysctlbyname("hw.memsize", &memsize, &len, NULL, 0) == 0) total = (double)memsize / 1e9; mach_msg_type_number_t cnt = HOST_VM_INFO64_COUNT; vm_statistics64_data_t vm; - if(host_statistics64(mach_host_self(), HOST_VM_INFO64, (host_info64_t)&vm, &cnt) != KERN_SUCCESS) return 0; - return ((double)vm.free_count + (double)vm.inactive_count + (double)vm.purgeable_count) - * (double)sysconf(_SC_PAGESIZE) / 1e9; + if(host_statistics64(mach_host_self(), HOST_VM_INFO64, (host_info64_t)&vm, &cnt) == KERN_SUCCESS) + avail = ((double)vm.free_count + (double)vm.inactive_count + (double)vm.purgeable_count) + * (double)sysconf(_SC_PAGESIZE) / 1e9; #elif defined(_WIN32) MEMORYSTATUSEX msx = {0}; msx.dwLength = sizeof(msx); - if(!GlobalMemoryStatusEx(&msx)) return 0; - double phys = (double)msx.ullAvailPhys / 1e9; - double commit = (double)msx.ullAvailPageFile / 1e9; - return commit > 0 && commit < phys ? commit : phys; + if(GlobalMemoryStatusEx(&msx)){ + total = (double)msx.ullTotalPhys / 1e9; + double phys = (double)msx.ullAvailPhys / 1e9; + double commit = (double)msx.ullAvailPageFile / 1e9; + avail = commit > 0 && commit < phys ? commit : phys; + } #else - FILE *f = fopen("/proc/meminfo", "r"); if(!f) return 0; - char ln[256]; double kb = 0; - while(fgets(ln, sizeof ln, f)) if(sscanf(ln, "MemAvailable: %lf", &kb) == 1) break; - fclose(f); return kb / 1e6; + FILE *f = fopen("/proc/meminfo", "r"); + if(f){ + char ln[256]; double kb; + /* /proc/meminfo's "kB" is KiB (1024 B), so a GB is kb*1024/1e9, not + * kb/1e6. The old kb/1e6 understated by 2.3% -- harmless while the + * number was only ever compared against itself, but glm53 now weighs + * it against a model size computed from byte counts (/1e9, true GB), + * and a budget that subtracts true GB from understated GB is wrong in + * the direction that matters: it hands back less than it should. */ + /* MemTotal precedes MemAvailable in /proc/meminfo, but do not rely on + * the order: stop only once both have been seen. */ + while((total == 0 || avail == 0) && fgets(ln, sizeof ln, f)){ + if(total == 0 && sscanf(ln, "MemTotal: %lf", &kb) == 1){ total = kb * 1024.0 / 1e9; continue; } + if(avail == 0 && sscanf(ln, "MemAvailable: %lf", &kb) == 1) avail = kb * 1024.0 / 1e9; + } + fclose(f); + } #endif + if(total_gb) *total_gb = total; + if(avail_gb) *avail_gb = avail; +} + +static inline double compat_mem_available_gb(void){ + double avail = 0; + compat_meminfo_gb(NULL, &avail); + return avail; } #endif /* COMPAT_H */ diff --git a/c/deepseek_v4.c b/c/deepseek_v4.c index 66a831415..bc8f38622 100644 --- a/c/deepseek_v4.c +++ b/c/deepseek_v4.c @@ -1368,7 +1368,7 @@ static int build_runtime_plan(ColiV4Engine *engine, uint64_t maximum_layer = 0, dense_total = 0; for (int layer = 0; layer < config->num_hidden_layers; layer++) { ColiDeepSeekV4LayerPlan layer_plan; - ColiDeepSeekV4LayerStats stats; + ColiDeepSeekV4LayerStats stats = {0}; if (coli_v4_layer_plan(&layer_plan, config, layer, error, error_size) || coli_v4_layer_validate(&layer_plan, index, &stats, @@ -8241,8 +8241,9 @@ static size_t hot_slot_index(const V4ExpertStoreState *state, * * FLOCK-packed checkpoints store [scales][weights] contiguously and need one * request. Standard HF checkpoints keep the ranges apart: weights use direct - * I/O while the much smaller scales use buffered pread. Any direct-I/O error - * falls back to the exact buffered path. */ + * I/O while the much smaller scales use buffered pread. REAP-style + * per_matrix records issue one window per scale/weight segment. Any + * direct-I/O error falls back to the exact buffered path. */ static uint64_t v4_direct_reads; static uint64_t v4_direct_flock_reads; static uint64_t v4_direct_payload_bytes; @@ -8298,33 +8299,87 @@ static int v4_read_direct_window(const V4ExpertStoreState *state, int shard, return 0; } +/* Direct window into an interior slab offset. v4_read_direct_window bounces + * at slab[0], so a second per_matrix segment would clobber earlier bytes. */ +static int v4_read_direct_copy(const V4ExpertStoreState *state, int shard, + int rep, unsigned char *destination, + uint64_t offset, size_t length) { + if (!destination) return -1; + if (!length) return 0; + if (length > SIZE_MAX - 8192u) return -1; + unsigned char *bounce = NULL; + if (posix_memalign((void **)&bounce, 4096, length + 8192u)) return -1; + int result = v4_read_direct_window(state, shard, rep, bounce, offset, + length, 0); + if (!result) memcpy(destination, bounce, length); + compat_aligned_free(bounce); + return result; +} + +static int v4_try_direct_segment(V4ExpertStoreState *state, int shard, int rep, + V4ExpertSlot *slot, uint64_t dest, + uint64_t offset, uint64_t bytes) { + if (!slot->aligned_slab || + !coli_st_streaming_direct_available_rep(state->index, shard, rep)) + return -1; + size_t length = (size_t)bytes; + if (dest == 0) + return v4_read_direct_window(state, shard, rep, slot->slab, offset, + length, 0); + return v4_read_direct_copy(state, shard, rep, slot->slab + dest, offset, + length); +} + +static int v4_read_per_matrix_segment(V4ExpertStoreState *state, int shard, + int rep, V4ExpertSlot *slot, + uint64_t dest, uint64_t offset, + uint64_t bytes, int *used_direct, + int *used_fallback) { + if (!v4_try_direct_segment(state, shard, rep, slot, dest, offset, bytes)) { + *used_direct = 1; + return 0; + } + if (slot->aligned_slab && + coli_st_streaming_direct_available_rep(state->index, shard, rep)) + *used_fallback = 1; + return coli_st_read_at_rep(state->index, shard, rep, offset, (size_t)bytes, + slot->slab + dest); +} + static int v4_read_expert_record(V4ExpertStoreState *state, const V4ExpertRecord *record, V4ExpertSlot *slot, int rep) { if (record->per_matrix) { - int direct_available = slot->aligned_slab && - coli_st_streaming_direct_available_rep(state->index, record->m_scale_shard[0], rep); - if (direct_available) - __atomic_fetch_add(&v4_direct_fallbacks, UINT64_C(1), - __ATOMIC_RELAXED); + int used_direct = 0; + int used_fallback = 0; uint64_t scale_cursor = 0; for (int matrix = 0; matrix < V4_MATRIX_COUNT; matrix++) { - if (coli_st_read_at_rep(state->index, record->m_scale_shard[matrix], rep, - record->m_scale_offset[matrix], - (size_t)record->m_scale_bytes[matrix], - slot->slab + scale_cursor) != 0) + if (v4_read_per_matrix_segment( + state, record->m_scale_shard[matrix], rep, slot, + scale_cursor, record->m_scale_offset[matrix], + record->m_scale_bytes[matrix], &used_direct, + &used_fallback) != 0) return -1; scale_cursor += record->m_scale_bytes[matrix]; } uint64_t weight_cursor = scale_cursor; for (int matrix = 0; matrix < V4_MATRIX_COUNT; matrix++) { - if (coli_st_read_at_rep(state->index, record->m_weight_shard[matrix], rep, - record->m_weight_offset[matrix], - (size_t)record->m_weight_bytes[matrix], - slot->slab + weight_cursor) != 0) + if (v4_read_per_matrix_segment( + state, record->m_weight_shard[matrix], rep, slot, + weight_cursor, record->m_weight_offset[matrix], + record->m_weight_bytes[matrix], &used_direct, + &used_fallback) != 0) return -1; weight_cursor += record->m_weight_bytes[matrix]; } + if (used_direct && !used_fallback) { + __atomic_fetch_add(&v4_direct_reads, UINT64_C(1), __ATOMIC_RELAXED); + __atomic_fetch_add(&v4_direct_payload_bytes, record->record_bytes, + __ATOMIC_RELAXED); + } else if (used_fallback) { + __atomic_fetch_add(&v4_direct_fallbacks, UINT64_C(1), + __ATOMIC_RELAXED); + } return 0; } int direct_available = slot->aligned_slab && @@ -8782,6 +8837,37 @@ int coli_v4_test_expert_slot_index(ColiExpertStore *store, ColiExpertKey key) { pthread_mutex_unlock(&state->mutex); return result; } + +void coli_v4_test_reset_direct_io_stats(void) { + __atomic_store_n(&v4_direct_reads, 0, __ATOMIC_RELAXED); + __atomic_store_n(&v4_direct_flock_reads, 0, __ATOMIC_RELAXED); + __atomic_store_n(&v4_direct_payload_bytes, 0, __ATOMIC_RELAXED); + __atomic_store_n(&v4_direct_fallbacks, 0, __ATOMIC_RELAXED); +} + +uint64_t coli_v4_test_direct_reads(void) { + return __atomic_load_n(&v4_direct_reads, __ATOMIC_RELAXED); +} + +uint64_t coli_v4_test_direct_fallbacks(void) { + return __atomic_load_n(&v4_direct_fallbacks, __ATOMIC_RELAXED); +} + +int coli_v4_test_force_streaming_direct(ColiExpertStore *store) { + if (!store || !store->state) return -1; + V4ExpertStoreState *state = store->state; + if (!state->index) return -1; + int enabled = 0; + for (int i = 0; i < state->index->nfd; i++) { + if (state->index->dfds[i] < 0 && state->index->fds[i] >= 0) { + int twin = dup(state->index->fds[i]); + if (twin < 0) return -1; + state->index->dfds[i] = twin; + } + if (state->index->dfds[i] >= 0) enabled = 1; + } + return enabled ? 0 : -1; +} #endif /* Let the layer currently sweeping a batched CPU prefill borrow the complete @@ -11635,6 +11721,7 @@ int coli_v4_prompt_build(char **output, size_t *output_length, #include "json.h" #include "native_quant.h" #include "serve_codec.h" +#include "decode_batch.h" /* coli_logprob_tail: the numeric channel's tail, same bytes as the other engines */ #include "tok.h" static int load_embedding(float *state, const ColiSafetensorsIndex *index, @@ -11772,55 +11859,39 @@ static int head_argmax(ColiV4Engine *engine, const float *hidden, g_v4_prof_head_s += spec_now() - t0; return result; } -static int head_argmax_impl(ColiV4Engine *engine, const float *hidden, +/* Every head score of one hidden row, in vocabulary order. head_argmax used + * to run this matmul and keep only the maximum; the numeric channel (SUBMIT + * logprobs=k, docs/brio.md) needs the whole row, so the row is computed here + * once and the argmax is a scan over it. Same head_bf16_dot per row, same scan + * order: the token picked and its logit do not change. */ +static int head_scores_impl(ColiV4Engine *engine, const float *hidden, const ColiSafetensorsIndex *index, - const ColiDeepSeekV4Config *config, - int *best_token, float *best_logit) { + const ColiDeepSeekV4Config *config, float *scores) { const ColiSafetensorsTensor *head = coli_st_find(index, "head.weight"); int d = config->hidden_size, vocab = config->vocab_size; - if (!head || head->dtype != COLI_ST_BF16 || d < 1 || vocab < 1) + if (!head || head->dtype != COLI_ST_BF16 || d < 1 || vocab < 1 || !scores) return -1; int shard = coli_st_tensor_shard(index, head); size_t resident_bytes = (size_t)vocab * (size_t)d * sizeof(uint16_t); const uint16_t *resident = coli_v4_head_cache_data( engine, shard, (uint64_t)head->off, resident_bytes); - /* The normal V4 memory plan keeps the BF16 head resident. Compute all * rows in one OpenMP team directly from that allocation: the old tiled * path copied the complete ~1 GiB head and created ~2,000 teams per token. * Each row retains the same scalar accumulation order and the final scan * retains vocabulary order, so logits/tie-breaking do not change. */ if (resident) { - float *scores = malloc((size_t)vocab * sizeof(*scores)); - if (!scores) return -1; #pragma omp parallel for schedule(static) for (int row = 0; row < vocab; row++) { const uint16_t *weight = resident + (size_t)row * d; scores[row] = head_bf16_dot(weight, hidden, d); } - int winner = -1; - float maximum = -FLT_MAX; - for (int row = 0; row < vocab; row++) - if (scores[row] > maximum) { - maximum = scores[row]; - winner = row; - } - free(scores); - *best_token = winner; - *best_logit = maximum; - return winner < 0 ? -1 : 0; + return 0; } - /* Low-memory fallback: stream small row tiles exactly as before. */ enum { ROWS = 64 }; uint16_t *raw = malloc((size_t)ROWS * d * sizeof(*raw)); - float *scores = malloc((size_t)ROWS * sizeof(*scores)); - if (!raw || !scores) { - free(scores); free(raw); - return -1; - } - int winner = -1; - float maximum = -FLT_MAX; + if (!raw) return -1; for (int start = 0; start < vocab; start += ROWS) { int rows = vocab - start < ROWS ? vocab - start : ROWS; size_t bytes = (size_t)rows * d * sizeof(*raw); @@ -11828,26 +11899,54 @@ static int head_argmax_impl(ColiV4Engine *engine, const float *hidden, engine, index, shard, (uint64_t)head->off + (uint64_t)start * d * sizeof(*raw), bytes, raw)) { - free(scores); free(raw); + free(raw); return -1; } #pragma omp parallel for for (int row = 0; row < rows; row++) { const uint16_t *weight = raw + (size_t)row * d; - scores[row] = head_bf16_dot(weight, hidden, d); + scores[start + row] = head_bf16_dot(weight, hidden, d); } - for (int row = 0; row < rows; row++) - if (scores[row] > maximum) { - maximum = scores[row]; - winner = start + row; - } } - free(scores); free(raw); + free(raw); + return 0; +} +/* First maximum in vocabulary order: the tie-break head_argmax always had. */ +static int head_scores_argmax(const float *scores, int vocab, + int *best_token, float *best_logit) { + int winner = -1; + float maximum = -FLT_MAX; + for (int row = 0; row < vocab; row++) + if (scores[row] > maximum) { + maximum = scores[row]; + winner = row; + } *best_token = winner; *best_logit = maximum; return winner < 0 ? -1 : 0; } - +static int head_argmax_impl(ColiV4Engine *engine, const float *hidden, + const ColiSafetensorsIndex *index, + const ColiDeepSeekV4Config *config, + int *best_token, float *best_logit) { + int vocab = config->vocab_size; + if (vocab < 1) return -1; + float *scores = malloc((size_t)vocab * sizeof(*scores)); + if (!scores) return -1; + int result = head_scores_impl(engine, hidden, index, config, scores); + if (!result) result = head_scores_argmax(scores, vocab, best_token, best_logit); + free(scores); + return result; +} +/* The whole row, under the same head-time meter as head_argmax. */ +static int head_scores(ColiV4Engine *engine, const float *hidden, + const ColiSafetensorsIndex *index, + const ColiDeepSeekV4Config *config, float *scores) { + double t0 = spec_now(); + int result = head_scores_impl(engine, hidden, index, config, scores); + g_v4_prof_head_s += spec_now() - t0; + return result; +} static int head_argmax_batch(ColiV4Engine *engine, const float *hidden, const ColiSafetensorsIndex *index, const ColiDeepSeekV4Config *config, int batch, @@ -12651,6 +12750,8 @@ static void session_free_attention(ColiV4Session *session) { void coli_v4_session_destroy(ColiV4Session *session) { if (!session) return; kv_prefix_free(&session->fed); + free(session->pin_ids); free(session->pin_scores); + free(session->echo_hidden); free(session->echo_scores); session_free_attention(session); session_free_buffers(session); if (session->tokenizer_ready) { @@ -13135,7 +13236,8 @@ int coli_v4_session_generate(ColiV4Session *session, ColiV4SessionGenerateStats *stats_out, char *error, size_t error_size) { if (!session || !session->engine || !prompt || !options || - options->max_new_tokens < 1) { + (options->max_new_tokens < 1 && + !(options->max_new_tokens == 0 && options->logprobs > 0))) { if (error && error_size) snprintf(error, error_size, "invalid V4 session generate arguments"); return -1; @@ -13243,6 +13345,34 @@ int coli_v4_session_generate(ColiV4Session *session, fprintf(stderr, "[PREFIX] hint boundary at %d tokens\n", ckpt_at); } session->prefix_reused = reuse; + /* The numeric channel (docs/brio.md). Scratch sized to the head, kept on + * the session so every early return below leaves nothing behind. */ + const int vocab = config->vocab_size; + const int echo = options->logprobs > 0 && options->on_echo != NULL; + const int want_scores = echo || options->pin || options->on_scores != NULL; + if (want_scores && (!session->echo_hidden || !session->echo_scores)) { + free(session->echo_hidden); + free(session->echo_scores); + session->echo_hidden = malloc((size_t)config->hidden_size * sizeof(float)); + session->echo_scores = malloc((size_t)vocab * sizeof(float)); + if (!session->echo_hidden || !session->echo_scores) { + if (error && error_size) + snprintf(error, error_size, "out of memory for the logprob channel"); + return -1; + } + } + /* Position `reuse` is the first fresh token, and its predictor lives in + * the state we continue from, which nothing below recomputes. When that + * state is the pinned prompt end, its scores were kept for exactly this: a + * closed-set caller pins the prompt, then asks about each option, and the + * option's first token is usually its only one. Any other reuse has no + * predictor to report; the caller sees the position missing, as with the + * other engines. */ + if (echo && reuse > 0 && reuse < prompt_count && session->pin_scores && + session->pin_len == reuse && + !memcmp(session->pin_ids, session->prompt_ids, (size_t)reuse * sizeof(int))) + options->on_echo(options->scores_user_data, reuse, + session->prompt_ids[reuse], session->pin_scores, vocab); if (reuse && getenv("V4_PREFIX_LOG")) fprintf(stderr, "[PREFIX] reusing %d of %d prompt tokens\n", reuse, prompt_count); @@ -13313,6 +13443,25 @@ int coli_v4_session_generate(ColiV4Session *session, } kv_prefix_record(&session->fed, session->prompt_ids + done_upto, done_upto, seg); + /* Read-out of the prefill: row `item` of this segment is position + * done_upto+item and predicts the token at the next one. One head pass + * per row, paid only by the requests that opened the channel. */ + if (echo) { + for (int item = 0; item < seg; item++) { + int at = done_upto + item + 1; + if (at >= prompt_count) break; + if (final_hidden(session->echo_hidden, state + (size_t)item * hd, + index, config, error, error_size) || + head_scores(engine, session->echo_hidden, index, config, + session->echo_scores)) { + kv_prefix_taint(&session->fed); + return -1; + } + options->on_echo(options->scores_user_data, at, + session->prompt_ids[at], session->echo_scores, + vocab); + } + } done_upto += seg; session->fed.len = done_upto; tail_rows = seg; @@ -13346,10 +13495,36 @@ int coli_v4_session_generate(ColiV4Session *session, int current = 0; float current_logit = 0.0f; if (final_hidden(hidden, last, index, config, error, error_size) || - head_argmax(engine, hidden, index, config, ¤t, ¤t_logit)) { + (want_scores + ? (head_scores(engine, hidden, index, config, session->echo_scores) || + head_scores_argmax(session->echo_scores, vocab, ¤t, + ¤t_logit)) + : head_argmax(engine, hidden, index, config, ¤t, + ¤t_logit))) { kv_prefix_taint(&session->fed); return -1; } + if (options->pin) { + /* Keep what the snapshot cannot: the scores at the prompt end. The + * attention state goes to a v4_ckpt slot regardless of the size gate + * above: a pinned prompt is short by nature (a document and a + * question) and is about to be extended by every option. */ + int *ids = realloc(session->pin_ids, (size_t)prompt_count * sizeof(int)); + float *keep = realloc(session->pin_scores, (size_t)vocab * sizeof(float)); + if (ids) session->pin_ids = ids; + if (keep) session->pin_scores = keep; + if (ids && keep) { + memcpy(session->pin_ids, session->prompt_ids, + (size_t)prompt_count * sizeof(int)); + memcpy(session->pin_scores, session->echo_scores, + (size_t)vocab * sizeof(float)); + session->pin_len = prompt_count; + } else { + session->pin_len = 0; /* an optimisation, never an error */ + } + if (v4_ckpt_min_tokens() && !v4_ckpt_have(session->prompt_ids, prompt_count)) + v4_ckpt_store(session, prompt_count, 1); + } /* The prompt is in the attention state from here on; record it before the * decode loop so a failure mid-generation still leaves fed describing what * was actually fed. */ @@ -13357,10 +13532,19 @@ int coli_v4_session_generate(ColiV4Session *session, session->fed.len = prompt_count; int generated_count = 0; int last_processed = prompt_count - 1; - generated[generated_count++] = current; - int done = session_emit_token(session, on_token, user_data, current, + /* max_new == 0 is the read-only request of the numeric channel: the + * prompt is in the state, its read-out went through on_echo, nothing is + * generated and `done` skips the loop; the tail then reports zero. */ + int done = 1; + if (max_new > 0) { + generated[generated_count++] = current; + if (options->on_scores) + options->on_scores(options->scores_user_data, last_processed, current, + session->echo_scores, vocab); + done = session_emit_token(session, on_token, user_data, current, current_logit, last_processed, generated_count, options->stop_at_sentence); + } double first_at = spec_now(); int draft_limit = getenv("V4_DRAFT") ? atoi(getenv("V4_DRAFT")) : 0; @@ -13372,7 +13556,11 @@ int coli_v4_session_generate(ColiV4Session *session, while (!done && generated_count < max_new) { int remaining = max_new - generated_count; - if (!options->no_dspark && !session->spec_disabled && remaining >= 3) { + /* A draft block accepts several tokens from one target pass and has + * no per-token scores to report, so the numeric channel takes the + * plain path: same greedy tokens, one head row each. */ + if (!options->no_dspark && options->logprobs <= 0 && + !session->spec_disabled && remaining >= 3) { int inputs[25] = {0}, drafts[24] = {0}; int predictions[25] = {0}; float logits[25] = {0}; @@ -13590,12 +13778,20 @@ int coli_v4_session_generate(ColiV4Session *session, session->state = state; session->next = next; if (final_hidden(hidden, state, index, config, error, error_size) || - head_argmax(engine, hidden, index, config, ¤t, ¤t_logit)) { + (options->on_scores + ? (head_scores(engine, hidden, index, config, session->echo_scores) || + head_scores_argmax(session->echo_scores, vocab, ¤t, + ¤t_logit)) + : head_argmax(engine, hidden, index, config, ¤t, + ¤t_logit))) { kv_prefix_taint(&session->fed); return -1; } last_processed = position; generated[generated_count++] = current; + if (options->on_scores) + options->on_scores(options->scores_user_data, last_processed, current, + session->echo_scores, vocab); done = session_emit_token(session, on_token, user_data, current, current_logit, last_processed, generated_count, @@ -13810,6 +14006,8 @@ typedef struct { float top_p; int extension_bytes; int prefix_bytes; + int logprobs; /* SUBMIT logprobs=k: 0 = channel closed (opt-in) */ + int pin; /* SUBMIT pin=1: keep the prompt end for the next prompts */ } V4ServeRequest; typedef struct { @@ -13817,6 +14015,8 @@ typedef struct { const char *request_id; int cancelled; int fatal; + int logprobs; + char tail[1024]; /* the next DATA frame's logprob tail, from on_scores */ } V4ServeStream; static const ColiServeWireProfile v4_wire = { @@ -14049,6 +14249,8 @@ static int v4_serve_read_request(FILE *input, FILE *output, request->top_p = command.top_p; request->extension_bytes = (int)command.extension_bytes; request->prefix_bytes = prefix_bytes; + request->logprobs = command.logprobs; + request->pin = command.pin; coli_serve_command_dispose(&command); return 2; } @@ -14092,8 +14294,15 @@ static int v4_serve_token(void *user_data, int token, float logit, char piece[1024]; int bytes = tok_decode(&stream->session->tokenizer, &token, 1, piece, (int)sizeof(piece) - 1); - v4_serve_data(stdout, stream->request_id, piece, bytes); + /* With the channel open the frame carries the tail on_scores left + * here: "DATA [tid tlp]*k", one frame per token. */ + if (stream->logprobs > 0 && bytes > 0) + coli_serve_write_data_lp(stdout, stream->request_id, piece, + (size_t)bytes, stream->tail); + else + v4_serve_data(stdout, stream->request_id, piece, bytes); } + stream->tail[0] = 0; if (v4_serve_drain_commands(stream)) { stream->cancelled = 1; return 1; @@ -14129,6 +14338,34 @@ static void v4_serve_done(FILE *output, const char *id, int completion, coli_serve_write_done_i32_suffix(output, id, &done, &prefix_reused, 1); } +/* ECHO frame of the numeric channel, the same bytes the other engines' + * serve_echo writes: "ECHO [tid tlp]*k" and the + * token's bytes DATA-framed after it. `scores` are raw head logits; + * coli_logprob_tail does the normalisation and the top-k. */ +static void v4_serve_echo(void *user_data, int position, int token, + const float *scores, int vocab) { + V4ServeStream *stream = user_data; + char tail[1024], piece[1024]; + coli_logprob_tail(tail, sizeof tail, scores, vocab, token, stream->logprobs); + int bytes = tok_decode(&stream->session->tokenizer, &token, 1, piece, + (int)sizeof(piece) - 1); + if (bytes < 0) bytes = 0; + printf("ECHO %s %d %d%s\n", stream->request_id, bytes, position, tail); + if (bytes > 0) fwrite(piece, 1, (size_t)bytes, stdout); + fputc('\n', stdout); + fflush(stdout); +} + +/* The tail of the next DATA frame, computed while the scores exist and + * written by v4_serve_token right after. */ +static void v4_serve_scores(void *user_data, int position, int token, + const float *scores, int vocab) { + (void)position; + V4ServeStream *stream = user_data; + coli_logprob_tail(stream->tail, sizeof stream->tail, scores, vocab, token, + stream->logprobs); +} + static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session, V4ServeRequest *request) { if (request->extension_bytes) { @@ -14190,7 +14427,7 @@ static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session, engine->experts ? coli_v4_expert_store_matmul_sec(engine->experts) : 0.0; double block_before = g_v4_prof_block_s, head_before = g_v4_prof_head_s; long long forwards_before = g_v4_prof_forwards; - V4ServeStream stream = {session, request->id, 0, 0}; + V4ServeStream stream = {session, request->id, 0, 0, request->logprobs, {0}}; ColiV4SessionGenerateStats stats = {0}; char error[512] = {0}; double started = spec_now(); @@ -14203,6 +14440,11 @@ static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session, .should_abort = v4_serve_abort, .abort_user_data = &stream, .prefix_bytes = (size_t)request->prefix_bytes, + .logprobs = request->logprobs, + .pin = request->pin, + .on_echo = request->logprobs > 0 ? v4_serve_echo : NULL, + .on_scores = request->logprobs > 0 ? v4_serve_scores : NULL, + .scores_user_data = &stream, }, v4_serve_token, &stream, &stats, error, sizeof(error)); double elapsed = spec_now() - started; @@ -14219,6 +14461,7 @@ static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session, int completion = stats.generated_tokens - (stats.eos_stopped ? 1 : 0); if (completion < 0) completion = 0; int length_limited = !stream.cancelled && !stats.eos_stopped && + request->max_tokens > 0 && /* a read-only request is not cut short */ stats.generated_tokens >= request->max_tokens; double decode = stats.decode_sec > 0.0 ? stats.decode_sec : elapsed; /* Trailing field: prompt tokens served from the previous turn's attention @@ -16059,6 +16302,9 @@ int coli_v4_config_load(ColiDeepSeekV4Config *config, const char *model_dir, #ifdef __AVX2__ #include #endif +#ifdef __ARM_NEON +#include +#endif float coli_e8m0_decode(uint8_t value) { if (value == 0xff) return NAN; @@ -16424,6 +16670,87 @@ void coli_fp4_matmul_batch_rows16_order(float *y, const uint8_t *q4, _mm256_storeu_ps(y + (int64_t)s * O + tile * 16 + 8, sum1[s]); } } +#elif defined(__ARM_NEON) + /* NEON port of the AVX2 arm (issue #1696): same algorithm — vqtbl1q_u8 + * nibble LUT decode of doubled e2m1 ints with an exact x0.5f un-double, + * 4x4 float transposes (vtrnq_f32 + vcombine_f32) making rows column- + * major, then strict (x*w)*scale rounding with separate mul/mul/add (no + * FMA fusion) so results stay bit-exact with the scalar and AVX2 arms. + * Tiles of 16 rows ride four float32x4_t lanes; one x broadcast serves + * all 4 columns of a group, keeping 4 independent add chains busy. */ + #pragma omp parallel for schedule(static) + for (int64_t tile = 0; tile < O / 16; tile++) { + /* doubled e2m1 codes as uint8: 0,2,4,6,8,12,16,24 / 0,0xFE..0xE8 */ + static const uint8_t lut2u[16] = {0,1,2,3,4,6,8,12, + 0,0xFF,0xFE,0xFD,0xFC,0xFA,0xF8,0xF4}; + const uint8x16_t lut2 = vld1q_u8(lut2u); + const uint8x16_t m4 = vdupq_n_u8(0x0F); + const float32x4_t half = vdupq_n_f32(0.5f); + float32x4_t acc[4][128]; + for (int g = 0; g < 4; g++) + for (int s = 0; s < S; s++) acc[g][s] = vdupq_n_f32(0.0f); + for (int base = 0; base < I; base += 32) { + float sc[16]; + float32x4_t rowv[16][8]; + for (int r = 0; r < 16; r++) { + int64_t row = tile * 16 + r; + sc[r] = e8lut[e8s[row * ng + base / 32]]; + uint8x16_t by = vld1q_u8(q4 + row * rb + base / 2); + uint8x16_t lo = vandq_u8(by, m4); + uint8x16_t hi = vandq_u8(vshrq_n_u8(by, 4), m4); + uint8x16_t z0 = vzip1q_u8(lo, hi); + uint8x16_t z1 = vzip2q_u8(lo, hi); + int8x16_t n0 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z0)); + int8x16_t n1 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z1)); + int16x8_t sa = vmovl_s8(vget_low_s8(n0)); + int16x8_t sb = vmovl_s8(vget_high_s8(n0)); + int16x8_t sc16 = vmovl_s8(vget_low_s8(n1)); + int16x8_t sd = vmovl_s8(vget_high_s8(n1)); + rowv[r][0] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sa))), half); + rowv[r][1] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sa))), half); + rowv[r][2] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sb))), half); + rowv[r][3] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sb))), half); + rowv[r][4] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sc16))), half); + rowv[r][5] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sc16))), half); + rowv[r][6] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sd))), half); + rowv[r][7] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sd))), half); + } + for (int g = 0; g < 4; g++) { + const float32x4_t sc4 = vld1q_f32(sc + g * 4); + for (int k = 0; k < 8; k++) { + const float32x4_t a0 = rowv[g*4+0][k], a1 = rowv[g*4+1][k]; + const float32x4_t a2 = rowv[g*4+2][k], a3 = rowv[g*4+3][k]; + float32x4x2_t t0 = vtrnq_f32(a0, a1); + float32x4x2_t t1 = vtrnq_f32(a2, a3); + /* vcombine, not vzip: vzip yields lane order (r0,r2,r1,r3) + * which would swap columns 1<->2 of each group. */ + const float32x4_t colv[4] = { + vcombine_f32(vget_low_f32(t0.val[0]), vget_low_f32(t1.val[0])), + vcombine_f32(vget_low_f32(t0.val[1]), vget_low_f32(t1.val[1])), + vcombine_f32(vget_high_f32(t0.val[0]), vget_high_f32(t1.val[0])), + vcombine_f32(vget_high_f32(t0.val[1]), vget_high_f32(t1.val[1])), + }; + for (int s = 0; s < S; s++) { + const float xs0 = x[(int64_t)s * I + base + k * 4 + 0]; + const float xs1 = x[(int64_t)s * I + base + k * 4 + 1]; + const float xs2 = x[(int64_t)s * I + base + k * 4 + 2]; + const float xs3 = x[(int64_t)s * I + base + k * 4 + 3]; + float32x4_t xw0 = vmulq_f32(vdupq_n_f32(xs0), colv[0]); + float32x4_t xw1 = vmulq_f32(vdupq_n_f32(xs1), colv[1]); + float32x4_t xw2 = vmulq_f32(vdupq_n_f32(xs2), colv[2]); + float32x4_t xw3 = vmulq_f32(vdupq_n_f32(xs3), colv[3]); + acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw0, sc4)); + acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw1, sc4)); + acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw2, sc4)); + acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw3, sc4)); + } + } + } + } + for (int s = 0; s < S; s++) + for (int g = 0; g < 4; g++) + vst1q_f32(y + (int64_t)s * O + tile * 16 + g * 4, acc[g][s]); + } #else #pragma omp parallel for schedule(static) for (int o = 0; o < O; o++) { @@ -16546,6 +16873,70 @@ void coli_fp4_matvec_rows16_order(float *y, const uint8_t *q4, _mm256_storeu_ps(y + tile * 16, sum0); _mm256_storeu_ps(y + tile * 16 + 8, sum1); } +#elif defined(__ARM_NEON) + /* NEON port of the batch-one arm above: 4 row-groups of 4 rows so the + * accumulators stay float32x4_t registers for the whole row tile. + * Same doubled-int LUT decode and column-ascending (x*w)*scale-then-add + * order as the rows16 kernels — bit-exact vs the scalar arm below. */ + #pragma omp parallel for schedule(static) + for (int64_t tile = 0; tile < O / 16; tile++) { + static const uint8_t lut2u[16] = {0,1,2,3,4,6,8,12, + 0,0xFF,0xFE,0xFD,0xFC,0xFA,0xF8,0xF4}; + const uint8x16_t lut2 = vld1q_u8(lut2u); + const uint8x16_t m4 = vdupq_n_u8(0x0F); + const float32x4_t half = vdupq_n_f32(0.5f); + float32x4_t acc[4]; + for (int g = 0; g < 4; g++) acc[g] = vdupq_n_f32(0.0f); + for (int base = 0; base < I; base += 32) { + float sc[16]; + float32x4_t rowv[16][8]; + for (int r = 0; r < 16; r++) { + int64_t row = tile * 16 + r; + sc[r] = e8lut[e8s[row * ng + base / 32]]; + uint8x16_t by = vld1q_u8(q4 + row * rb + base / 2); + uint8x16_t lo = vandq_u8(by, m4); + uint8x16_t hi = vandq_u8(vshrq_n_u8(by, 4), m4); + uint8x16_t z0 = vzip1q_u8(lo, hi); + uint8x16_t z1 = vzip2q_u8(lo, hi); + int8x16_t n0 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z0)); + int8x16_t n1 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z1)); + int16x8_t sa = vmovl_s8(vget_low_s8(n0)); + int16x8_t sb = vmovl_s8(vget_high_s8(n0)); + int16x8_t sc16 = vmovl_s8(vget_low_s8(n1)); + int16x8_t sd = vmovl_s8(vget_high_s8(n1)); + rowv[r][0] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sa))), half); + rowv[r][1] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sa))), half); + rowv[r][2] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sb))), half); + rowv[r][3] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sb))), half); + rowv[r][4] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sc16))), half); + rowv[r][5] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sc16))), half); + rowv[r][6] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sd))), half); + rowv[r][7] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sd))), half); + } + for (int g = 0; g < 4; g++) { + const float32x4_t sc4 = vld1q_f32(sc + g * 4); + for (int k = 0; k < 8; k++) { + const float32x4_t a0 = rowv[g*4+0][k], a1 = rowv[g*4+1][k]; + const float32x4_t a2 = rowv[g*4+2][k], a3 = rowv[g*4+3][k]; + float32x4x2_t t0 = vtrnq_f32(a0, a1); + float32x4x2_t t1 = vtrnq_f32(a2, a3); + const float32x4_t colv[4] = { + vcombine_f32(vget_low_f32(t0.val[0]), vget_low_f32(t1.val[0])), + vcombine_f32(vget_low_f32(t0.val[1]), vget_low_f32(t1.val[1])), + vcombine_f32(vget_high_f32(t0.val[0]), vget_high_f32(t1.val[0])), + vcombine_f32(vget_high_f32(t0.val[1]), vget_high_f32(t1.val[1])), + }; + for (int ci = 0; ci < 4; ci++) { + const float xc = x[base + k * 4 + ci]; + acc[g] = vaddq_f32(acc[g], vmulq_f32( + vmulq_f32(vdupq_n_f32(xc), colv[ci]), sc4)); + } + } + } + } + for (int g = 0; g < 4; g++) + vst1q_f32(y + tile * 16 + g * 4, acc[g]); + } #else #pragma omp parallel for schedule(static) for (int o = 0; o < O; o += 4) { diff --git a/c/deepseek_v4.h b/c/deepseek_v4.h index d57d1ab62..5b1ad3756 100644 --- a/c/deepseek_v4.h +++ b/c/deepseek_v4.h @@ -124,12 +124,31 @@ typedef struct { * uninterruptible as before. */ typedef int (*ColiV4SessionAbortFn)(void *user_data); +/* The numeric channel (SUBMIT logprobs=k, docs/brio.md): raw head scores, + * vocab_size floats, valid only for the duration of the callback. on_echo + * fires once per prompt position whose predictor this call computed, with the + * token that actually stands there; on_scores fires right before the on_token + * of every generated token, with the scores it was picked from. */ +typedef void (*ColiV4SessionScoresFn)(void *user_data, int position, int token, + const float *scores, int vocab); + typedef struct { - int max_new_tokens; /* required; clamped by session cap */ + int max_new_tokens; /* required; clamped by session cap; 0 only with logprobs > 0 */ int stop_at_sentence; int no_dspark; /* disable speculative draft/verification */ ColiV4SessionAbortFn should_abort; /* optional prefill abort poll */ void *abort_user_data; + /* SUBMIT logprobs=k / pin=1. logprobs > 0 opens the channel above, turns + * speculative decoding off for the request (a draft accepted in a block + * has no scores of its own) and allows max_new_tokens == 0, "read the + * prompt and stop". pin keeps the prompt-end scores and a snapshot of the + * attention state, so the prompts that extend this one start from here + * with their first fresh token's predictor intact. */ + int logprobs; + int pin; + ColiV4SessionScoresFn on_echo; + ColiV4SessionScoresFn on_scores; + void *scores_user_data; /* Optional: byte length of the prompt's stable leading prefix (the * rendered system turn). The session snapshots the attention state at * that token boundary during this prefill so later conversations that diff --git a/c/deepseek_v41.c b/c/deepseek_v41.c index e4f31b28f..93860212a 100644 --- a/c/deepseek_v41.c +++ b/c/deepseek_v41.c @@ -3522,6 +3522,16 @@ static int serve_eos(Model *m, const char *snap, int *ids, int cap) { return n; } +/* max_tokens is a ceiling, as on V4 and Qwen (#1641). A score-only + * request may fill the context; generation needs at least one free position. */ +static int serve_budget(int prompt, int requested, int context, int logprobs) { + if (prompt < 1 || prompt > context) return -1; + int budget = requested > 0 ? requested : (logprobs > 0 ? 0 : 256); + int room = context - prompt; + if (budget > 0 && room == 0) return -1; + return budget < room ? budget : room; +} + static void serve_loop(Model *m, Tok *tokenizer, const char *snap) { Cfg *c = &m->c; coli_serve_stdio_init(); @@ -3530,7 +3540,9 @@ static void serve_loop(Model *m, Tok *tokenizer, const char *snap) { coli_serve_write_ready(stdout, rss_gb()); serve_emap(m); float *logits = xmalloc((size_t)c->vocab * sizeof(float), "logits"); - int *ids = xmalloc((size_t)c->max_positions * sizeof(int), "prompt ids"); + /* tok_encode stops at its output capacity: one extra id distinguishes + * a full, valid read-only prompt from a silently truncated one. */ + int *ids = xmalloc(((size_t)c->max_positions + 1) * sizeof(int), "prompt ids"); float *pending_image = NULL; int pending_h = 0, pending_w = 0; @@ -3586,7 +3598,22 @@ static void serve_loop(Model *m, Tok *tokenizer, const char *snap) { mir_reads0[r] = g_mir_nread[r]; } int n_prompt = tok_encode(tokenizer, (const char *)command.payload, - (int)command.payload_bytes, ids, c->max_positions); + (int)command.payload_bytes, ids, c->max_positions + 1); + int budget = serve_budget(n_prompt, command.max_tokens, c->max_positions, + command.logprobs); + if (budget < 0) { + char message[128]; + snprintf(message, sizeof(message), + "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d", + n_prompt, command.max_tokens, c->max_positions); + coli_serve_write_error(stdout, command.id, + n_prompt < 1 ? "EMPTY_PROMPT" : message); + coli_serve_command_dispose(&command); continue; + } + if (command.max_tokens > budget) + fprintf(stderr, "[serve] max_tokens %d clamped to %d (context %d - prompt %d); " + "raise CTX for longer answers\n", + command.max_tokens, budget, c->max_positions, n_prompt); /* Decided BEFORE the reset, because the reset is what it decides about. * A chat client resends the whole transcript every turn; if this prompt * begins with the ids the state was built from, that state already IS @@ -3634,23 +3661,6 @@ static void serve_loop(Model *m, Tok *tokenizer, const char *snap) { fflush(stderr); } if (!reuse) model_reset(m); - if (n_prompt < 1) { - coli_serve_write_error(stdout, command.id, "EMPTY_PROMPT"); - coli_serve_command_dispose(&command); continue; - } - /* max_tokens=0 con logprobs>0 vuol dire "leggi e fermati": non e un - * valore mancante da rimpiazzare con un default, ed e' proprio il caso - * in cui un menu chiuso non vuole pagare un passo di decodifica per - * opzione. Senza logprobs 0 resta "non specificato" -> 256, come prima. */ - int budget = command.max_tokens > 0 ? command.max_tokens - : (command.logprobs > 0 ? 0 : 256); - if (n_prompt + budget > c->max_positions) { - char message[128]; - snprintf(message, sizeof(message), "CONTEXT_EXCEEDED %d %d", - n_prompt + budget, c->max_positions); - coli_serve_write_error(stdout, command.id, message); - coli_serve_command_dispose(&command); continue; - } coli_serve_write_accept(stdout, command.id, n_prompt); float *aligned = NULL; uint8_t *image_mask = NULL; diff --git a/c/deepseek_v4_internal.h b/c/deepseek_v4_internal.h index 2bcdf9cde..9686a56ac 100644 --- a/c/deepseek_v4_internal.h +++ b/c/deepseek_v4_internal.h @@ -966,6 +966,20 @@ struct ColiV4Session { uint64_t spec_drafted; uint64_t spec_accepted; int spec_disabled; + /* Prompt-end capture for SUBMIT pin=1: the ids fed and the head scores + * that predict the token after them. A later prompt that starts with + * exactly these ids gets its first fresh token's predictor from here; that + * token is the one a closed-set caller asks about (docs/brio.md). The + * attention state itself goes to a v4_ckpt slot; this is the part the + * snapshot does not hold. */ + int *pin_ids; + int pin_len; + float *pin_scores; + /* Scratch for the numeric channel, one hidden row and one row of head + * scores, allocated on first use and freed with the session so the many + * early returns of generate() leave nothing behind. */ + float *echo_hidden; + float *echo_scores; }; /* RAM-tiered expert open used by coli_v4_engine_open (replaces ld --wrap). @@ -1002,6 +1016,12 @@ extern void (*coli_v4_test_expert_wait_hook)(ColiExpertKey key); extern uint64_t coli_v4_test_fp4_batch_calls; extern uint64_t coli_v4_test_expert_victim_probes; int coli_v4_test_expert_slot_index(ColiExpertStore *store, ColiExpertKey key); +void coli_v4_test_reset_direct_io_stats(void); +uint64_t coli_v4_test_direct_reads(void); +uint64_t coli_v4_test_direct_fallbacks(void); +/* Point missing O_DIRECT twins at a dup of the buffered fd so tests can + * exercise the direct-window path on filesystems that refuse O_DIRECT. */ +int coli_v4_test_force_streaming_direct(ColiExpertStore *store); ColiV4Session *coli_v4_test_session_bare_create(ColiV4Engine *engine); void coli_v4_test_session_bare_destroy(ColiV4Session *session); diff --git a/c/doctor.py b/c/doctor.py index b8fc8783e..9b7045bb6 100644 --- a/c/doctor.py +++ b/c/doctor.py @@ -457,6 +457,26 @@ def deep_container_report(model, mirror_dir=None): } +def windows_backend_dll(image): + """Which GPU backend DLL a Windows host compiled in, or None if CPU-only. + + backend_loader.c bakes exactly one basename: coli_hip.dll under COLI_HIP_DLL + and coli_cuda.dll otherwise. That string is the build marker. The GLM/Qwen + banner "[CUDA] mode: routed experts" is only printed by those two engines; + a Kimi K3 CUDA_DLL host links the same loader and prints [K3-CUDA] instead. + DeepSeek V4 has its own pair and is not this function's job. + """ + if not image or b"[DSV4 CUDA]" in image: + return None + if b"coli_hip.dll" in image: + return "coli_hip.dll" + if b"coli_cuda.dll" in image: + return "coli_cuda.dll" + if b"[CUDA] mode: routed experts" in image or b"[K3-CUDA]" in image: + return "coli_cuda.dll" + return None + + def cuda_linkage(engine_path): """Return CUDA linkage state without loading the executable or CUDA runtime.""" engine = Path(engine_path) @@ -484,17 +504,11 @@ def cuda_linkage(engine_path): if sys.platform == "win32": # Windows DLL-split builds never link the GPU runtime directly: the host # LoadLibrary's its backend at runtime (backend_loader.c), so there's no - # import-table entry for ldd/dumpbin to see. Detect the GPU build via a - # marker string baked into the engine's #ifdef COLI_CUDA block, then - # require the backend artifact to sit next to the executable. - # - # WHICH artifact is not a guess. backend_loader.c compiles exactly one - # basename into the host -- COLI_BACKEND_DLL is "coli_hip.dll" under - # COLI_HIP_DLL and "coli_cuda.dll" otherwise -- so the binary states - # what it will load and we check for that. Asking for coli_cuda.dll - # unconditionally failed a working HIP host (a hard error, not a - # warning), and accepting either name would have passed a HIP host that - # only had a stray CUDA backend beside it. + # import-table entry for ldd/dumpbin to see. Detect the GPU build from + # the backend basename compiled into the host, then require that file + # next to the executable. Asking for coli_cuda.dll unconditionally + # failed a working HIP host (a hard error, not a warning), and requiring + # the GLM routed-experts banner missed every Kimi K3 CUDA_DLL build. try: image = engine.read_bytes() except OSError: @@ -506,10 +520,7 @@ def cuda_linkage(engine_path): present = any((engine.parent / name).is_file() for name in ("coli_cuda_dsv4_dg.dll", "coli_cuda_dsv4.dll")) return {"linked": present, "missing": not present} - if b"[CUDA] mode: routed experts" not in image: - return {"linked": False, "missing": False} - expected = next((name for name in ("coli_hip.dll", "coli_cuda.dll") - if name.encode() in image), None) + expected = windows_backend_dll(image) if expected is None: return {"linked": False, "missing": False} dll_present = (engine.parent / expected).is_file() diff --git a/c/exact_dot.h b/c/exact_dot.h new file mode 100644 index 000000000..bc40f1692 --- /dev/null +++ b/c/exact_dot.h @@ -0,0 +1,155 @@ +/* exact_dot.h — order-independent dot products for the opt-in exact verify mode. + * + * Every product is formed exactly (integer mantissas, never rounded) and accumulated by + * exponent bin; the bins are folded into one wide two's-complement integer and rounded to + * double ONCE, correctly (round-to-nearest-even), then to float. No summation order, no + * contraction flag and no SIMD width can change the result, so CPU and GPU agree to the bit + * by construction and near-ties are decided identically everywhere. The cost is throughput: + * an integer path, no FMA, no tensor cores. That is why it is opt-in and verify-only. + * + * Covers the two shapes the engine's verify rows need: + * exd_add_ff(acc, a, b) a*b for f32 a, b (attention q.k, p.v over f32 rows) + * exd_add_wsx(acc, w, s, x) (w*s)*x for int w, f32 s, x (quantised weight rows: int4/int8 + * * per-row/per-group scale * f32 act) + * Products are exact int64 (24+24 or 8+24+24 mantissa bits <= 56 bits); bins hold __int128 + * partial sums (72 bits of headroom), so any n up to 2^72 products per bin is safe. + * + * NaN / inf: a float dot with a NaN input is NaN, with an inf input is +-inf or NaN; the exact + * path reproduces the same *classification* (flags), so callers see the same special values. + * + * Copyright (c) 2026 Anomly, Inc. Licensed under the same terms as colibri (see LICENSE). + * Author: Ry Bruscoe. */ +#ifndef COLI_EXACT_DOT_H +#define COLI_EXACT_DOT_H +#include +#include +#include + +#if defined(__SIZEOF_INT128__) +typedef __int128 exd_i128; +#else +#error "exact_dot.h needs a 128-bit integer type (GCC/Clang); the exact verify mode is unavailable on this compiler" +#endif + +/* Product exponents: f32 mantissa m (24 bits, value m*2^e with e = exp-150), normals e in + * [-149, 104]; products e in [-298, 208]; with an 8-bit integer weight (value w) the exponent is + * the same range. Bin index = e + EXD_BIAS. */ +#define EXD_BIAS 320 +#define EXD_NBIN 640 +/* wide accumulator: EXD_NBIN + 64 bits of headroom, in 64-bit limbs (two's complement) */ +#define EXD_LIMBS 12 /* 768 bits */ + +typedef struct { + exd_i128 bin[EXD_NBIN]; + int lo, hi; /* used bin range (inclusive); lo > hi means empty */ + int nan, pinf, ninf; /* special-value flags, order-independent by construction */ +} exd_acc; + +static inline void exd_init(exd_acc *a){ a->lo = EXD_NBIN; a->hi = -1; a->nan = a->pinf = a->ninf = 0; } + +/* bins are zeroed lazily as the used range [lo, hi] grows (a full memset would cost more than a + * typical 512-element dot); returns the bin to add into */ +static inline exd_i128 *exd_touch(exd_acc *a, int idx){ + if(a->lo > a->hi){ a->bin[idx] = 0; a->lo = a->hi = idx; return &a->bin[idx]; } + if(idx < a->lo){ for(int i = idx; i < a->lo; i++) a->bin[i] = 0; a->lo = idx; } + else if(idx > a->hi){ for(int i = a->hi + 1; i <= idx; i++) a->bin[i] = 0; a->hi = idx; } + return &a->bin[idx]; +} + +/* decode f32 into (signed integer mantissa, exponent) with value = m * 2^e; returns 0 for + * zero (m=0), 1 for finite non-zero, 2 for inf, 3 for nan. */ +static inline int exd_decode(float f, int64_t *m, int *e){ + uint32_t u; memcpy(&u, &f, 4); + int s = (int)(u >> 31), ex = (int)((u >> 23) & 0xFF); uint32_t fr = u & 0x7FFFFF; + if(ex == 0xFF){ *m = 0; *e = 0; return fr ? 3 : 2; } + if(ex == 0){ if(!fr){ *m = 0; *e = 0; return 0; } *m = s ? -(int64_t)fr : (int64_t)fr; *e = -149; return 1; } + int64_t mm = (int64_t)(fr | 0x800000); *m = s ? -mm : mm; *e = ex - 150; return 1; +} + +static inline void exd_special(exd_acc *a, int ca, int cb, int64_t ma, int64_t mb, float fa, float fb){ + if(ca == 3 || cb == 3){ a->nan = 1; return; } + if(ca == 2 || cb == 2){ + /* inf * 0 = nan; inf * x has the sign of the product */ + if((ca == 2 && (cb == 0)) || (cb == 2 && (ca == 0))){ a->nan = 1; return; } + int sa = signbit(fa) ? 1 : 0, sb = signbit(fb) ? 1 : 0; (void)ma; (void)mb; + if(sa ^ sb) a->ninf = 1; else a->pinf = 1; + } +} + +static inline void exd_add_ff(exd_acc *a, float fa, float fb){ + int64_t ma, mb; int ea, eb; + int ca = exd_decode(fa, &ma, &ea), cb = exd_decode(fb, &mb, &eb); + if(ca > 1 || cb > 1){ exd_special(a, ca, cb, ma, mb, fa, fb); return; } + if(ca == 0 || cb == 0) return; + *exd_touch(a, ea + eb + EXD_BIAS) += (exd_i128)ma * mb; +} + +/* (w * s) * x with an exact integer w (|w| < 2^8), f32 scale s, f32 activation x */ +static inline void exd_add_wsx(exd_acc *a, int w, float s, float x){ + if(w == 0) return; + int64_t ms, mx; int es, ex; + int cs = exd_decode(s, &ms, &es), cx = exd_decode(x, &mx, &ex); + if(cs > 1 || cx > 1){ exd_special(a, cs, cx, ms, mx, s, x); return; } + if(cs == 0 || cx == 0) return; + *exd_touch(a, es + ex + EXD_BIAS) += (exd_i128)w * ms * mx; /* <= 8 + 24 + 24 bits: exact in int128 */ +} + +/* ---- fold bins into a 768-bit two's-complement integer scaled by 2^(lo - EXD_BIAS) ---- */ +static inline void exd_wide_add_shifted(uint64_t *w, exd_i128 v, int shift){ + /* add v * 2^shift into w (12 limbs, little-endian, two's complement), 0 <= shift < 704 */ + uint64_t t[EXD_LIMBS]; const uint64_t sx = (v < 0) ? ~(uint64_t)0 : 0; + for(int i = 0; i < EXD_LIMBS; i++) t[i] = sx; + t[0] = (uint64_t)v; t[1] = (uint64_t)(v >> 64); + int limbs = shift >> 6, bits = shift & 63; + if(bits){ uint64_t prev = 0; for(int i = 0; i < EXD_LIMBS; i++){ uint64_t cur = t[i]; t[i] = (cur << bits) | prev; prev = cur >> (64 - bits); } } + if(limbs){ for(int i = EXD_LIMBS - 1; i >= 0; i--) t[i] = (i - limbs >= 0) ? t[i - limbs] : 0; } + unsigned __int128 carry = 0; + for(int i = 0; i < EXD_LIMBS; i++){ unsigned __int128 sum = (unsigned __int128)w[i] + t[i] + carry; w[i] = (uint64_t)sum; carry = sum >> 64; } +} + +/* correctly rounded (nearest-even) conversion of a two's-complement 768-bit integer * 2^e2 */ +static inline double exd_wide_to_double(const uint64_t *w, int e2){ + uint64_t m[EXD_LIMBS]; memcpy(m, w, sizeof m); + int neg = (m[EXD_LIMBS - 1] >> 63) & 1; + if(neg){ unsigned __int128 c = 1; for(int i = 0; i < EXD_LIMBS; i++){ unsigned __int128 v = (unsigned __int128)(~m[i]) + c; m[i] = (uint64_t)v; c = v >> 64; } } + int top = -1; + for(int i = EXD_LIMBS - 1; i >= 0 && top < 0; i--) if(m[i]){ int b = 63; while(!((m[i] >> b) & 1)) b--; top = i * 64 + b; } + if(top < 0) return 0.0; + /* take the top 54 bits (53 + round bit) and a sticky bit for the rest */ + int hi = top, lo = top - 53; /* bits [lo, hi] = 54 bits */ + uint64_t bits54 = 0; int sticky = 0; + for(int b = hi; b >= lo; b--){ + int v = (b >= 0) ? (int)((m[b >> 6] >> (b & 63)) & 1) : 0; + bits54 = (bits54 << 1) | (uint64_t)v; + } + for(int i = 0; i < EXD_LIMBS && !sticky; i++){ + int base = i * 64; + if(base + 63 < lo){ if(m[i]) sticky = 1; continue; } + for(int b = base; b < base + 64 && b < lo; b++) if((m[i] >> (b - base)) & 1){ sticky = 1; break; } + } + uint64_t mant = bits54 >> 1; int round = (int)(bits54 & 1); + if(round && (sticky || (mant & 1))) mant += 1; + /* mant may have become 2^53: ldexp handles it exactly */ + double d = ldexp((double)mant, lo + 1 + e2); /* mant = bits [lo+1, hi] */ + return neg ? -d : d; +} + +static inline double exd_finish_double(const exd_acc *a){ + if(a->nan || (a->pinf && a->ninf)) return NAN; + if(a->pinf) return INFINITY; + if(a->ninf) return -INFINITY; + if(a->lo > a->hi) return 0.0; + uint64_t w[EXD_LIMBS]; memset(w, 0, sizeof w); + for(int i = a->lo; i <= a->hi; i++) if(a->bin[i]) exd_wide_add_shifted(w, a->bin[i], i - a->lo); + return exd_wide_to_double(w, a->lo - EXD_BIAS); +} + +static inline float exd_finish(const exd_acc *a){ return (float)exd_finish_double(a); } + +/* convenience: exact f32 dot of length n */ +static inline float exd_dot_ff(const float *x, const float *y, int n){ + exd_acc a; exd_init(&a); + for(int i = 0; i < n; i++) exd_add_ff(&a, x[i], y[i]); + return exd_finish(&a); +} +#endif /* COLI_EXACT_DOT_H */ diff --git a/c/family_registry.py b/c/family_registry.py index 7b3c2e364..d1fb3877d 100644 --- a/c/family_registry.py +++ b/c/family_registry.py @@ -702,7 +702,7 @@ def _dsv4_geometry(config, context, _model_dir): """ layers = _required_int(config, "num_hidden_layers", "deepseek_v4") experts = _required_int(config, "n_routed_experts", "deepseek_v4") - hidden = _required_int(config, "hidden_size", "deepseek_v4") + _required_int(config, "hidden_size", "deepseek_v4") heads = _required_int(config, "num_attention_heads", "deepseek_v4") head_dim = _required_int(config, "head_dim", "deepseek_v4") q_rank = _required_int(config, "q_lora_rank", "deepseek_v4") @@ -1268,6 +1268,10 @@ def _qwen38_resident_inventory(name, size, _config, dtype=None): # links NOCUDA_LDFLAGS. Left at the default this advertised a VRAM tier. supports_accelerator=False, expert_inventory=_individual_expert_inventory(_GLM_EXPERT), + # coli convert routes to convert_olmoe_merged.py (d4d11ef dispatch); + # the converter takes no precision flags (--ebits / --group-size etc.). + converter="convert_olmoe_merged.py", + converter_accepts=(), config_section="root", # implicit_cap 0, not 8: the engine sizes its expert cache from the RAM # budget once the dense weights are resident (#1443), so "nobody chose a diff --git a/c/fp8_format.h b/c/fp8_format.h new file mode 100644 index 000000000..d10ea461e --- /dev/null +++ b/c/fp8_format.h @@ -0,0 +1,35 @@ +/* fmt=8 (fp8-e4m3-b128) block geometry -- the single definition site for the + * 128x128 scale-block edge, shared by the CPU side (quant.h: matmul_fp8, + * e4m3 dequant plumbing; colibri.c: qt_addrow/qt_matvec_rows, qt_from_disk) + * and the CUDA backend's three converted sites (backend_cuda.cu: + * absorb_scale, quant_matmul's fmt=8 branch, the upload-time ng/scale_count + * computation); the f8-warp and f8-group kernels there keep their own + * 128 / >>7 literals, pinned against drift by backend_cuda.cu's + * static_assert(FP8_BLOCK == 128). Before this header the two backends + * agreed by IDENTICAL LITERALS restated in each file, so an edit to one side + * could not break the other side's build -- only a full-scale CPU-vs-CUDA + * parity run would have noticed. Kept deliberately tiny (no LUTs, no + * functions with OpenMP pragmas, no intrinsics) so the CUDA translation unit + * can include it without dragging in quant.h. + * + * The in-repo FP8 tests are NOT independent of this constant: the reference + * decoders in test_fp8_passthrough.c, test_fp8_load.c, test_fp8_e2e_loader.c, + * test_qwen38_native_weights.c, test_qt_addrow.c and test_shard_kvb_refuse.c + * all index with FP8_BLOCK/fp8_nblk via quant.h, so they move in lockstep + * with an edit here (test_backend_metal.mm's ref_fp8_nblk and + * test_backend_cuda.cu's fmt=8 reference keep their own literals, on + * purpose). An edit to FP8_BLOCK is a format change, not a tunable -- those + * tests cannot catch a wrong edit on their own. */ +#ifndef COLI_FP8_FORMAT_H +#define COLI_FP8_FORMAT_H + +#include + +#define FP8_BLOCK 128 + +/* Blocks covering n elements: ceil(n/FP8_BLOCK). Host-side helper (device + * code uses the FP8_BLOCK macro arithmetic directly -- this is not decorated + * for device compilation on purpose, to keep the header plain C). */ +static inline int64_t fp8_nblk(int n){ return ((int64_t)n + FP8_BLOCK - 1) / FP8_BLOCK; } + +#endif /* COLI_FP8_FORMAT_H */ diff --git a/c/glm53.c b/c/glm53.c index 8f33d4cd9..ac34e953b 100644 --- a/c/glm53.c +++ b/c/glm53.c @@ -971,7 +971,7 @@ static void mv(float *out, const Mat *w, const float *x) { } #endif #ifdef COLI_VULKAN - if (g_vk_ready && (w->fmt == 1 || w->fmt == 4)) { + if (g_vk_ready && w->resident && (w->fmt == 1 || w->fmt == 4)) { Mat *mutable_w = (Mat *)w; if (coli_vk_matmul((ColiVkTensor **)&mutable_w->vk, out, x, w->fmt == 4 ? (const void *)w->q4 : (const void *)w->q8, @@ -1277,15 +1277,6 @@ static void expert_table_init(GModel *m) { /* Quanti slot per layer: il budget diviso i layer sparsi. Il pavimento e' 1 e * non topk, perche' un pavimento a topk impegnerebbe topk*layer slot comunque, * cioe' molti GB, a dispetto del budget chiesto. */ -/* Quanta RAM il sistema dice di poter dare adesso. MemAvailable e non MemFree: - * la seconda ignora la page cache riutilizzabile e farebbe stimare molto meno - * di quello che c'e'. */ -static double memory_available_gb(void) { - /* #1375: era una lettura di /proc/meminfo, che su Windows e macOS non - * esiste: 0 -> budget 1 GB -> uno slot per layer, in silenzio. */ - return compat_mem_available_gb(); -} - static void expert_cache_init(GModel *m) { const Cfg *c = &m->c; const char *setting = getenv("GLM53_EXPERT_GB"); @@ -1295,19 +1286,39 @@ static void expert_cache_init(GModel *m) { * versi -- su una macchina piccola va in OOM, su una grande lascia RAM * inutilizzata mentre il disco fa tutto il lavoro, che e' esattamente * quello che e' successo alla prima esecuzione vera. */ + const int from = c->first_dense > m->layer_begin ? c->first_dense : m->layer_begin; + int sparse = m->layer_end - from; + if (sparse < 0) sparse = 0; double budget; if (setting) budget = atof(setting); else { - const double free_now = memory_available_gb(); - budget = free_now - 3.0; + /* MemAvailable counts reclaimable page cache as free. Sizing this LRU + * from it means allocating, as anonymous memory, the very pages the + * next slot miss would have been served from: both caches hold the + * same bytes, the model is paid for twice, and the cheap copy is the + * one that loses. So leave the model room to stay in page cache and + * take only what is left over. */ + double total = 0.0, free_now = 0.0; + compat_meminfo_gb(&total, &free_now); /* one pass, both fields */ + /* the routed experts dominate; the rest of the weights are ~7% */ + const double model_gb = (double)sparse * c->n_experts * (double)m->e_slot / 1e9 * 1.07; + double margin = total * 0.08; + if (margin < 4.0) margin = 4.0; + if (total <= 0.0 || model_gb >= total - margin) { + /* The model does not fit in RAM anyway, so page cache cannot help: + * keep as many slots as possible, exactly as before. `total <= 0` + * is "could not measure" and takes the same safe path. */ + budget = free_now - 3.0; + } else { + budget = total - model_gb - margin; + if (budget > free_now - 3.0) budget = free_now - 3.0; + } if (budget < 1.0) budget = 1.0; if (getenv("GLM53_VERBOSE")) - fprintf(stderr, "expert budget: %.1f GB (%.1f available, 3 GB reserved)\n", - budget, free_now); + fprintf(stderr, "expert budget: %.1f GB (%.1f total, %.1f model, " + "%.1f margin, %.1f available)\n", + budget, total, model_gb, margin, free_now); } - const int from = c->first_dense > m->layer_begin ? c->first_dense : m->layer_begin; - int sparse = m->layer_end - from; - if (sparse < 0) sparse = 0; int cap = (int)((budget * 1e9) / ((double)m->e_slot * (sparse > 0 ? sparse : 1))); if (g_cap_override > 0) cap = g_cap_override; /* scelta esplicita: vince */ if (cap < 1) cap = 1; @@ -2153,7 +2164,7 @@ static void model_load_range(GModel *m, const char *dir, int layer_begin, else snprintf(spv, sizeof(spv), "%s/qmatmul.spv", given ? given : "shaders"); g_vk_ready = coli_vk_init(spv) && coli_vk_available(); if (g_vk_ready) coli_vk_set_swiglu_limit(m->c.swiglu_limit); - fprintf(stderr, g_vk_ready) + fprintf(stderr, g_vk_ready ? "Vulkan: active for resident matrices\n" : "Vulkan: no usable device (%s), falling back to CPU\n", spv); } @@ -2376,6 +2387,9 @@ static float *run_layers(GModel *m, GSession *s, float *streams, float *next, static void mat_release(Mat *mat) { #ifdef COLI_METAL if (mat->metal) coli_metal_tensor_free((ColiMetalTensor *)mat->metal); +#endif +#ifdef COLI_VULKAN + if (mat->vk) coli_vk_tensor_free((ColiVkTensor *)mat->vk); #endif free((void *)mat->f); free((void *)mat->q8); free((void *)mat->q4); free((void *)mat->s); diff --git a/c/gsgemv.h b/c/gsgemv.h new file mode 100644 index 000000000..fede49259 --- /dev/null +++ b/c/gsgemv.h @@ -0,0 +1,212 @@ +#ifndef COLIBRI_GSGEMV_H +#define COLIBRI_GSGEMV_H +/* Group-scaled int8 GEMV: one f32 scale per `gs` input elements per row, the + * layout the gs64 expert containers use. Row layout of `scale`: [O][I/gs] + * row-major. Its own header so tests/test_gsgemv.c can link the very kernel + * the engine runs; qwen36.c carries a main and cannot be linked into a test. + * + * This is qwen36's hottest kernel: every expert matmul goes through it, three + * per expert, eight experts, forty layers. Each row's float operations must + * stay in exactly this order -- float addition is not associative and the + * engine's token stream is required to be byte-identical to the reference. + * tests/test_gsgemv.c holds the pre-restructure kernel verbatim and compares + * raw float bits, so any reassociation fails there rather than surfacing as + * drifted text much later. + * + * The SSE4.1 tier below is the exception to that byte-identical rule, by + * design: it is new code with no pre-existing output to match, its tree + * reduction is a genuinely different shape than the scalar reference, and it + * is checked by tests/test_gsgemv.c against a tolerance, not memcmp. It + * routes its FMA and float loads through sse41_kernels.h -- the same shared + * primitives header olmoe.c uses -- rather than inlining its own copy. */ +#include +#include +#if (defined(__AVX2__) && defined(__FMA__)) || defined(__SSE4_1__) +#include +#endif +#if defined(__SSE4_1__) +#include "sse41_kernels.h" +#endif + +#if defined(__AVX2__) && defined(__FMA__) +/* The group reduction, lifted verbatim so the four-row body and the tail row + * cannot drift apart. Order of the adds is load-bearing, not stylistic. */ +static inline float gs_group_sum(__m256 a0, __m256 a1) { + a0 = _mm256_add_ps(a0, a1); + __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1)); + s = _mm_add_ps(s, _mm_movehl_ps(s,s)); + s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1)); + return _mm_cvtss_f32(s); +} +#elif defined(__SSE4_1__) +/* Same reduction as gs_group_sum above, one level shallower: a0/a1 are + * already the two four-lane halves, so there is no 256->128 fold first. */ +static inline float gs_group_sum_sse41(__m128 a0, __m128 a1) { + a0 = _mm_add_ps(a0, a1); + a0 = _mm_add_ps(a0, _mm_movehl_ps(a0,a0)); + a0 = _mm_add_ss(a0, _mm_shuffle_ps(a0,a0,1)); + return _mm_cvtss_f32(a0); +} +#endif + +static void matmul_q_gs(float *y, const float *x, const int8_t *q, const float *scale, + int I, int O, int gs) { + int ng = (I + gs - 1) / gs; +#if defined(__AVX2__) && defined(__FMA__) + if ((gs & 31) == 0) { + /* Four output rows in flight at once. Each row keeps its own pair of + * accumulators, its own reduction and its own running acc, so its float + * operations happen in exactly the order the single-row loop used them + * -- the interleave is a schedule change, not an algebraic one. + * + * Why it pays: one group is 8 FMAs but a dependency chain of roughly 36 + * cycles (accumulate, then the reduction tree, then the loop-carried + * acc +=), against a throughput floor near 4. The single-row loop stalls + * on that chain about nine times out of ten. Four independent rows fill + * the gaps, and the x block gets loaded once instead of four times. */ + int o4 = O & ~3; + #pragma omp parallel for schedule(static) if(O >= 256) + for (int ob = 0; ob < o4; ob += 4) { + const int8_t *w0 = q + (int64_t)(ob+0) * I, *w1 = q + (int64_t)(ob+1) * I; + const int8_t *w2 = q + (int64_t)(ob+2) * I, *w3 = q + (int64_t)(ob+3) * I; + const float *sc0 = scale + (int64_t)(ob+0) * ng, *sc1 = scale + (int64_t)(ob+1) * ng; + const float *sc2 = scale + (int64_t)(ob+2) * ng, *sc3 = scale + (int64_t)(ob+3) * ng; + float acc0 = 0.f, acc1 = 0.f, acc2 = 0.f, acc3 = 0.f; + for (int gi = 0; gi < ng; gi++) { + __m256 a00 = _mm256_setzero_ps(), a01 = _mm256_setzero_ps(); + __m256 a10 = _mm256_setzero_ps(), a11 = _mm256_setzero_ps(); + __m256 a20 = _mm256_setzero_ps(), a21 = _mm256_setzero_ps(); + __m256 a30 = _mm256_setzero_ps(), a31 = _mm256_setzero_ps(); + int base = gi * gs, end = base + gs; if (end > I) end = I; + for (int i = base; i + 16 <= end; i += 16) { + __m256 xl = _mm256_loadu_ps(x+i), xh = _mm256_loadu_ps(x+i+8); + __m128i b0 = _mm_loadu_si128((const __m128i*)(w0 + i)); + a00 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a00); + a01 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a01); + __m128i b1 = _mm_loadu_si128((const __m128i*)(w1 + i)); + a10 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a10); + a11 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a11); + __m128i b2 = _mm_loadu_si128((const __m128i*)(w2 + i)); + a20 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b2)), a20); + a21 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b2,8))), a21); + __m128i b3 = _mm_loadu_si128((const __m128i*)(w3 + i)); + a30 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b3)), a30); + a31 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b3,8))), a31); + } + /* fmaf, not `acc += sum * sc`: the shipped single-row loop was + * written as a separate multiply and add, but -ffp-contract=fast + * fused it, and that fused form is what produced the reference + * output. Whether the optimizer also fuses it in this differently + * shaped loop is not something to leave to chance -- spelling it + * out pins the single rounding the reference depends on. */ + acc0 = fmaf(gs_group_sum(a00, a01), sc0[gi], acc0); + acc1 = fmaf(gs_group_sum(a10, a11), sc1[gi], acc1); + acc2 = fmaf(gs_group_sum(a20, a21), sc2[gi], acc2); + acc3 = fmaf(gs_group_sum(a30, a31), sc3[gi], acc3); + } + y[ob+0] = acc0; y[ob+1] = acc1; y[ob+2] = acc2; y[ob+3] = acc3; + } + for (int o = o4; o < O; o++) { /* at most three rows */ + const int8_t *w = q + (int64_t)o * I; + const float *sc = scale + (int64_t)o * ng; + float acc = 0.f; + for (int gi = 0; gi < ng; gi++) { + __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps(); + int base = gi * gs, end = base + gs; if (end > I) end = I; + for (int i = base; i + 16 <= end; i += 16) { + __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); + a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0); + a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1); + } + acc = fmaf(gs_group_sum(a0, a1), sc[gi], acc); + } + y[o] = acc; + } + return; + } +#elif defined(__SSE4_1__) + if ((gs & 15) == 0) { + /* Same four-row interleave as the AVX2 tier above, sized down to this + * ISA: each __m128 lane holds 4 lanes instead of 8, so a group step is + * 8 int8s, not 16 -- the gate halves to 16 and each row keeps two + * __m128 accumulators instead of two __m256. The multiply-add goes + * through COLIBRI_FMA (sse41_kernels.h): on real SSE4.1-only hardware + * (no __FMA__) that macro expands to the same explicit mul-then-add + * this tier always used, so nothing changes there; it only takes the + * single-rounded _mm_fmadd_ps path if __FMA__ is somehow defined + * without __AVX2__, which does not happen on real hardware but keeps + * the tier correct rather than silently wrong if it ever did. Float + * loads of `x` go through colibri_sse41_loadu_ps for the same reason + * olmoe.c does: one definition shared with future consumers instead + * of a fourth inline copy of _mm_loadu_ps. */ + int o4 = O & ~3; + #pragma omp parallel for schedule(static) if(O >= 256) + for (int ob = 0; ob < o4; ob += 4) { + const int8_t *w0 = q + (int64_t)(ob+0) * I, *w1 = q + (int64_t)(ob+1) * I; + const int8_t *w2 = q + (int64_t)(ob+2) * I, *w3 = q + (int64_t)(ob+3) * I; + const float *sc0 = scale + (int64_t)(ob+0) * ng, *sc1 = scale + (int64_t)(ob+1) * ng; + const float *sc2 = scale + (int64_t)(ob+2) * ng, *sc3 = scale + (int64_t)(ob+3) * ng; + float acc0 = 0.f, acc1 = 0.f, acc2 = 0.f, acc3 = 0.f; + for (int gi = 0; gi < ng; gi++) { + __m128 a00 = _mm_setzero_ps(), a01 = _mm_setzero_ps(); + __m128 a10 = _mm_setzero_ps(), a11 = _mm_setzero_ps(); + __m128 a20 = _mm_setzero_ps(), a21 = _mm_setzero_ps(); + __m128 a30 = _mm_setzero_ps(), a31 = _mm_setzero_ps(); + int base = gi * gs, end = base + gs; if (end > I) end = I; + for (int i = base; i + 8 <= end; i += 8) { + __m128 xl = colibri_sse41_loadu_ps(x+i), xh = colibri_sse41_loadu_ps(x+i+4); + __m128i b0 = _mm_loadl_epi64((const __m128i*)(w0 + i)); + a00 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b0)), a00); + a01 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,4))), a01); + __m128i b1 = _mm_loadl_epi64((const __m128i*)(w1 + i)); + a10 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b1)), a10); + a11 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b1,4))), a11); + __m128i b2 = _mm_loadl_epi64((const __m128i*)(w2 + i)); + a20 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b2)), a20); + a21 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b2,4))), a21); + __m128i b3 = _mm_loadl_epi64((const __m128i*)(w3 + i)); + a30 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b3)), a30); + a31 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b3,4))), a31); + } + acc0 = fmaf(gs_group_sum_sse41(a00, a01), sc0[gi], acc0); + acc1 = fmaf(gs_group_sum_sse41(a10, a11), sc1[gi], acc1); + acc2 = fmaf(gs_group_sum_sse41(a20, a21), sc2[gi], acc2); + acc3 = fmaf(gs_group_sum_sse41(a30, a31), sc3[gi], acc3); + } + y[ob+0] = acc0; y[ob+1] = acc1; y[ob+2] = acc2; y[ob+3] = acc3; + } + for (int o = o4; o < O; o++) { /* at most three rows */ + const int8_t *w = q + (int64_t)o * I; + const float *sc = scale + (int64_t)o * ng; + float acc = 0.f; + for (int gi = 0; gi < ng; gi++) { + __m128 a0 = _mm_setzero_ps(), a1 = _mm_setzero_ps(); + int base = gi * gs, end = base + gs; if (end > I) end = I; + for (int i = base; i + 8 <= end; i += 8) { + __m128i b0 = _mm_loadl_epi64((const __m128i*)(w + i)); + a0 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b0)), a0); + a1 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+4), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,4))), a1); + } + acc = fmaf(gs_group_sum_sse41(a0, a1), sc[gi], acc); + } + y[o] = acc; + } + return; + } +#endif + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + const float *sc = scale + (int64_t)o * ng; + float acc = 0.f; + for (int gi = 0; gi < ng; gi++) { + int base = gi * gs, end = base + gs; if (end > I) end = I; + float part = 0.f; + for (int i = base; i < end; i++) part += x[i] * (float)w[i]; + acc += part * sc[gi]; + } + y[o] = acc; + } +} + +#endif /* COLIBRI_GSGEMV_H */ diff --git a/c/idot.h b/c/idot.h new file mode 100644 index 000000000..34eaf3eb4 --- /dev/null +++ b/c/idot.h @@ -0,0 +1,1028 @@ +/* idot.h -- integer dot-product kernels shared by every engine. + * + * The activation is quantized to int8 once (one scale per row, amax/127, + * qrow_i8) and the weights, int8 rows or int4 planar blocks with per-row or + * per-group scales, are multiplied with integer instructions: maddubs on + * AVX2, vpdpbusd on AVX-VNNI and AVX-512 VNNI, sdot/smmla on NEON, and the + * AMX tile kernel where the CPU has it. Exact int32 sums, scaled once per + * output, so every ISA path is bit-identical to the scalar reference. + * + * Lived inside quant.h; moved here so an engine that has its own dense + * kernels (qwen36.c uses qgemv.h/gsgemv.h, whose matmul_q would collide + * with quant.h's) can still take the integer path for its dense trunk. + * quant.h includes this file at the top, so its other consumers see the + * same declarations in the same place as before. */ +#ifndef COLIBRI_IDOT_H +#define COLIBRI_IDOT_H +#include +#include +#include +#include +#include +/* ---- SIMD includes -------------------------------------------------------- */ +#ifdef __AVX2__ +#include +static inline float hsum256(__m256 v){ + __m128 lo=_mm256_castps256_ps128(v), hi=_mm256_extractf128_ps(v,1); + lo=_mm_add_ps(lo,hi); __m128 sh=_mm_movehl_ps(lo,lo); lo=_mm_add_ps(lo,sh); + sh=_mm_shuffle_ps(lo,lo,1); lo=_mm_add_ss(lo,sh); return _mm_cvtss_f32(lo); +} +static inline int hsum256_i32(__m256i v){ + __m128i lo=_mm256_castsi256_si128(v), hi=_mm256_extracti128_si256(v,1); + lo=_mm_add_epi32(lo,hi); lo=_mm_hadd_epi32(lo,lo); lo=_mm_hadd_epi32(lo,lo); + return _mm_cvtsi128_si32(lo); +} +#endif +#if defined(__AVXVNNI__) && defined(__AVX2__) +static inline int hsum128_i32(__m128i v){ + v=_mm_hadd_epi32(v,v); v=_mm_hadd_epi32(v,v); return _mm_cvtsi128_si32(v); +} +#endif +#ifdef __ARM_NEON +#include +#endif +#ifdef __VSX__ +#include +#undef vector +#undef pixel +#undef bool +#endif + + +/* ---- IDOT: integer dot kernels (int8-quantized activations) --------------- */ +#if defined(__AVX512VNNI__) && defined(__AVX512BW__) +#define IDOT_KERNEL "avx512-vnni" +#elif defined(__AVXVNNI__) && defined(__AVX2__) +#define IDOT_KERNEL "avx-vnni" +#elif defined(__AVX2__) +#define IDOT_KERNEL "avx2" +#elif defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8) +#define IDOT_KERNEL "neon-i8mm" +#elif defined(__ARM_NEON) +#define IDOT_KERNEL "neon" +#elif defined(__VSX__) +#define IDOT_KERNEL "vsx" +#else +#define IDOT_KERNEL "scalar" +#endif + +static inline float qrow_i8(const float *x, int8_t *q, int I){ + float amax=0; for(int i=0;iamax)amax=a; } + float s=amax/127.f; if(s<1e-12f) s=1e-12f; float inv=1.f/s; + for(int i=0;i int8 with one scale (shared by every engine whose trunk + * meets the integer kernels below), plus the int32 sum of every block of 64 + * (the K1b kernel subtracts 8*sum per group because its nibbles are unsigned). + * Vectorized: the scalar lrintf loop was measured at ~7 us for 4096 values, + * which times ~150 GEMVs per token is a millisecond thrown away. Rounding is + * to nearest even in both paths, so the vector path equals qrow_i8 bit for bit. */ +static float dense_act_i8(const float *x, int I, int8_t *xq, int32_t *xsg){ + float amax = 0.f; + int i = 0; +#ifdef __AVX2__ + { + __m256 am = _mm256_setzero_ps(); + const __m256 sign = _mm256_set1_ps(-0.0f); + for (; i + 8 <= I; i += 8) am = _mm256_max_ps(am, _mm256_andnot_ps(sign, _mm256_loadu_ps(x + i))); + float tmp[8]; _mm256_storeu_ps(tmp, am); + for (int k = 0; k < 8; k++) if (tmp[k] > amax) amax = tmp[k]; + } +#endif + for (; i < I; i++) { float a = fabsf(x[i]); if (a > amax) amax = a; } + float s = amax / 127.f; if (s < 1e-12f) s = 1e-12f; + float inv = 1.f / s; + i = 0; +#ifdef __AVX2__ + { + const __m256 vinv = _mm256_set1_ps(inv); + for (; i + 32 <= I; i += 32) { + __m256i a = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i), vinv)); + __m256i b = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i + 8), vinv)); + __m256i c = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i + 16), vinv)); + __m256i d = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i + 24), vinv)); + /* packs interleave 128-bit lanes: fix the order with one permute */ + __m256i ab = _mm256_packs_epi32(a, b), cd = _mm256_packs_epi32(c, d); + __m256i abcd = _mm256_packs_epi16(ab, cd); + abcd = _mm256_permutevar8x32_epi32(abcd, _mm256_setr_epi32(0, 4, 1, 5, 2, 6, 3, 7)); + _mm256_storeu_si256((__m256i *)(xq + i), abcd); + } + } +#endif + for (; i < I; i++) xq[i] = (int8_t)lrintf(x[i] * inv); + if (xsg) { + int ng = I / 64; + for (int g = 0; g < ng; g++) { + int32_t sum = 0; + for (int k = 0; k < 64; k++) sum += xq[g * 64 + k]; + xsg[g] = sum; + } + } + return s; +} + +/* dot int8*int8 */ +static inline int32_t dot_i8i8(const int8_t *w, const int8_t *x, int I){ + int32_t sum=0; int i=0; +#if defined(__AVX512VNNI__) && defined(__AVX512BW__) + __m512i acc=_mm512_setzero_si512(); + for(;i+64<=I;i+=64){ + __m512i wv=_mm512_loadu_si512((const void*)(w+i)); + __m512i xv=_mm512_loadu_si512((const void*)(x+i)); + __mmask64 neg=_mm512_movepi8_mask(wv); + __m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv); + acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs); + } + sum=_mm512_reduce_add_epi32(acc); +#elif defined(__AVXVNNI__) && defined(__AVX2__) + /* 4 accumulatori indipendenti (64 byte/iter): un solo acc incatena i vpdpbusd + * (latenza-bound ~5c). Somme intere associative -> bit-identico. Stessa struttura + * dei 4 accumulatori del ramo NEON piu' sotto. + * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer + * adds are associative, so the result is bit-identical (mirrors the NEON path). */ + __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128(); + for(;i+64<=I;i+=64){ + __m128i w0=_mm_loadu_si128((const __m128i*)(w+i)), x0=_mm_loadu_si128((const __m128i*)(x+i)); + __m128i w1=_mm_loadu_si128((const __m128i*)(w+i+16)), x1=_mm_loadu_si128((const __m128i*)(x+i+16)); + __m128i w2=_mm_loadu_si128((const __m128i*)(w+i+32)), x2=_mm_loadu_si128((const __m128i*)(x+i+32)); + __m128i w3=_mm_loadu_si128((const __m128i*)(w+i+48)), x3=_mm_loadu_si128((const __m128i*)(x+i+48)); + a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0)); + a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1)); + a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2)); + a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3)); + } + __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3)); + for(;i+16<=I;i+=16){ + __m128i wv=_mm_loadu_si128((const __m128i*)(w+i)); + __m128i xv=_mm_loadu_si128((const __m128i*)(x+i)); + acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),_mm_sign_epi8(xv,wv)); + } + sum=hsum128_i32(acc); +#elif defined(__AVX2__) + __m256i acc=_mm256_setzero_si256(); const __m256i ones=_mm256_set1_epi16(1); + for(;i+32<=I;i+=32){ + __m256i wv=_mm256_loadu_si256((const __m256i*)(w+i)); + __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i)); + __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv)); + acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones)); + } + sum=hsum256_i32(acc); +#elif defined(__ARM_NEON) +#if defined(__ARM_FEATURE_DOTPROD) + int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); + for(;i+64<=I;i+=64){ + a0=vdotq_s32(a0,vld1q_s8(w+i), vld1q_s8(x+i)); + a1=vdotq_s32(a1,vld1q_s8(w+i+16),vld1q_s8(x+i+16)); + a2=vdotq_s32(a2,vld1q_s8(w+i+32),vld1q_s8(x+i+32)); + a3=vdotq_s32(a3,vld1q_s8(w+i+48),vld1q_s8(x+i+48)); + } + int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); + for(;i+16<=I;i+=16) acc=vdotq_s32(acc,vld1q_s8(w+i),vld1q_s8(x+i)); + sum=vaddvq_s32(acc); +#else + int32x4_t acc=vdupq_n_s32(0); + for(;i+16<=I;i+=16){ + int8x16_t wv=vld1q_s8(w+i), xv=vld1q_s8(x+i); + int16x8_t p=vmull_s8(vget_low_s8(wv),vget_low_s8(xv)); + p=vmlal_s8(p,vget_high_s8(wv),vget_high_s8(xv)); + acc=vpadalq_s16(acc,p); + } + sum=vaddvq_s32(acc); +#endif +#elif defined(__VSX__) + __vector signed int acc=vec_splats(0); + const __vector signed char vz=vec_splats((signed char)0); + for(;i+16<=I;i+=16){ + __vector signed char wv=vec_xl(0,(const signed char*)(w+i)); + __vector signed char xv=vec_xl(0,(const signed char*)(x+i)); + __vector __bool char neg=vec_cmplt(wv,vz); + __vector signed char xs=vec_sel(xv,vec_sub(vz,xv),neg); + __vector unsigned char wa=(__vector unsigned char)vec_sel(wv,vec_sub(vz,wv),neg); + acc=vec_msum(xs,wa,acc); + } + sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3); +#endif + for(;i>1))); + __m256i lo=_mm256_and_si256(by,m4v), hi=_mm256_and_si256(_mm256_srli_epi16(by,4),m4v); + __m256i z0=_mm256_unpacklo_epi8(lo,hi), z1=_mm256_unpackhi_epi8(lo,hi); + __m512i wv=_mm512_sub_epi8(_mm512_inserti64x4(_mm512_castsi256_si512(z0),z1,1),b8v); + __m512i xv=_mm512_permutexvar_epi64(xidx,_mm512_loadu_si512((const void*)(x+i))); + __mmask64 neg=_mm512_movepi8_mask(wv); + __m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv); + acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs); + } + sum=_mm512_reduce_add_epi32(acc); +#elif defined(__AVXVNNI__) && defined(__AVX2__) + /* 4 accumulatori indipendenti (64 elementi = 32 byte packed/iter): un solo acc + * incatena i vpdpbusd (latenza-bound ~5c). Somme intere associative -> bit-identico. + * Stessa struttura dei 4 accumulatori del ramo NEON piu' sotto. + * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer + * adds are associative, so the result is bit-identical (mirrors the NEON path). */ + const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8); + __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128(); + for(;i+64<=I;i+=64){ + __m128i by0=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); /* elem i..i+31 */ + __m128i by1=_mm_loadu_si128((const __m128i*)(w4+(i>>1)+16)); /* elem i+32..i+63 */ + __m128i lo0=_mm_and_si128(by0,m4), hi0=_mm_and_si128(_mm_srli_epi16(by0,4),m4); + __m128i lo1=_mm_and_si128(by1,m4), hi1=_mm_and_si128(_mm_srli_epi16(by1,4),m4); + __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo0,hi0),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo0,hi0),b8); + __m128i w2=_mm_sub_epi8(_mm_unpacklo_epi8(lo1,hi1),b8), w3=_mm_sub_epi8(_mm_unpackhi_epi8(lo1,hi1),b8); + __m128i x0=_mm_loadu_si128((const __m128i*)(x+i)), x1=_mm_loadu_si128((const __m128i*)(x+i+16)); + __m128i x2=_mm_loadu_si128((const __m128i*)(x+i+32)), x3=_mm_loadu_si128((const __m128i*)(x+i+48)); + a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0)); + a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1)); + a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2)); + a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3)); + } + __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3)); + for(;i+32<=I;i+=32){ /* 32-nibble remainder: 2 dpbusd, same unpack */ + __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); + __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4); + __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo,hi),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo,hi),b8); + __m128i x0=_mm_loadu_si128((const __m128i*)(x+i)); + __m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16)); + acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0)); + acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1)); + } + sum=hsum128_i32(acc); +#elif defined(__AVX2__) + const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi8(8); + const __m256i ones=_mm256_set1_epi16(1); + __m256i acc=_mm256_setzero_si256(); + for(;i+32<=I;i+=32){ + __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); + __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4); + __m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi); + __m256i wv=_mm256_sub_epi8(_mm256_set_m128i(n1,n0),b8); + __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i)); + __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv)); + acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones)); + } + sum=hsum256_i32(acc); +#elif defined(__ARM_NEON) + const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8); +#if defined(__ARM_FEATURE_DOTPROD) + int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); + for(;i+64<=I;i+=64){ + uint8x16_t byA=vld1q_u8(w4+(i>>1)), byB=vld1q_u8(w4+(i>>1)+16); + uint8x16x2_t zA=vzipq_u8(vandq_u8(byA,m4q), vshrq_n_u8(byA,4)); + uint8x16x2_t zB=vzipq_u8(vandq_u8(byB,m4q), vshrq_n_u8(byB,4)); + a0=vdotq_s32(a0,vsubq_s8(vreinterpretq_s8_u8(zA.val[0]),b8q),vld1q_s8(x+i)); + a1=vdotq_s32(a1,vsubq_s8(vreinterpretq_s8_u8(zA.val[1]),b8q),vld1q_s8(x+i+16)); + a2=vdotq_s32(a2,vsubq_s8(vreinterpretq_s8_u8(zB.val[0]),b8q),vld1q_s8(x+i+32)); + a3=vdotq_s32(a3,vsubq_s8(vreinterpretq_s8_u8(zB.val[1]),b8q),vld1q_s8(x+i+48)); + } + int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); + for(;i+32<=I;i+=32){ + uint8x16_t by=vld1q_u8(w4+(i>>1)); + uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4)); + acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q),vld1q_s8(x+i)); + acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q),vld1q_s8(x+i+16)); + } + sum=vaddvq_s32(acc); +#else + int32x4_t acc=vdupq_n_s32(0); + for(;i+32<=I;i+=32){ + uint8x16_t by=vld1q_u8(w4+(i>>1)); + uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4)); + int8x16_t w0=vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q); + int8x16_t w1=vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q); + int8x16_t x0=vld1q_s8(x+i), x1=vld1q_s8(x+i+16); + int16x8_t p=vmull_s8(vget_low_s8(w0),vget_low_s8(x0)); + p=vmlal_s8(p,vget_high_s8(w0),vget_high_s8(x0)); + acc=vpadalq_s16(acc,p); + p=vmull_s8(vget_low_s8(w1),vget_low_s8(x1)); + p=vmlal_s8(p,vget_high_s8(w1),vget_high_s8(x1)); + acc=vpadalq_s16(acc,p); + } + sum=vaddvq_s32(acc); +#endif +#elif defined(__VSX__) + const __vector unsigned char m4v=vec_splats((unsigned char)0x0F); + const __vector unsigned char sh4=vec_splats((unsigned char)4); + const __vector signed char b8v=vec_splats((signed char)8); + const __vector signed char vz=vec_splats((signed char)0); + __vector signed int acc=vec_splats(0); + for(;i+32<=I;i+=32){ + __vector unsigned char by=vec_xl(0,w4+(i>>1)); + __vector unsigned char lo=vec_and(by,m4v), hi=vec_sr(by,sh4); + __vector signed char w0=vec_sub((__vector signed char)vec_mergeh(lo,hi),b8v); + __vector signed char w1=vec_sub((__vector signed char)vec_mergel(lo,hi),b8v); + __vector signed char x0=vec_xl(0,(const signed char*)(x+i)); + __vector signed char x1=vec_xl(0,(const signed char*)(x+i+16)); + __vector __bool char n0=vec_cmplt(w0,vz), n1=vec_cmplt(w1,vz); + acc=vec_msum(vec_sel(x0,vec_sub(vz,x0),n0), + (__vector unsigned char)vec_sel(w0,vec_sub(vz,w0),n0),acc); + acc=vec_msum(vec_sel(x1,vec_sub(vz,x1),n1), + (__vector unsigned char)vec_sel(w1,vec_sub(vz,w1),n1),acc); + } + sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3); +#endif + for(;i+1>1]; sum+=((int)(b&0xF)-8)*x[i]+((int)(b>>4)-8)*x[i+1]; } + if(i>1]; sum+=((int)(b&0xF)-8)*x[i]; } + return sum; +} + +/* ---- ARM i8mm SMMLA tiled kernels ---------------------------------------- */ +#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8) +static inline int32x4_t mm_tile16(int32x4_t acc, int8x16_t wo, int8x16_t wo1, + int8x16_t xs, int8x16_t xs1){ + acc=vmmlaq_s32(acc, vcombine_s8(vget_low_s8(wo), vget_low_s8(wo1)), + vcombine_s8(vget_low_s8(xs), vget_low_s8(xs1))); + return vmmlaq_s32(acc, vcombine_s8(vget_high_s8(wo), vget_high_s8(wo1)), + vcombine_s8(vget_high_s8(xs), vget_high_s8(xs1))); +} +static void matmul_q_idot_mm(float *y, const int8_t *xq, const float *sx, const int8_t *q, + const float *scale, int S, int I, int O){ + #pragma omp parallel for schedule(static) + for(int o=0;o<(O&~1);o+=2){ + const int8_t *wo=q+(int64_t)o*I, *wo1=q+(int64_t)(o+1)*I; + float sc0=scale[o], sc1=scale[o+1]; + for(int s=0;s<(S&~1);s+=2){ + const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I; + int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0; + for(;i+64<=I;i+=64){ + a0=mm_tile16(a0,vld1q_s8(wo+i), vld1q_s8(wo1+i), vld1q_s8(xs+i), vld1q_s8(xs1+i)); + a1=mm_tile16(a1,vld1q_s8(wo+i+16),vld1q_s8(wo1+i+16),vld1q_s8(xs+i+16),vld1q_s8(xs1+i+16)); + a2=mm_tile16(a2,vld1q_s8(wo+i+32),vld1q_s8(wo1+i+32),vld1q_s8(xs+i+32),vld1q_s8(xs1+i+32)); + a3=mm_tile16(a3,vld1q_s8(wo+i+48),vld1q_s8(wo1+i+48),vld1q_s8(xs+i+48),vld1q_s8(xs1+i+48)); + } + for(;i+16<=I;i+=16) + a0=mm_tile16(a0,vld1q_s8(wo+i),vld1q_s8(wo1+i),vld1q_s8(xs+i),vld1q_s8(xs1+i)); + int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); + int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1); + int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3); + for(;i>1)), byo1=vld1q_u8(wo1+(i>>1)); + uint8x16_t cyo=vld1q_u8(wo+(i>>1)+16), cyo1=vld1q_u8(wo1+(i>>1)+16); + uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4)); + uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4)); + uint8x16x2_t ko =vzipq_u8(vandq_u8(cyo, m4q), vshrq_n_u8(cyo, 4)); + uint8x16x2_t ko1=vzipq_u8(vandq_u8(cyo1,m4q), vshrq_n_u8(cyo1,4)); + a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q), + vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q), + vld1q_s8(xs+i), vld1q_s8(xs1+i)); + a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q), + vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q), + vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16)); + a2=mm_tile16(a2, vsubq_s8(vreinterpretq_s8_u8(ko.val[0]),b8q), + vsubq_s8(vreinterpretq_s8_u8(ko1.val[0]),b8q), + vld1q_s8(xs+i+32), vld1q_s8(xs1+i+32)); + a3=mm_tile16(a3, vsubq_s8(vreinterpretq_s8_u8(ko.val[1]),b8q), + vsubq_s8(vreinterpretq_s8_u8(ko1.val[1]),b8q), + vld1q_s8(xs+i+48), vld1q_s8(xs1+i+48)); + } + for(;i+32<=I;i+=32){ + uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1)); + uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4)); + uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4)); + a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q), + vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q), + vld1q_s8(xs+i), vld1q_s8(xs1+i)); + a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q), + vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q), + vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16)); + } + int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); + int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1); + int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3); + for(;i+1>1], bo1=wo1[i>>1]; + int a0=(int)(bo&0xF)-8, a1=(int)(bo>>4)-8, b0=(int)(bo1&0xF)-8, b1=(int)(bo1>>4)-8; + int u0=xs[i],u1=xs[i+1],v0=xs1[i],v1=xs1[i+1]; + d00+=a0*u0+a1*u1; d01+=a0*v0+a1*v1; d10+=b0*u0+b1*u1; d11+=b0*v0+b1*v1; } + if(i>1], bo1=wo1[i>>1]; + int a0=(int)(bo&0xF)-8, b0=(int)(bo1&0xF)-8; + d00+=a0*xs[i]; d01+=a0*xs1[i]; d10+=b0*xs[i]; d11+=b0*xs1[i]; } + y[(int64_t)s*O+o] =(float)d00*sc0*sx[s]; + y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s]; + y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1]; + y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1]; + } + if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I; + y[(int64_t)s*O+o] =(float)dot_i4i8(wo, xs,I)*sc0*sx[s]; + y[(int64_t)s*O+(o+1)]=(float)dot_i4i8(wo1,xs,I)*sc1*sx[s]; } + } + if(O&1){ int o=O-1; const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o]; + #pragma omp parallel for schedule(static) + for(int s=0;s=2){ matmul_q_idot_mm(y,xq,sx,q,scale,S,I,O); return; } +#endif + #pragma omp parallel for schedule(static) + for(int o=0;o=2){ matmul_i4_idot_mm(y,xq,sx,q4,scale,S,I,O); return; } +#endif + #pragma omp parallel for schedule(static) + for(int o=0;oplanar repack. */ +static void planarize_i4_row(uint8_t *row, int I){ + uint8_t tmp[32]; + int nb=I/64; + for(int b=0;b>1]>>((src_lo&1)*4))&0xF; + uint8_t nib_hi=(blk[src_hi>>1]>>((src_hi&1)*4))&0xF; + tmp[k]=(uint8_t)(nib_lo|(nib_hi<<4)); + } + memcpy(blk,tmp,32); + } +} +static void planarize_i4(uint8_t *q4, int O, int I){ + int rb=(I+1)/2; + #pragma omp parallel for schedule(static) + for(int o=0;o>1))); + __m256i b1=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)+32)); + a0=coli_dpbusd256(a0,_mm256_and_si256(b0,m4), + _mm256_loadu_si256((const __m256i*)(x+i))); + a1=coli_dpbusd256(a1,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4), + _mm256_loadu_si256((const __m256i*)(x+i+32))); + a2=coli_dpbusd256(a2,_mm256_and_si256(b1,m4), + _mm256_loadu_si256((const __m256i*)(x+i+64))); + a3=coli_dpbusd256(a3,_mm256_and_si256(_mm256_srli_epi16(b1,4),m4), + _mm256_loadu_si256((const __m256i*)(x+i+96))); + } + __m256i acc=_mm256_add_epi32(_mm256_add_epi32(a0,a1),_mm256_add_epi32(a2,a3)); + for(;i+64<=I;i+=64){ + __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1))); + acc=coli_dpbusd256(acc,_mm256_and_si256(b0,m4), + _mm256_loadu_si256((const __m256i*)(x+i))); + acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4), + _mm256_loadu_si256((const __m256i*)(x+i+32))); + } + sum=hsum256_i32(acc); +#elif defined(__AVX2__) + const __m256i m4=_mm256_set1_epi8(0x0F); + const __m256i ones=_mm256_set1_epi16(1); + __m256i acc=_mm256_setzero_si256(); + for(;i+64<=I;i+=64){ + __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1))); + /* maddubs(u8, s8): u<=15, |x|<=127 -> coppia <= 3810, int16 sicuro */ + __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(b0,m4), + _mm256_loadu_si256((const __m256i*)(x+i))); + __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(b0,4),m4), + _mm256_loadu_si256((const __m256i*)(x+i+32))); + acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p0,ones)); + acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p1,ones)); + } + sum=hsum256_i32(acc); +#elif defined(__ARM_NEON) + int32x4_t acc=vdupq_n_s32(0); + for(;i+64<=I;i+=64){ + uint8x16_t b0=vld1q_u8(w4+(i>>1)), b1=vld1q_u8(w4+(i>>1)+16); + int8x16_t lo0=vreinterpretq_s8_u8(vandq_u8(b0,vdupq_n_u8(0x0F))); + int8x16_t lo1=vreinterpretq_s8_u8(vandq_u8(b1,vdupq_n_u8(0x0F))); + int8x16_t hi0=vreinterpretq_s8_u8(vshrq_n_u8(b0,4)); + int8x16_t hi1=vreinterpretq_s8_u8(vshrq_n_u8(b1,4)); +#if defined(__ARM_FEATURE_DOTPROD) + acc=vdotq_s32(acc,lo0,vld1q_s8(x+i)); + acc=vdotq_s32(acc,lo1,vld1q_s8(x+i+16)); + acc=vdotq_s32(acc,hi0,vld1q_s8(x+i+32)); + acc=vdotq_s32(acc,hi1,vld1q_s8(x+i+48)); +#else + int8x16_t xs0=vld1q_s8(x+i), xs1=vld1q_s8(x+i+16); + int8x16_t xs2=vld1q_s8(x+i+32), xs3=vld1q_s8(x+i+48); + int16x8_t m; + m=vmull_s8(vget_low_s8(lo0),vget_low_s8(xs0)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_high_s8(lo0),vget_high_s8(xs0)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_low_s8(lo1),vget_low_s8(xs1)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_high_s8(lo1),vget_high_s8(xs1)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_low_s8(hi0),vget_low_s8(xs2)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_high_s8(hi0),vget_high_s8(xs2)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_low_s8(hi1),vget_low_s8(xs3)); acc=vpadalq_s16(acc,m); + m=vmull_s8(vget_high_s8(hi1),vget_high_s8(xs3)); acc=vpadalq_s16(acc,m); +#endif + } + sum=vaddvq_s32(acc); +#endif + for(;i+64<=I;i+=64){ /* fallback scalare sui blocchi planari */ + const uint8_t *blk=w4+(i>>1); + for(int k=0;k<32;k++){ + sum+=(int32_t)(blk[k]&0xF)*x[i+k]; + sum+=(int32_t)(blk[k]>>4)*x[i+k+32]; + } + } + for(;i>1]; + sum+=(int32_t)(byte&0xF)*x[i]; + if(i+1>4)*x[i+1]; + } + return sum; +} + +/* ---- K1b (OPT-IN, IDOT_GS=1): IDOT planare A GRUPPI (fmt=4, gs%64==0) ----- + * Con gs=64 il gruppo di scala COINCIDE col blocco-piano da 64 elementi: il + * dot unsigned del blocco (2 dpbusd) -> int32 di gruppo, meno 8*somma(x) del + * gruppo, per la scala f32 del gruppo. Attivazioni int8 (stessa famiglia + * qrow_i8 del resto dell'IDOT): NON bit-identico al kernel f32 a gruppi -- + * per questo e' dietro flag, in attesa dell'ablazione. xsg = somme int32 + * per (riga, gruppo), calcolate dal chiamante in una passata esatta. + * EN: grouped planar IDOT, opt-in. With gs=64 the scale group IS the plane + * block; per-group unsigned dot minus 8*group-sum, times the group scale. + * int8 activations: not bit-identical to the f32 grouped kernel, hence the + * flag until the ablation blesses a default. + * + * Three execution shapes below, one contract: per (row, output) the float + * accumulation is the SAME sequence of fmaf((float)group_int, scale[g], a) + * in ascending g, and every group_int is an exact int32 — so the per-row + * path, the 1x4 row tile, and the AMX tile are bit-identical to each other + * and to the pure-C reference on every ISA. */ + +/* one planar 64-element block, unpacked once and shared across a row tile: + * lo nibbles = elements base..base+31 in order, hi = base+32..base+63. */ +#if defined(coli_dpbusd256) +static inline void i4p_blk256(const uint8_t *blk, __m256i *lo, __m256i *hi){ + const __m256i m4=_mm256_set1_epi8(0x0F); + __m256i b=_mm256_loadu_si256((const __m256i*)blk); + *lo=_mm256_and_si256(b,m4); + *hi=_mm256_and_si256(_mm256_srli_epi16(b,4),m4); +} +#endif +#if defined(__AVX512VNNI__) && defined(__AVX512BW__) +/* whole block in one zmm: lanes 0..31 = lo elements, 32..63 = hi — matches a + * straight 64-byte load of the activation block, so ONE dpbusd per block. */ +static inline __m512i i4p_blk512(const uint8_t *blk){ + const __m256i m4=_mm256_set1_epi8(0x0F); + __m256i b=_mm256_loadu_si256((const __m256i*)blk); + return _mm512_inserti64x4(_mm512_castsi256_si512(_mm256_and_si256(b,m4)), + _mm256_and_si256(_mm256_srli_epi16(b,4),m4),1); +} +#endif + +/* ---- K1c: AMX int8 tile kernel (Sapphire Rapids+, opt-in via the same + * IDOT_GS=1 family gate; AMX=0 kills it, AMX_S_MIN sets the row threshold). + * With gs a multiple of 64, K=64 tile-multiplies cover a scale group exactly: + * B tile = 16 output rows' int4 block unpacked once to SIGNED int8 (v-8, so + * tdpbssd returns d - 8*sum(x_group) directly — the same int32 the vector + * path computes as d_unsigned - 8*xsg), A tile = up to 16 activation rows. + * The unpack cost is paid once per (output tile, group) and amortized over + * every activation row — the multi-row reuse the pair-layout kernels lack. + * Linux-only arming: tile data needs an ARCH_REQ_XCOMP_PERM handshake. */ +#if defined(__AMX_INT8__) && defined(__AMX_TILE__) && defined(__AVX512F__) +#define COLI_HAVE_AMX_I4P 1 +#if defined(__linux__) +#include +#include +#elif defined(_WIN32) +#ifndef WIN32_LEAN_AND_MEAN +#define WIN32_LEAN_AND_MEAN +#endif +#include +#endif +static int coli_amx_state=-1; +static int coli_amx_s_min=8; +static int coli_amx_ok(void){ + if(coli_amx_state<0){ + const char *e=getenv("AMX"); + const char *sm=getenv("AMX_S_MIN"); if(sm&&atoi(sm)>0) coli_amx_s_min=atoi(sm); +#if defined(__linux__) + /* ARCH_REQ_XCOMP_PERM(XFEATURE_XTILEDATA): kernel >= 5.16 grants tile + * state per-process; without it the first tile op SIGILLs. */ + coli_amx_state=(e&&*e=='0')?0:(syscall(SYS_arch_prctl,0x1023,18)==0); +#elif defined(_WIN32) + /* Windows 11: tile data is an opt-in per-process XState feature. + * Dynamic lookup keeps older kernels building/running (arming just + * fails closed there). XSTATE_AMX_TILE_DATA = 18. */ + if(e&&*e=='0') coli_amx_state=0; + else { + HMODULE k32=GetModuleHandleA("kernel32.dll"); + typedef BOOL (WINAPI *coli_xstate_fn)(ULONG64); + coli_xstate_fn f=k32?(coli_xstate_fn)(void*)GetProcAddress(k32,"EnableProcessOptionalXStateFeatures"):NULL; + coli_amx_state=(f && f(1ull<<18))?1:0; + } +#else + (void)e; coli_amx_state=0; +#endif + if(coli_amx_state) + fprintf(stderr,"[K1c] AMX int8 tile kernel armed for gs64 tensors " + "(AMX=0 disables, engages at S>=%d)\n",coli_amx_s_min); + } + return coli_amx_state; +} +/* ldtilecfg layout (palette 1): tmm0=C [rows x 16 i32], tmm1=A [rows x 64 i8], + * tmm2=B [16 x 64 i8 VNNI]. rows<16 reconfigures for the last partial s-tile. */ +struct coli_tilecfg { uint8_t palette,start_row,rsvd[14]; uint16_t colsb[16]; uint8_t rows[16]; }; +static void coli_amx_cfg(int arows){ + struct coli_tilecfg c; memset(&c,0,sizeof c); c.palette=1; + c.rows[0]=(uint8_t)arows; c.colsb[0]=64; + c.rows[1]=(uint8_t)arows; c.colsb[1]=64; + c.rows[2]=16; c.colsb[2]=64; + _tile_loadconfig(&c); +} +static void matmul_i4p_gidot_amx(float *y, const int8_t *xq, const float *sx, + const uint8_t *q4, const float *scale, + int S, int I, int O16, int O, int gs){ + int rb=(I+1)/2, ng=(I+gs-1)/gs, bpg=gs/64; + #pragma omp parallel + { + float *acc=malloc((size_t)S*16*sizeof(float)); + if(!acc){ fprintf(stderr,"OOM: amx acc\n"); exit(1); } + int8_t bstage[4*1024] __attribute__((aligned(64))); /* bpg<=4 gated below */ + int32_t cbuf[16*16] __attribute__((aligned(64))); + float sclT[16]; + int cur=16; coli_amx_cfg(16); + #pragma omp for schedule(static) + for(int ot=0; ot>1; + for(int n=0;n<16;n++){ + const uint8_t *blk=q4+(int64_t)(ot+n)*rb+boff; + for(int k=0;k<32;k++){ + dst[(k>>2)*64+n*4+(k&3)] =(int8_t)((blk[k]&0xF)-8); + dst[((k+32)>>2)*64+n*4+((k+32)&3)]=(int8_t)((blk[k]>>4)-8); + } + } + } + for(int st=0; st>1)); + A0=_mm512_dpbusd_epi32(A0,wz,_mm512_loadu_si512((const void*)(x0+base))); + A1=_mm512_dpbusd_epi32(A1,wz,_mm512_loadu_si512((const void*)(x1+base))); + A2=_mm512_dpbusd_epi32(A2,wz,_mm512_loadu_si512((const void*)(x2+base))); + A3=_mm512_dpbusd_epi32(A3,wz,_mm512_loadu_si512((const void*)(x3+base))); + } + d0=_mm512_reduce_add_epi32(A0); d1=_mm512_reduce_add_epi32(A1); + d2=_mm512_reduce_add_epi32(A2); d3=_mm512_reduce_add_epi32(A3); +#elif defined(coli_dpbusd256) + __m256i A0=_mm256_setzero_si256(),A1=_mm256_setzero_si256(); + __m256i A2=_mm256_setzero_si256(),A3=_mm256_setzero_si256(); + for(int b=0;b>1),&lo,&hi); + A0=coli_dpbusd256(A0,lo,_mm256_loadu_si256((const __m256i*)(x0+base))); + A0=coli_dpbusd256(A0,hi,_mm256_loadu_si256((const __m256i*)(x0+base+32))); + A1=coli_dpbusd256(A1,lo,_mm256_loadu_si256((const __m256i*)(x1+base))); + A1=coli_dpbusd256(A1,hi,_mm256_loadu_si256((const __m256i*)(x1+base+32))); + A2=coli_dpbusd256(A2,lo,_mm256_loadu_si256((const __m256i*)(x2+base))); + A2=coli_dpbusd256(A2,hi,_mm256_loadu_si256((const __m256i*)(x2+base+32))); + A3=coli_dpbusd256(A3,lo,_mm256_loadu_si256((const __m256i*)(x3+base))); + A3=coli_dpbusd256(A3,hi,_mm256_loadu_si256((const __m256i*)(x3+base+32))); + } + d0=hsum256_i32(A0); d1=hsum256_i32(A1); + d2=hsum256_i32(A2); d3=hsum256_i32(A3); +#elif defined(__AVX2__) + const __m256i ones=_mm256_set1_epi16(1); + const __m256i m4=_mm256_set1_epi8(0x0F); + __m256i A0=_mm256_setzero_si256(),A1=_mm256_setzero_si256(); + __m256i A2=_mm256_setzero_si256(),A3=_mm256_setzero_si256(); + for(int b=0;b>1))); + __m256i lo=_mm256_and_si256(bb,m4); + __m256i hi=_mm256_and_si256(_mm256_srli_epi16(bb,4),m4); + /* maddubs(u8,s8): u<=15, |x|<=127 -> pair <= 3810, int16-safe */ + A0=_mm256_add_epi32(A0,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x0+base))),ones)); + A0=_mm256_add_epi32(A0,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x0+base+32))),ones)); + A1=_mm256_add_epi32(A1,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x1+base))),ones)); + A1=_mm256_add_epi32(A1,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x1+base+32))),ones)); + A2=_mm256_add_epi32(A2,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x2+base))),ones)); + A2=_mm256_add_epi32(A2,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x2+base+32))),ones)); + A3=_mm256_add_epi32(A3,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x3+base))),ones)); + A3=_mm256_add_epi32(A3,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x3+base+32))),ones)); + } + d0=hsum256_i32(A0); d1=hsum256_i32(A1); + d2=hsum256_i32(A2); d3=hsum256_i32(A3); +#else + d0=d1=d2=d3=0; + for(int b=0;b>1); + for(int k=0;k<32;k++){ + int32_t ul=(int32_t)(blk[k]&0xF), uh=(int32_t)(blk[k]>>4); + d0+=ul*x0[base+k]+uh*x0[base+k+32]; + d1+=ul*x1[base+k]+uh*x1[base+k+32]; + d2+=ul*x2[base+k]+uh*x2[base+k+32]; + d3+=ul*x3[base+k]+uh*x3[base+k+32]; + } + } +#endif + a0=fmaf((float)(d0-8*g0[g]),scl[g],a0); + a1=fmaf((float)(d1-8*g1[g]),scl[g],a1); + a2=fmaf((float)(d2-8*g2[g]),scl[g],a2); + a3=fmaf((float)(d3-8*g3[g]),scl[g],a3); + } + if(g*gs>1]; + int32_t u=(int32_t)((i&1)?(byte>>4):(byte&0xF)); + d0+=u*x0[i]; d1+=u*x1[i]; d2+=u*x2[i]; d3+=u*x3[i]; + } + a0=fmaf((float)(d0-8*g0[g]),scl[g],a0); + a1=fmaf((float)(d1-8*g1[g]),scl[g],a1); + a2=fmaf((float)(d2-8*g2[g]),scl[g],a2); + a3=fmaf((float)(d3-8*g3[g]),scl[g],a3); + } + y[(int64_t)s*O+o] =a0*sx[s]; + y[(int64_t)(s+1)*O+o]=a1*sx[s+1]; + y[(int64_t)(s+2)*O+o]=a2*sx[s+2]; + y[(int64_t)(s+3)*O+o]=a3*sx[s+3]; + } + for(; s>1)), + _mm512_loadu_si512((const void*)(xr+base))); + } + d=_mm512_reduce_add_epi32(acc); +#else + for(int b=0;b>1); + const int8_t *xb=xr+base; +#if defined(coli_dpbusd256) + __m256i lo,hi; i4p_blk256(blk,&lo,&hi); + __m256i acc=_mm256_setzero_si256(); + acc=coli_dpbusd256(acc,lo,_mm256_loadu_si256((const __m256i*)xb)); + acc=coli_dpbusd256(acc,hi,_mm256_loadu_si256((const __m256i*)(xb+32))); + d+=hsum256_i32(acc); +#elif defined(__AVX2__) + const __m256i m4=_mm256_set1_epi8(0x0F); + const __m256i ones=_mm256_set1_epi16(1); + __m256i bb=_mm256_loadu_si256((const __m256i*)blk); + __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(bb,m4), + _mm256_loadu_si256((const __m256i*)xb)); + __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(bb,4),m4), + _mm256_loadu_si256((const __m256i*)(xb+32))); + __m256i acc=_mm256_add_epi32(_mm256_madd_epi16(p0,ones), + _mm256_madd_epi16(p1,ones)); + d+=hsum256_i32(acc); +#else + for(int k=0;k<32;k++){ + d+=(int32_t)(blk[k]&0xF)*xb[k]; + d+=(int32_t)(blk[k]>>4)*xb[k+32]; + } +#endif + } +#endif + a=fmaf((float)(d-8*xg[g]),scl[g],a); + } + if(g*gs>1]; + d+=(int32_t)((i&1)?(byte>>4):(byte&0xF))*xr[i]; + } + a=fmaf((float)(d-8*xg[g]),scl[g],a); + } + y[(int64_t)s*O+o]=a*sx[s]; + } + } +} + +static void matmul_i4p_grouped_idot(float *y, const int8_t *xq, const float *sx, + const int32_t *xsg, const uint8_t *q4, + const float *scale, int S, int I, int O, int gs){ +#if defined(COLI_HAVE_AMX_I4P) + /* AMX takes the aligned bulk (full 16-output tiles, full groups only — + * the real gs64 checkpoints have I%gs==0); the vector path finishes any + * output remainder. Below the row threshold the B-tile unpack does not + * amortize and the vector tile is the better kernel. */ + if(coli_amx_ok() && S>=coli_amx_s_min && gs%64==0 && gs<=256 && I%gs==0 && O>=16){ + int O16=O&~15; + matmul_i4p_gidot_amx(y,xq,sx,q4,scale,S,I,O16,O,gs); + if(O16 + * bit-identico al path per-riga per associativita'. + * EN: 1x4 register tile — the weight block's load+mask cost is paid + * once per 4 activation rows. Integer sums: bit-identical to the + * per-row path by associativity. */ + const __m256i m4t=_mm256_set1_epi8(0x0F); + for(;s+4<=S;s+=4){ + const int8_t *x0=xq+(int64_t)s*I, *x1=x0+I, *x2=x1+I, *x3=x2+I; + __m256i a0=_mm256_setzero_si256(), a1=_mm256_setzero_si256(); + __m256i a2=_mm256_setzero_si256(), a3=_mm256_setzero_si256(); + int i=0; + for(;i+64<=I;i+=64){ + __m256i b =_mm256_loadu_si256((const __m256i*)(w+(i>>1))); + __m256i lo=_mm256_and_si256(b,m4t); + __m256i hi=_mm256_and_si256(_mm256_srli_epi16(b,4),m4t); + a0=coli_dpbusd256(a0,lo,_mm256_loadu_si256((const __m256i*)(x0+i))); + a0=coli_dpbusd256(a0,hi,_mm256_loadu_si256((const __m256i*)(x0+i+32))); + a1=coli_dpbusd256(a1,lo,_mm256_loadu_si256((const __m256i*)(x1+i))); + a1=coli_dpbusd256(a1,hi,_mm256_loadu_si256((const __m256i*)(x1+i+32))); + a2=coli_dpbusd256(a2,lo,_mm256_loadu_si256((const __m256i*)(x2+i))); + a2=coli_dpbusd256(a2,hi,_mm256_loadu_si256((const __m256i*)(x2+i+32))); + a3=coli_dpbusd256(a3,lo,_mm256_loadu_si256((const __m256i*)(x3+i))); + a3=coli_dpbusd256(a3,hi,_mm256_loadu_si256((const __m256i*)(x3+i+32))); + } + int32_t d0=hsum256_i32(a0), d1=hsum256_i32(a1); + int32_t d2=hsum256_i32(a2), d3=hsum256_i32(a3); + /* coda a coppie, unsigned: stessa identita' -8*xsum del per-riga */ + for(;i>1]; + d0+=(int32_t)(byte&0xF)*x0[i]; d1+=(int32_t)(byte&0xF)*x1[i]; + d2+=(int32_t)(byte&0xF)*x2[i]; d3+=(int32_t)(byte&0xF)*x3[i]; + if(i+1>4)*x0[i+1]; d1+=(int32_t)(byte>>4)*x1[i+1]; + d2+=(int32_t)(byte>>4)*x2[i+1]; d3+=(int32_t)(byte>>4)*x3[i+1]; + } + } + y[(int64_t)(s+0)*O+o]=(float)(d0-8*xsum[s+0])*sc*sx[s+0]; + y[(int64_t)(s+1)*O+o]=(float)(d1-8*xsum[s+1])*sc*sx[s+1]; + y[(int64_t)(s+2)*O+o]=(float)(d2-8*xsum[s+2])*sc*sx[s+2]; + y[(int64_t)(s+3)*O+o]=(float)(d3-8*xsum[s+3])*sc*sx[s+3]; + } +#endif + for(;s 0.0) return gb * 1e9; + static int noted = 0; + if (!noted) { + noted = 1; + fprintf(stderr, "[inkling] could not measure available RAM on this platform; " + "auto cache falls back to 16 experts/layer. Pass --cap to set it.\n"); + } return 0; -#endif } /* ---------- routed-expert slots: serial bookkeeping, parallel fills ---------- */ @@ -2198,15 +2193,20 @@ static void apply_rep_penalty(float *logit, int n, const int *hist, int nhist, f } } -/* reject a prompt that would overrun the served KV bound (CTX_MAX, default 8192). - * The refusal is the frame the gateway turns into a 400 context_length_exceeded - * (#506, #1381); free text here reached the client as a 500. One request is - * served at a time, so the returned buffer is only read before the next call. */ +/* Refuse only a prompt that does not fit the served KV bound (CTX_MAX, + * default 8192). max_tokens is a ceiling: coli chat's interactive default + * (16384) used to 400 every turn because 2 + 16384 > 8192. The refusal is + * the frame the gateway turns into a 400 context_length_exceeded (#506, + * #1381); free text here reached the client as a 500. One request is served + * at a time, so the returned buffer is only read before the next call. */ +static int ink_ctx_max(void) { + const char *cm = getenv("CTX_MAX"); + return cm ? atoi(cm) : 8192; +} static const char *prompt_reject(int np, int want) { static char message[96]; - const char *cm = getenv("CTX_MAX"); - int ctx_max = cm ? atoi(cm) : 8192; - if (np + want <= ctx_max) return NULL; + int ctx_max = ink_ctx_max(); + if (coli_serve_budget(np, want, ctx_max, 0) >= 0) return NULL; snprintf(message, sizeof(message), "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d", np, want, ctx_max); @@ -2323,8 +2323,19 @@ static int serve_one(Model *m, Tok *T, SReq *q) { int *ids = malloc((size_t)cap * sizeof(int)); int np = tok_encode(T, q->payload, q->plen, ids, cap); if (np <= 0) { coli_serve_write_error(stdout,q->id,"empty prompt"); free(ids); return 0; } - const char *bad = prompt_reject(np, q->max_tok); - if (bad) { coli_serve_write_error(stdout,q->id,bad); free(ids); return 0; } + int ctx_max = ink_ctx_max(); + int budget = coli_serve_budget(np, q->max_tok, ctx_max, q->logprobs > 0); + if (budget < 0) { + const char *bad = prompt_reject(np, q->max_tok); + coli_serve_write_error(stdout,q->id,bad ? bad : "CONTEXT_EXCEEDED"); + free(ids); return 0; + } + if (budget < q->max_tok) { + fprintf(stderr, "[serve] max_tokens %d clamped to %d (context %d - prompt %d); " + "raise CTX_MAX for longer answers\n", + q->max_tok, budget, ctx_max, np); + q->max_tok = budget; + } /* audio: every <|audio|> placeholder must have exactly one DMel frame */ int naud = q->alen / m->c.mel_bins; if (q->alen % m->c.mel_bins != 0 || audio_tok_count(m, ids, np) != naud) { @@ -2482,6 +2493,25 @@ static void serve_hwinfo(Model *m) { if (sscanf(ln, "MemTotal: %lf", &v) == 1) rt = v/1e6; if (sscanf(ln, "MemAvailable: %lf", &v) == 1) ra = v/1e6; } fclose(mi); } + if (rt <= 0.0 || ra <= 0.0) { + double t2 = 0, a2 = 0; + compat_meminfo_gb(&t2, &a2); + if (rt <= 0.0) rt = t2; + if (ra <= 0.0) ra = a2; + } +#ifdef _WIN32 + if (cores <= 0) { + SYSTEM_INFO si; + GetSystemInfo(&si); + cores = (int)si.dwNumberOfProcessors; + } +#endif +#ifdef __APPLE__ + if (!cpu[0]) { + size_t sl = sizeof(cpu); + if (sysctlbyname("machdep.cpu.brand_string", cpu, &sl, NULL, 0)) cpu[0] = 0; + } +#endif int ngpu = 0; double vram = 0; const char *gpu = ""; #ifdef COLI_CUDA diff --git a/c/json.h b/c/json.h index 649b92cae..3a938b67a 100644 --- a/c/json.h +++ b/c/json.h @@ -28,6 +28,7 @@ typedef struct { size_t acap, aoff; int depth; /* annidamento corrente: bound contro lo stack-overflow * da JSON malevolo tipo [[[[...]]]] (discesa ricorsiva) */ + int error; /* used by json_parse_checked; legacy parsing stays permissive */ } jparser; /* tetto di annidamento: gli header safetensors / config sono piatti (profondita' @@ -43,7 +44,13 @@ static char *j_dup(jparser *p, const char *b, int n) { return d; } -static void j_ws(jparser *p) { while (*p->s && isspace((unsigned char)*p->s)) p->s++; } +static void j_ws(jparser *p) { + while (*p->s && isspace((unsigned char)*p->s)) { + char c=*p->s++; + /* Preserve legacy consumption, but only JSON whitespace is valid. */ + if (c!=' ' && c!='\t' && c!='\r' && c!='\n') p->error=1; + } +} static jval *j_new(jtype t) { jval *v = (jval *)calloc(1, sizeof(jval)); @@ -52,12 +59,12 @@ static jval *j_new(jtype t) { static jval *j_parse_val(jparser *p); -static char *j_parse_str_raw(jparser *p) { +static char *j_parse_str_raw(jparser *p, int is_key) { /* SEC (GHSA-2qrj): fail closed if not actually at a quote. The old comment * "assume *p->s == '\"'" was violated on the object-key path, and the * unconditional p->s++ would step past the buffer's NUL terminator and scan * adjacent heap (OOB read leaking into tensor names). */ - if (*p->s != '"') return j_dup(p, "", 0); + if (*p->s != '"') { p->error = 1; return j_dup(p, "", 0); } p->s++; /* buffer su heap che CRESCE: niente troncamento silenzioso a 64KB (le stringhe * lunghe di tokenizer.json/config venivano tagliate) e niente 64KB di stack. */ @@ -67,6 +74,7 @@ static char *j_parse_str_raw(jparser *p) { if (!tmp) { fprintf(stderr, "OOM parsing JSON string\n"); exit(1); } } tmp[n++] = (char)(ch); }while(0) while (*p->s && *p->s != '"') { char c = *p->s++; + if ((unsigned char)c < 0x20) p->error = 1; if (c == '\\' && *p->s) { char e = *p->s++; switch (e) { @@ -75,9 +83,12 @@ static char *j_parse_str_raw(jparser *p) { case 'f': c = '\f'; break; case '/': c = '/'; break; case '\\': c = '\\'; break; case '"': c = '"'; break; case 'u': { /* \uXXXX -> codepoint UTF-8 (con coppie surrogate) */ - if (!p->s[0]||!p->s[1]||!p->s[2]||!p->s[3]) { c='?'; break; } /* \u troncato: non leggere oltre il NUL */ + if (!p->s[0]||!p->s[1]||!p->s[2]||!p->s[3]) { p->error=1; c='?'; break; } /* \u troncato: non leggere oltre il NUL */ + for (int i=0; i<4; i++) if (!isxdigit((unsigned char)p->s[i])) p->error=1; unsigned cp = (unsigned)strtoul((char[]){p->s[0],p->s[1],p->s[2],p->s[3],0}, NULL, 16); p->s += 4; + /* Checked keys must retain their identity in C-string lookups. */ + if (is_key && cp==0) p->error=1; if (cp >= 0xD800 && cp <= 0xDBFF && p->s[0]=='\\' && p->s[1]=='u' && p->s[2] && p->s[3] && p->s[4] && p->s[5]) { unsigned lo = (unsigned)strtoul((char[]){p->s[2],p->s[3],p->s[4],p->s[5],0}, NULL, 16); @@ -89,13 +100,13 @@ static char *j_parse_str_raw(jparser *p) { else { J_PUT(0xF0|(cp>>18)); J_PUT(0x80|((cp>>12)&0x3F)); J_PUT(0x80|((cp>>6)&0x3F)); J_PUT(0x80|(cp&0x3F)); } continue; } - default: c = e; break; + default: p->error = 1; c = e; break; } } J_PUT(c); } #undef J_PUT - if (*p->s == '"') p->s++; + if (*p->s == '"') p->s++; else p->error = 1; char *out = j_dup(p, tmp, (int)n); free(tmp); return out; } @@ -103,9 +114,9 @@ static char *j_parse_str_raw(jparser *p) { static jval *j_parse_val(jparser *p) { j_ws(p); char c = *p->s; - if (c == '"') { jval *v = j_new(J_STR); v->str = j_parse_str_raw(p); return v; } + if (c == '"') { jval *v = j_new(J_STR); v->str = j_parse_str_raw(p,0); return v; } if (c == '{') { - if (++p->depth > J_MAX_DEPTH) { p->depth--; return j_new(J_NULL); } + if (++p->depth > J_MAX_DEPTH) { p->error=1; p->depth--; return j_new(J_NULL); } p->s++; jval *v = j_new(J_OBJ); int cap = 8; v->keys = malloc(cap * sizeof(char*)); @@ -116,9 +127,9 @@ static jval *j_parse_val(jparser *p) { if (*p->s == '}') { p->s++; p->depth--; return v; } for (;;) { j_ws(p); - if (*p->s != '"') break; /* SEC (GHSA-2qrj): object key must be a quoted string; stop on malformed input */ - char *key = j_parse_str_raw(p); - j_ws(p); if (*p->s == ':') p->s++; + if (*p->s != '"') { p->error=1; break; } /* SEC (GHSA-2qrj): object key must be a quoted string; stop on malformed input */ + char *key = j_parse_str_raw(p,1); + j_ws(p); if (*p->s == ':') p->s++; else p->error=1; jval *val = j_parse_val(p); if (v->len == cap) { cap *= 2; char **nk = (char**)realloc(v->keys, cap*sizeof(char*)); @@ -131,13 +142,14 @@ static jval *j_parse_val(jparser *p) { j_ws(p); if (*p->s == ',') { p->s++; continue; } if (*p->s == '}') { p->s++; break; } + p->error = 1; break; } p->depth--; return v; } if (c == '[') { - if (++p->depth > J_MAX_DEPTH) { p->depth--; return j_new(J_NULL); } + if (++p->depth > J_MAX_DEPTH) { p->error=1; p->depth--; return j_new(J_NULL); } p->s++; jval *v = j_new(J_ARR); int cap = 8; v->kids = malloc(cap * sizeof(jval*)); @@ -154,6 +166,7 @@ static jval *j_parse_val(jparser *p) { j_ws(p); if (*p->s == ',') { p->s++; continue; } if (*p->s == ']') { p->s++; break; } + p->error = 1; break; } p->depth--; @@ -163,12 +176,28 @@ static jval *j_parse_val(jparser *p) { if (c == 'f' && !strncmp(p->s, "false", 5)) { p->s += 5; jval *v = j_new(J_BOOL); v->boolean = 0; return v; } if (c == 'n' && !strncmp(p->s, "null", 4)) { p->s += 4; return j_new(J_NULL); } /* numero */ - { char *end; double d = strtod(p->s, &end); p->s = end; jval *v = j_new(J_NUM); v->num = d; return v; } + { const char *start=p->s, *q=start; + if (*q=='-') q++; + if (*q=='0') q++; + else if (*q>='1' && *q<='9') { do { q++; } while (*q>='0' && *q<='9'); } + else p->error=1; + if (*q=='.') { + q++; if (*q<'0' || *q>'9') p->error=1; + while (*q>='0' && *q<='9') q++; + } + if (*q=='e' || *q=='E') { + q++; if (*q=='+' || *q=='-') q++; + if (*q<'0' || *q>'9') p->error=1; + while (*q>='0' && *q<='9') q++; + } + char *end; double d = strtod(start, &end); + if (end==start || end!=q) p->error=1; + p->s = end; jval *v = j_new(J_NUM); v->num = d; return v; } } /* API */ static jval *json_parse(const char *text, char **arena_out) { - jparser p = { text, NULL, 0, 0, 0 }; + jparser p = { text, NULL, 0, 0, 0, 0 }; jval *v = j_parse_val(&p); if (arena_out) *arena_out = p.arena; else free(p.arena); return v; @@ -195,4 +224,16 @@ static void json_free(jval *v) { free(v); } +/* An oracle must not accept a partial tree from a truncated/malformed file. + * Keep the existing API's permissive behavior for other engine consumers. */ +static jval *json_parse_checked(const char *text) { + if (!text) return NULL; + jparser p = { text, NULL, 0, 0, 0, 0 }; + jval *v = j_parse_val(&p); + j_ws(&p); + if (p.error || *p.s) { json_free(v); v=NULL; } + free(p.arena); + return v; +} + #endif diff --git a/c/kimi_k3.c b/c/kimi_k3.c index 5543ad615..054b0e0d4 100644 --- a/c/kimi_k3.c +++ b/c/kimi_k3.c @@ -117,6 +117,7 @@ #include "pin_pool.h" /* coli_pin_slots_wanted: quanti scatti tenere */ #include "hybrid_split.h" /* KV prefix reuse (shared) */ #include "serve_codec.h" +#include "serve_budget.h" #ifdef COLI_SEGMENT_ADAPTER #include "segment_runtime.h" #include "segment_adapters.h" @@ -189,7 +190,9 @@ typedef struct { /* ---------- routed-expert streaming (native MXFP4 from the HF shards) ---- */ typedef struct { int fd[6]; int64_t off[6]; int contig; } ERef; /* w1p w1s w2p w2s w3p w3s */ +#ifndef KIMI_K3_NO_MAIN static char g_k3_usage[2100]; /* /.coli_usage, or COLI_USAGE */ +#endif typedef struct { int eid; uint8_t *buf, *base; uint64_t used; int pinned; } Slot; /* pinned: seeded from .coli_usage at startup and never evicted. The LRU adapts to * THIS session; the pin knows the history of every session before it. Capped at @@ -1612,28 +1615,22 @@ static inline float situf_(float g, float u, float b1, float b2){ } #ifdef COLI_CUDA -/* CUDA apply for one expert, decode only (S==1). - * - * Same shape as the Vulkan path below and the CPU expert_apply above -- w1/w3, - * SiTU-GLU on the host, then w2 down -- but stateless: the routed tier streams, - * so there is nothing resident to keep a device handle for. Weights ride up - * with the call. - * - * Returns 0 with u untouched on ANY failure, so the caller falls through to the - * disk+CPU path exactly as it does when Vulkan declines. That is the contract - * vLLM's MXFP4 backends use too -- FlashInfer/AITER when they can, an emulation - * path when they cannot -- and it is what makes the fast path safe to attempt - * unconditionally. */ +/* Keep SiTU-GLU and intermediate activations on the GPU when supported. + * Older DLLs lack the optional fused entry point, so retain the three-matmul + * path as fallback. Accumulate into u only after a complete expert succeeds. */ static int cuda_expert_apply(Model *m, const uint8_t *w1p, const uint8_t *w1s, const uint8_t *w2p, const uint8_t *w2s, const uint8_t *w3p, const uint8_t *w3s, const float *z, float wk, float *u, float *gate, float *up, float *hz){ Cfg *c=&m->c; - if(!coli_cuda_matmul_mxfp4(gate,z,w1p,w1s,1,c->latent,c->moe_inter)) return 0; - if(!coli_cuda_matmul_mxfp4(up, z,w3p,w3s,1,c->latent,c->moe_inter)) return 0; - for(int i=0;imoe_inter;i++) gate[i]=situf_(gate[i],up[i],c->situ_b1,c->situ_b2); - if(!coli_cuda_matmul_mxfp4(hz,gate,w2p,w2s,1,c->moe_inter,c->latent)) return 0; + if(!coli_cuda_expert_mxfp4(hz,z,w1p,w1s,w3p,w3s,w2p,w2s, + 1,c->latent,c->moe_inter,c->situ_b1,c->situ_b2)) { + if(!coli_cuda_matmul_mxfp4(gate,z,w1p,w1s,1,c->latent,c->moe_inter)) return 0; + if(!coli_cuda_matmul_mxfp4(up, z,w3p,w3s,1,c->latent,c->moe_inter)) return 0; + for(int i=0;imoe_inter;i++) gate[i]=situf_(gate[i],up[i],c->situ_b1,c->situ_b2); + if(!coli_cuda_matmul_mxfp4(hz,gate,w2p,w2s,1,c->moe_inter,c->latent)) return 0; + } for(int i=0;ilatent;i++) u[i]+=wk*hz[i]; return 1; } @@ -2370,7 +2367,8 @@ static int sample_tok(const float *lo, int V, float temp, float top_p){ * (K3_THINK=0 opens directly = non-thinking mode). The model then * closes think, opens response, and finishes with <|end_of_msg|> (the eos). */ typedef struct { Tok *T; int *ids; int n, cap; - int sp_open, sp_close, sp_sep, sp_eom; } ChatB; + int sp_open, sp_close, sp_sep, sp_eom; + int cont; } ChatB; /* cont: the final turn was left open (continuation) */ static void cb_special(ChatB *b, int id){ if(b->n>=b->cap){ fprintf(stderr,"chat prompt too long\n"); exit(1); } b->ids[b->n++]=id; @@ -2398,7 +2396,7 @@ static int chat_build(Tok *T, const char *sys, const char *user, int thinking, int *ids, int cap, int *sp){ ChatB b={T,ids,0,cap, chat_special(T,"<|open|>"), chat_special(T,"<|close|>"), - chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>")}; + chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>"), 0}; if(b.sp_open<0||b.sp_close<0||b.sp_sep<0||b.sp_eom<0){ fprintf(stderr,"chat: XTML special tokens not in tokenizer.json\n"); exit(1); } sp[0]=b.sp_open; sp[1]=b.sp_close; sp[2]=b.sp_sep; sp[3]=b.sp_eom; @@ -2475,7 +2473,7 @@ static int chat_build_wire(Tok *T, const char *wire, int nwire, int *thinking, int *ids, int cap, int *sp){ ChatB b={T,ids,0,cap, chat_special(T,"<|open|>"), chat_special(T,"<|close|>"), - chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>")}; + chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>"), 0}; /* -2, not -1: the caller must be able to tell a bad payload from a snapshot * whose tokenizer has no XTML tokens. Serve reported both as "invalid K3 * chat payload", which sent at least one user hunting through a request @@ -2489,6 +2487,7 @@ static int chat_build_wire(Tok *T, const char *wire, int nwire, int *thinking, while(p \n -- the same + * body as A, but rendered as the trailing open turn (no + * <|close|>, no <|end_of_msg|>, and the caller appends no + * fresh generation cue). An old engine has no 'C' record: + * it falls to the 'M' branch below, fails to parse, and + * rejects the payload -- fail-closed, never miswired. */ + int nr=-1, nt=-1; + if(sscanf(p,"C %d %d",&nr,&nt)!=2||nr<0||nt<0||nl+1+nr+nt>end) return -1; + char *reason=malloc((size_t)nr+1), *text=malloc((size_t)nt+1); + if(!reason||!text){ fprintf(stderr,"OOM chat continuation\n"); exit(1); } + memcpy(reason,nl+1,(size_t)nr); reason[nr]=0; + memcpy(text,nl+1+nr,(size_t)nt); text[nt]=0; + cb_open(&b,"message","assistant"); + if(nr){ cb_open(&b,"think",NULL); cb_text(&b,reason); cb_close(&b,"think"); } + cb_open(&b,"response",NULL); cb_text(&b,text); /* OPEN: no close, no eom */ + b.cont=1; + free(reason); free(text); p=nl+1+nr+nt; continue; + } if(*p=='Y'){ /* typed system message (#1143): tool-declare / tool-choice */ int ntp=-1, nb=-1; if(sscanf(p,"Y %d %d",&ntp,&nb)!=2||ntp<1||ntp>64||nb<0||nl+1+ntp+nb>end) return -1; @@ -2601,8 +2619,10 @@ static int chat_build_wire(Tok *T, const char *wire, int nwire, int *thinking, chat_message(&b,r,text,!strcmp(r,"assistant")); free(text); p=nl+1+nb; } - cb_open(&b,"message","assistant"); - cb_open(&b,*thinking?"think":"response",NULL); + if(!b.cont){ /* a continuation already emitted the open final turn */ + cb_open(&b,"message","assistant"); + cb_open(&b,*thinking?"think":"response",NULL); + } return b.n; } @@ -3053,12 +3073,24 @@ static int serve_one(Model *m, Tok *T, ServeReq *q){ np+=tok_encode(T,q->payload,q->plen,ids+np,cap-np); } int max_ctx=getenv("K3_MAXT")?atoi(getenv("K3_MAXT")):8192; - if(np<1||(int64_t)np+q->max_tok>max_ctx){ /* SEC (GHSA-gf38): int64 so np+max_tok can't wrap negative */ - char message[160]; - snprintf(message,sizeof(message), - "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d", - np,q->max_tok,max_ctx); - coli_serve_write_error(stdout,q->id,message); free(ids); return 0; + int budget=coli_serve_budget(np,q->max_tok,max_ctx,q->logprobs>0); + if(budget<0){ + if(np<1){ + coli_serve_write_error(stdout,q->id,"EMPTY_PROMPT"); + }else{ + char message[160]; + snprintf(message,sizeof(message), + "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d", + np,q->max_tok,max_ctx); + coli_serve_write_error(stdout,q->id,message); + } + free(ids); return 0; + } + if(budgetmax_tok){ + fprintf(stderr,"[serve] max_tokens %d clamped to %d (context %d - prompt %d); " + "raise K3_MAXT for longer answers\n", + q->max_tok,budget,max_ctx,np); + q->max_tok=budget; } coli_serve_write_accept(stdout,q->id,np); /* Declare the structured sideband before any generated DATA. Even an diff --git a/c/olmoe.c b/c/olmoe.c index 01f929690..d56b606d1 100644 --- a/c/olmoe.c +++ b/c/olmoe.c @@ -18,6 +18,11 @@ * PILOT_EVICT_GUARD=0/1 : 1=enable LFRU prefetch eviction guard (default), 0=disable * EXPERT_DROP=0/1: 1=fadvise(DONTNEED) after each expert read (old behaviour, * for RAM-tight boxes); 0=keep pages cached (default) + * ROUTE_TRACE=: log every routing decision (one line per moe call, + * position and layer: " : ...") + * for offline analysis — tools/route_pairs.py, + * tools/route_coupling_report.py, tools/residency_sim.py. + * Measurement only: it cannot change which experts run. * (expert queue is sorted by eid for SSD read locality) */ #define _GNU_SOURCE @@ -41,6 +46,7 @@ #include "kv_prefix.h" #include "pin_pool.h" /* piu scatti annidati */ /* riuso del prefisso tra turni (shared) */ #include "serve_codec.h" +#include "serve_budget.h" #ifdef COLI_SEGMENT_ADAPTER #include "segment_runtime.h" #include "segment_adapters.h" @@ -127,6 +133,7 @@ typedef struct { } Model; static pthread_mutex_t g_pilot_mx = PTHREAD_MUTEX_INITIALIZER; +static pthread_cond_t g_pilot_cv = PTHREAD_COND_INITIALIZER; /* broadcast on every publish */ static struct { int l, e; } pilot_q[4096]; static volatile unsigned pilot_r = 0, pilot_w = 0; static Model *pilot_m = NULL; @@ -190,6 +197,28 @@ static void cache_publish(Model *m, int layer, Slot *s, int eid) { lc->slot_by_expert[eid] = (int)(s - lc->slots); } +/* A slot being read keeps the index entry of the expert it is loading, marked + * -(eid+2) as in colibri.c's ecache_reserve: lookups still miss and eviction + * still skips it (eid < 0), but a second loader of the same expert can see the + * read in flight instead of starting another one into another slot. */ +static void cache_reserve(Model *m, int layer, Slot *s, int eid) { + LCache *lc = &m->cache[layer]; + cache_hide(m, layer, s); + s->eid = -(eid + 2); + if (lc->slot_by_expert && eid >= 0 && eid < m->c.n_experts) + lc->slot_by_expert[eid] = (int)(s - lc->slots); +} + +/* Caller holds g_pilot_mx. */ +static int slot_in_flight(Model *m, int layer, int eid) { + if (layer < 0 || layer >= m->c.n_layers || eid < 0 || + eid >= m->c.n_experts) return 0; + LCache *lc = &m->cache[layer]; + if (!lc->slot_by_expert) return 0; + int i = lc->slot_by_expert[eid]; + return i >= 0 && i < lc->n && lc->slots[i].eid == -(eid + 2); +} + static void ensure_pilot_worker_started(Model *m) { if (!pilot_m) { pilot_m = m; @@ -318,6 +347,29 @@ static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) { return _mm_cvtsi128_si32(sum32); } #define HAVE_FAST_DOT_I8 1 +#elif defined(__SSE4_1__) +#include +#include "sse41_kernels.h" +/* Sandy Bridge-EP path: AVX 1.0 only, no FMA, no AVX-2. + * 16 int8 dot via two 8-wide SSE2 sign-extend + SSE4.1 madd pairs. + * Bit-for-bit identical to the AVX2 version above (just 2x 128-bit ops + * instead of 1x 256-bit op). NO FMA here -- this branch targets Sandy Bridge + * which has no FMA -- so use explicit mul+add for the inner accumulation. */ +static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) { + __m128i va_lo = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)a)); /* lower 8 int8 -> 8 int16 */ + __m128i vb_lo = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)b)); + __m128i va_hi = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)(a + 8))); /* upper 8 int8 -> 8 int16 */ + __m128i vb_hi = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)(b + 8))); + __m128i p_lo = _mm_madd_epi16(va_lo, vb_lo); /* 4 x int32 from 8 int16 pairs */ + __m128i p_hi = _mm_madd_epi16(va_hi, vb_hi); /* 4 x int32 from 8 int16 pairs */ + __m128i sum = _mm_add_epi32(p_lo, p_hi); + /* horizontal reduce 4 x int32 -> 1 x int32 */ + __m128i hi64 = _mm_unpackhi_epi64(sum, sum); + __m128i sum64 = _mm_add_epi32(sum, hi64); + __m128i hi32 = _mm_shuffle_epi32(sum64, _MM_SHUFFLE(2, 3, 0, 1)); + return _mm_cvtsi128_si32(_mm_add_epi32(sum64, hi32)); +} +#define HAVE_FAST_DOT_I8 1 #endif /* Test-only hook, compiled out of the shipping binary. * @@ -634,7 +686,16 @@ static void slot_ensure_allocated(Model *m, Slot *s) { s->pinned = 0; } +#ifdef COLI_CACHE_INDEX_TEST +/* Model-free tests stand in for the disk read, so they can hold a load open + * and count how many times each expert is read. */ +static void (*g_test_expert_load)(Model *m, int layer, int eid, Slot *s); +#endif + static void load_expert_merged(Model *m, int layer, int eid, Slot *s) { +#ifdef COLI_CACHE_INDEX_TEST + if (g_test_expert_load) { g_test_expert_load(m, layer, eid, s); return; } +#endif char nm[256], qsnm[256]; snprintf(nm, sizeof(nm), "model.layers.%d.mlp.experts.%d.merged_weight", layer, eid); snprintf(qsnm, sizeof(qsnm), "model.layers.%d.mlp.experts.%d.qs", layer, eid); @@ -681,6 +742,13 @@ static void expert_get(Model *m, int layer, int eid, Slot **out) { pthread_mutex_lock(&g_pilot_mx); ehit_mark(m, layer, eid); /* under the lock: the routing loop is parallel */ Slot *hit = slot_indexed(m, layer, eid); + /* The prefetcher usually reads the next layer's experts while this one + * computes, so a routed expert is often already on its way: wait for that + * read to publish rather than read the same bytes again into another slot. */ + while (!hit && slot_in_flight(m, layer, eid)) { + pthread_cond_wait(&g_pilot_cv, &g_pilot_mx); + hit = slot_indexed(m, layer, eid); + } if (hit) { m->hits++; hit->used = ++m->clock; *out = hit; if (m->last_access) m->last_access[layer * m->c.n_experts + eid] = m->clock; @@ -727,7 +795,7 @@ static void expert_get(Model *m, int layer, int eid, Slot **out) { s = &lc->slots[lru]; s->pinned = 0; } - cache_hide(m, layer, s); + cache_reserve(m, layer, s, eid); s->used = ++m->clock; pthread_mutex_unlock(&g_pilot_mx); @@ -739,6 +807,7 @@ static void expert_get(Model *m, int layer, int eid, Slot **out) { s->used = ++m->clock; if (m->last_access) m->last_access[layer * c->n_experts + eid] = m->clock; *out = s; + pthread_cond_broadcast(&g_pilot_cv); pthread_mutex_unlock(&g_pilot_mx); } @@ -917,11 +986,24 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { idx[kk] = best; val[kk] = pr[best]; } if (c->norm_topk) { float sm=0; for(int kk=0;kkhot_pinned && m->freq) { - uint32_t *freq_l = m->freq[layer]; - if (freq_l) for (int kk = 0; kk < K; kk++) if (idx[kk] >= 0) freq_l[idx[kk]]++; - } + /* IMPROVEMENT 2 activation heatmap AND the ROUTE_TRACE stream, in one + * call. The counters were the only thing this engine recorded, and it + * recorded them HERE, before pinning activates — rt_count keeps that + * placement exactly. The trace is the half olmoe never had: it emits a + * line per (moe call, position, layer), so tools/route_pairs.py, + * route_coupling_report.py and residency_sim.py can read this engine's + * routing the same way they read GLM's. Until now olmoe announced + * ROUTE_TRACE at startup and then wrote a zero-byte file, because + * rt_init() opens the stream but nothing here ever called rt_trace(): + * every consumer silently saw "no data" instead of an error. + * + * Only rt_route() is unconditional: it is a no-op for the counts when + * this engine has no counter row (the !hot_pinned guard below is + * unchanged) and a no-op for the trace when ROUTE_TRACE is unset, so a + * run without the variable behaves exactly as before. Measurement only, + * never the computation: idx[] and val[] are the ids and the + * post-normalisation gates the layer is about to apply. */ + if (!m->hot_pinned) rt_route(layer, s, idx, val, K); const float *xs = x + (int64_t)s*D; for (int kk = 0; kk < K; kk++) { Slot *e; expert_get(m, layer, idx[kk], &e); @@ -950,6 +1032,15 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { } } free(logits); free(g); free(u); free(hh); + /* Advance the trace call counter: once per moe() invocation, after all of + * its rows are traced. rt_trace_end() is a no-op when no stream is open. + * + * Outside the row loop on purpose. A batch of S == 0 traces no rows and + * must still consume a call id, or the ids stop being consecutive and + * residency_sim.py rejects the trace outright ("trace lacks advancing GLM + * call ids") rather than merging two forwards into one position space. GLM + * and glm53 advance theirs the same way. */ + rt_trace_end(); } /* PROF phases (#1449): wall time in attention, in the MoE blocks (expert @@ -1076,7 +1167,7 @@ static void pilot_realload(Model *m, int layer, int eid) { pthread_mutex_unlock(&g_pilot_mx); return; } - if (slot_indexed(m, layer, eid)) { + if (slot_indexed(m, layer, eid) || slot_in_flight(m, layer, eid)) { m->is_queued[layer * c->n_experts + eid] = 0; pthread_mutex_unlock(&g_pilot_mx); return; @@ -1113,7 +1204,7 @@ static void pilot_realload(Model *m, int layer, int eid) { s = &lc->slots[lru]; s->pinned = 0; } - cache_hide(m, layer, s); s->used = ++m->clock; + cache_reserve(m, layer, s, eid); s->used = ++m->clock; pthread_mutex_unlock(&g_pilot_mx); load_expert_merged(m, layer, eid, s); @@ -1124,6 +1215,7 @@ static void pilot_realload(Model *m, int layer, int eid) { s->used = ++m->clock; if (m->last_access) m->last_access[layer * c->n_experts + eid] = m->clock; m->is_queued[layer * c->n_experts + eid] = 0; + pthread_cond_broadcast(&g_pilot_cv); pthread_mutex_unlock(&g_pilot_mx); } @@ -1521,7 +1613,8 @@ static int serve_one(Model *m, Tok *T, SReq *q, int ctx_cap) { int *ids = malloc((size_t)cap * sizeof(int)); int np = tok_encode(T, q->payload, q->plen, ids, cap); if (np <= 0) { coli_serve_write_error(stdout, q->id, "empty prompt"); free(ids); return 0; } - if (np + q->max_tok > ctx_cap) { + int budget = coli_serve_budget(np, q->max_tok, ctx_cap, q->logprobs > 0); + if (budget < 0) { char message[128]; /* The frame the gateway turns into a 400 context_length_exceeded * (#506, #1381). Free text here reached the client as a 500. */ @@ -1530,6 +1623,12 @@ static int serve_one(Model *m, Tok *T, SReq *q, int ctx_cap) { np, q->max_tok, ctx_cap); coli_serve_write_error(stdout, q->id, message); free(ids); return 0; } + if (budget < q->max_tok) { + fprintf(stderr, "[serve] max_tokens %d clamped to %d (context %d - prompt %d); " + "raise CTX for longer answers\n", + q->max_tok, budget, ctx_cap, np); + q->max_tok = budget; + } g_temp = q->temp; g_nuc = q->top_p; /* A chat client resends the whole transcript every turn. If this prompt * begins with the ids the cache was built from, those keys and values ARE diff --git a/c/openai_server.py b/c/openai_server.py index 42fe36354..488aefcec 100644 --- a/c/openai_server.py +++ b/c/openai_server.py @@ -110,6 +110,8 @@ def _engine_error(fields, message): class GenerationScheduler: """Bounded FIFO admission for the engine's independent KV contexts.""" + _buckets = (0.001, 0.01, 0.05, 0.1, 0.5, 1, 5, 10, 30, 60, 300, math.inf) + def __init__(self, max_queue=8, queue_timeout=300, capacity=1): if max_queue < 0: raise ValueError("max_queue cannot be negative") @@ -127,9 +129,13 @@ def __init__(self, max_queue=8, queue_timeout=300, capacity=1): self.closed = False self.admitted = 0 self.completed = 0 + self.failed = 0 self.rejected = 0 self.timed_out = 0 self.cancelled = 0 + self.timings = {name: {"sum": 0.0, "buckets": [0] * len(self._buckets)} + for name in ("queue_wait_seconds", "slot_duration_seconds", + "first_output_seconds", "engine_call_seconds")} @contextlib.contextmanager def admit(self, cancelled=None, slot=None): @@ -140,7 +146,7 @@ def admit(self, cancelled=None, slot=None): if self.closed: raise APIError(503, "The inference scheduler is shutting down.", None, "scheduler_closed", "server_error") - if (self.active >= self.capacity or self.queue) and len(self.queue) >= self.max_queue: + if self._available_slot(slot) is None and len(self.queue) >= self.max_queue: self.rejected += 1 raise APIError(429, "The inference queue is full.", None, "queue_full", "rate_limit_error", {"Retry-After": "1"}) @@ -152,24 +158,6 @@ def admit(self, cancelled=None, slot=None): self.condition.notify_all() raise APIError(503, "The inference scheduler is shutting down.", None, "scheduler_closed", "server_error") - available = min(self.free_slots) if slot is None and self.free_slots else slot - # (#B2) Admit as soon as our target slot is free AND no strictly-earlier - # waiter also wants it (an earlier waiter "wants" it if it is any-slot or - # pinned to the same slot). This replaces the old strict FIFO-head rule, - # which let a head pinned to a busy slot block every request behind it — - # even ones targeting a currently-free slot (head-of-line blocking). - # ponytail: O(queue) scan per wakeup — negligible at the default max_queue; - # switch to per-slot wait sets if max_queue is ever raised to thousands. - can_admit = available in self.free_slots - if can_admit: - for t2, s2 in self.queue: - if t2 is ticket: - break - if s2 is None or s2 == available: - can_admit = False - break - if can_admit: - break if cancelled and cancelled(): self.queue.remove(entry) self.cancelled += 1 @@ -182,37 +170,100 @@ def admit(self, cancelled=None, slot=None): self.condition.notify_all() raise APIError(429, "Timed out waiting for the inference engine.", None, "queue_timeout", "rate_limit_error", {"Retry-After": "1"}) + available = self._available_slot(slot, ticket) + if available is not None: + break self.condition.wait(min(remaining, 0.25)) self.queue.remove(entry) self.free_slots.remove(available) self.active += 1 self.admitted += 1 - wait_seconds = time.monotonic() - queued_at - cancelled_after_admission = False + admitted_at = time.monotonic() + wait_seconds = admitted_at - queued_at + self._observe("queue_wait_seconds", wait_seconds) + outcome = "failed" try: yield wait_seconds, available + outcome = "completed" except ClientCancelled: - cancelled_after_admission = True + outcome = "cancelled" raise finally: with self.condition: self.active -= 1 self.free_slots.add(available) - if cancelled_after_admission: - self.cancelled += 1 - else: - self.completed += 1 + setattr(self, outcome, getattr(self, outcome) + 1) + self._observe("slot_duration_seconds", time.monotonic() - admitted_at) self.condition.notify_all() + def _available_slot(self, slot, ticket=None): + # Caller holds the condition lock. Pinned waiters reserve only their + # target; an older any-slot waiter has priority over every free slot. + candidates = self.free_slots.copy() if slot is None else self.free_slots & {slot} + for earlier_ticket, earlier_slot in self.queue: + if earlier_ticket is ticket: + break + if earlier_slot is None: + return None + candidates.discard(earlier_slot) + return min(candidates, default=None) + def snapshot(self): with self.condition: return {"active": self.active, "queued": len(self.queue), "capacity": self.capacity, "max_queue": self.max_queue, "queue_timeout_seconds": self.queue_timeout, - "admitted": self.admitted, "completed": self.completed, + "admitted": self.admitted, "completed": self.completed, "failed": self.failed, "rejected": self.rejected, "timed_out": self.timed_out, "cancelled": self.cancelled} + def _observe(self, name, seconds): + # Called with condition held. Cumulative buckets need no request history. + timing = self.timings[name] + timing["sum"] += seconds + for i, bound in enumerate(self._buckets): + if seconds <= bound: + timing["buckets"][i] += 1 + + def observe_timing(self, name, seconds): + with self.condition: + self._observe(name, seconds) + + def prometheus(self): + """One consistent, bounded snapshot; no prompt or request-ID labels.""" + gauges = {"active": "Currently admitted requests.", + "queued": "Requests waiting for a KV slot.", + "capacity": "Concurrent KV slots configured.", + "max_queue": "Maximum waiting requests configured."} + counters = {"admitted": "Requests admitted to a KV slot.", + "completed": "Admitted requests that returned normally.", + "failed": "Admitted requests that raised an error.", + "rejected": "Requests rejected because the queue was full.", + "timed_out": "Requests that timed out waiting for a slot.", + "cancelled": "Requests cancelled while queued or admitted."} + lines = [] + with self.condition: + for kind, fields in (("gauge", gauges), ("counter", counters)): + for field, help_text in fields.items(): + name = "colibri_scheduler_" + field + ("_total" if kind == "counter" else "") + value = len(self.queue) if field == "queued" else getattr(self, field) + lines.extend((f"# HELP {name} {help_text}", f"# TYPE {name} {kind}", + f"{name} {value}")) + for field, help_text in ( + ("queue_wait_seconds", "Queue wait of admitted requests only."), + ("slot_duration_seconds", "Slot occupancy of finished admitted requests, including errors and cancellation."), + ("first_output_seconds", "Engine-call start to first nonempty text or tool callback, excluding queue wait."), + ("engine_call_seconds", "Duration of finished engine generation calls, including errors and cancellation.")): + name = "colibri_scheduler_" + field + timing = self.timings[field] + lines.extend((f"# HELP {name} {help_text}", f"# TYPE {name} histogram")) + for bound, count in zip(self._buckets, timing["buckets"]): + label = "+Inf" if math.isinf(bound) else str(bound) + lines.append(f'{name}_bucket{{le="{label}"}} {count}') + lines.extend((f'{name}_sum {timing["sum"]}', + f'{name}_count {timing["buckets"][-1]}')) + return "\n".join(lines) + "\n" + def close(self): with self.condition: self.closed = True @@ -251,6 +302,63 @@ def content_text(content, param): _ARG_RE = re.compile(r"([^<]*)(.*?)", re.DOTALL) _NAME_RE = re.compile(r"\s*([A-Za-z0-9_.\-]+)") _TAG_RE = re.compile(r"|") + + +def _fallback_tool_preamble(tools): + """Tool declaration for a family with no native tool tokens. + + Mirrors the GLM-5.2 block because ``parse_tool_calls`` -- the parser these + families fall back to in ``parse_arch_tool_calls`` -- reads exactly that + wire format. Asking for a format the parser does not accept would produce + tool calls nobody can read back. + """ + out = ["You have access to the following functions. Call one only when it " + "is needed to answer the user.\n\n\n"] + for tool in tools: + fn = tool.get("function", tool) if isinstance(tool, dict) else {} + out.append(json.dumps(fn, ensure_ascii=False) + "\n") + out.append("\n\nTo call a function, reply with the call and nothing " + "else, in this exact format:\n" + BOX_START + "{function-name}" + "{arg-key}{arg-value}" + + BOX_END) + return "".join(out) + + +def _fallback_tool_calls(tool_calls, index): + """Render assistant tool_calls in the format parse_tool_calls() reads.""" + out = [] + for position, call in enumerate(tool_calls or []): + if not isinstance(call, dict): + raise APIError(400, "Each tool call must be an object.", + f"messages.{index}.tool_calls.{position}") + fn = call.get("function", call) + if not isinstance(fn, dict): + raise APIError(400, "`function` must be an object.", + f"messages.{index}.tool_calls.{position}.function") + name = fn.get("name") + if not isinstance(name, str) or not name: + raise APIError(400, "`function.name` must be a non-empty string.", + f"messages.{index}.tool_calls.{position}.function.name") + args = fn.get("arguments", "{}") + if isinstance(args, str): + try: + args = json.loads(args) if args else {} + except (json.JSONDecodeError, TypeError, ValueError): + raise APIError(400, "`function.arguments` must be a JSON object.", + f"messages.{index}.tool_calls.{position}.function.arguments") + out.append(BOX_START + name) + for key, value in (args or {}).items(): + rendered = value if isinstance(value, str) else json.dumps( + value, ensure_ascii=False) + out.append(f"{key}{rendered}") + out.append(BOX_END) + return "".join(out) + + +def _fallback_tool_result(message, index): + """Render a role:"tool" message as prose these templates can carry.""" + body = content_text(message.get("content"), f"messages.{index}.content") + return TR_OPEN + body + TR_CLOSE # A closing tag the model started but never finished ("K structure. Default OFF (never rewrites well-formed output). _SALVAGE = os.environ.get("COLI_TOOL_SALVAGE", "0") == "1" +# Families whose chat template has no tool syntax at all (OLMoE, Qwen3.6) refuse +# tools[] and role:"tool" rather than invent a format. COLI_TOOL_FALLBACK=1 opts +# into a prompt-injected translation for them: the declaration block, the prior +# assistant calls and the tool results are written as ordinary turns, in the +# same wire format parse_tool_calls() already reads back (#1378). Default OFF -- +# these models were never trained on tool syntax, so this trades a clean 400 for +# output the parser may or may not recognise. +_TOOL_FALLBACK = os.environ.get("COLI_TOOL_FALLBACK", "0") == "1" + def _tool_choice_name(tool_choice): """The tool name a dict `tool_choice` forces, or None. @@ -275,6 +392,21 @@ def _tool_choice_name(tool_choice): or tool_choice.get("name")) +def _tool_function(tool): + """The function object on a tools[] entry, or {} if it is missing or not an object. + + OpenAI dual spelling: {"function": {"name": ...}} or a bare function object. + .items() is taken only from a dict. Writing the name where the object goes + ({"type": "function", "function": "search"}) raised AttributeError in the GLM + and DeepSeek declaration blocks, and do_POST answered HTTP 500 "The colibri + engine failed to process the request." for a payload generation_options() + already has a 400 for. Same shape as the tool_choice fix (#1598): read the + member, then check it. + """ + fn = tool.get("function", tool) if isinstance(tool, dict) else {} + return fn if isinstance(fn, dict) else {} + + def _tool_param_order(tools): """name -> ordered param names (required first) from the request schema, for de-mangling.""" out = {} @@ -447,7 +579,7 @@ def _dsv4_tools_block(tools): """V4 tool-declaration block, rendered by the vendored reference template.""" schemas = [] for tool in (tools or []): - fn = tool.get("function", tool) if isinstance(tool, dict) else {} + fn = _tool_function(tool) # Gateway-side scrub: OpenAI clients attach routing hints the model # schema must not carry. schemas.append({k: v for k, v in fn.items() if k not in ("defer_loading", "strict")}) @@ -455,7 +587,7 @@ def _dsv4_tools_block(tools): def _dsv4_tool_calls(tool_calls): - """Render OpenAI-format tool_calls into a V4 DSML block (incl. the leading + """Render OpenAI-format tool_calls into a V4 DSML block (incl. the leading ).""" return v4_dsml.render_tool_calls(tool_calls) @@ -997,12 +1129,18 @@ def _k3_order_tool_results(messages): def render_chat_kimi(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): + tool_choice=None, add_generation_prompt=True): """Validated multi-turn K3 payload for the C engine. K3's rank-BPE makes ordinary-text segment boundaries part of the tokenizer contract. This private length-framed payload preserves roles, UTF-8 bytes, and message boundaries; kimi_k3.c constructs the native XTML tokens. + + add_generation_prompt=False continues a trailing assistant turn. Kimi frames turns + engine-side, so unlike the string renderers there is no terminator to drop here: the final + assistant turn is emitted as a `C` record (reasoning + text), which kimi_k3.c renders as + the open turn -- no <|close|>/<|end_of_msg|>, and no fresh generation cue. An engine that + predates the record rejects the payload rather than miswiring it. """ if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") @@ -1055,6 +1193,14 @@ def render_chat_kimi(messages, enable_thinking=False, reasoning_effort=None, too if role == "assistant": last_calls = calls or [] tool_index = 0 + if not add_generation_prompt and index == len(messages) - 1: + # Continuation: the trailing assistant turn is left OPEN. resolve_generation_prompt + # has already refused tools/tool_calls and a non-assistant trailing turn, so this is + # a plain assistant turn; the C record carries its reasoning (if any) and text, and + # kimi_k3.c renders it as the open turn with no cue. + r = reasoning or "" + parts.append(f"C {len(r.encode('utf-8'))} {len(text.encode('utf-8'))}\n{r}{text}") + continue if calls: if len(calls) > 64: raise APIError(400, "Too many tool calls in one message (max 64).", @@ -1083,13 +1229,17 @@ def render_chat_kimi(messages, enable_thinking=False, reasoning_effort=None, too def render_chat_v4(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): + tool_choice=None, add_generation_prompt=True): """DeepSeek V4's native multi-turn chat template. The target engine receives this as a raw prompt. Prior assistant turns end with the checkpoint's EOS marker; the final assistant marker selects the thinking or direct-answer prefix for the new turn. + add_generation_prompt=False continues a trailing assistant turn: the last assistant turn + is rendered open, i.e. without its closing EOS and with no cue, the position the model + occupies mid-turn. EOS is the terminator to drop here, as <|im_end|> is for ChatML. + Tool use follows the official DSML format (encoding/encoding_dsv4.py): tool schemas are declared on the first system/developer message, assistant tool calls are DSML blocks, and tool results are blocks merged into @@ -1163,7 +1313,7 @@ def render_chat_v4(messages, enable_thinking=False, reasoning_effort=None, tools effort = DSV4_REASONING_EFFORT.get(reasoning_effort, "low") if effort != "low": parts.append(DSV4_REASONING_EFFORT_PROMPTS[effort]) - for message in merged: + for m_index, message in enumerate(merged): role = message["role"] if role in ("system", "developer"): if role == "developer": @@ -1185,13 +1335,16 @@ def render_chat_v4(messages, enable_thinking=False, reasoning_effort=None, tools parts.append(message["content"]) if message.get("tool_calls"): parts.append(_dsv4_tool_calls(message["tool_calls"])) - parts.append(eos) - parts.extend((assistant, "" if enable_thinking else "")) + # A continued turn is the last message rendered open: no EOS, no cue below. + if add_generation_prompt or m_index != len(merged) - 1: + parts.append(eos) + if add_generation_prompt: + parts.extend((assistant, "" if enable_thinking else "")) return "".join(parts) def render_chat_olmoe(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): + tool_choice=None, add_generation_prompt=True): """OLMoE-Instruct's native chat_template (tokenizer_config.json): one bos_token, then per-message <|system|>/<|user|>/<|assistant|> turns each closed by a newline, prior assistant turns also closed by eos_token @@ -1199,20 +1352,32 @@ def render_chat_olmoe(messages, enable_thinking=False, reasoning_effort=None, to repurposed as this tokenizer's BOS/EOS marker), and a trailing "<|assistant|>\\n" generation prompt. No tool-call syntax and no thinking mode exist in this template, so both parameters are accepted but unused. - """ + + add_generation_prompt=False continues a trailing assistant turn. The template closes + even the last assistant turn with eos_token, so the open-turn shape is that turn without + the eos and with no cue -- the same drop-the-terminator move as the ChatML families, with + eos_token as the terminator here.""" if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") - if tools or tool_choice not in (None, "none"): - raise APIError(400, "Tool use is not wired up for the OLMoE engine yet.", - "tools", "unsupported_parameter") + if tool_choice == "none": + tools = None + if (tools or tool_choice not in (None, "none")) and not _TOOL_FALLBACK: + raise APIError(400, "Tool use is not wired up for the OLMoE engine yet. " + "Set COLI_TOOL_FALLBACK=1 to opt into prompt-injected " + "tool translation.", "tools", "unsupported_parameter") boundary = "|||IP_ADDRESS|||" # bos_token == eos_token in this tokenizer parts = [boundary] + if tools and _TOOL_FALLBACK: + parts.append(f"<|system|>\n{_fallback_tool_preamble(tools)}\n") last = len(messages) - 1 for index, message in enumerate(messages): if not isinstance(message, dict): raise APIError(400, "Each message must be an object.", f"messages.{index}") role = message.get("role") - if role not in ("system", "developer", "user", "assistant"): + allowed = ("system", "developer", "user", "assistant") + if _TOOL_FALLBACK: + allowed += ("tool",) + if role not in allowed: raise APIError(400, f"Unsupported role {role!r}.", f"messages.{index}.role") raw = message.get("content") text = content_text(raw, f"messages.{index}.content") if raw is not None else "" @@ -1220,42 +1385,83 @@ def render_chat_olmoe(messages, enable_thinking=False, reasoning_effort=None, to parts.append(f"<|system|>\n{text}\n") elif role == "user": parts.append(f"<|user|>\n{text}\n") + elif role == "tool": + # No tool role in this template: the result rides in as a user turn. + parts.append(f"<|user|>\n{_fallback_tool_result(message, index)}\n") else: - parts.append(f"<|assistant|>\n{text}{boundary}") + calls = (_fallback_tool_calls(message.get("tool_calls"), index) + if _TOOL_FALLBACK else "") + # A continued turn is the last message rendered open: no eos, no cue. + terminator = "" if (not add_generation_prompt and index == last) else boundary + parts.append(f"<|assistant|>\n{text}{calls}{terminator}") if index != last: parts.append("\n") - parts.append("<|assistant|>\n") + if add_generation_prompt: + parts.append("<|assistant|>\n") return "".join(parts) def render_chat_qwen(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): + tool_choice=None, add_generation_prompt=True): """Text-only subset of Qwen3.6's chat_template: <|im_start|>role\\n ... <|im_end|>\\n frames, then the generation prompt. The official template opens a mandatory block after `<|im_start|>assistant\\n` — the model was never trained on the bare `assistant\\n` state, and greedy argmax there lands on an EOS special (measured: gen=0). With thinking disabled the template pre-closes the block instead; both branches are - mirrored here byte for byte.""" + mirrored here byte for byte. + + add_generation_prompt=False continues a trailing assistant turn. The template renders an + assistant turn AFTER the last user query with its block (an earlier one, + from history, has it stripped) -- so the open-turn shape is that think-form minus the + <|im_end|> terminator and with no cue, not the bare history form the loop emits otherwise. + ChatML's per-turn terminator is why the marker has to be dropped explicitly, as on qwen38.""" if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") - if tools or tool_choice not in (None, "none"): - raise APIError(400, "Tool use is not wired up for the qwen36 engine yet.", - "tools", "unsupported_parameter") + if tool_choice == "none": + tools = None + if (tools or tool_choice not in (None, "none")) and not _TOOL_FALLBACK: + raise APIError(400, "Tool use is not wired up for the qwen36 engine yet. " + "Set COLI_TOOL_FALLBACK=1 to opt into prompt-injected " + "tool translation.", "tools", "unsupported_parameter") parts = [] + if tools and _TOOL_FALLBACK: + parts.append("<|im_start|>system\n" + + _fallback_tool_preamble(tools) + "<|im_end|>\n") for index, message in enumerate(messages): if not isinstance(message, dict): raise APIError(400, "Each message must be an object.", f"messages.{index}") role = message.get("role") if role == "developer": role = "system" - if role not in ("system", "user", "assistant"): + allowed = ("system", "user", "assistant") + if _TOOL_FALLBACK: + allowed += ("tool",) + if role not in allowed: raise APIError(400, f"Unsupported role {role!r}.", f"messages.{index}.role") raw = message.get("content") text = content_text(raw, f"messages.{index}.content") if raw is not None else "" + if not add_generation_prompt and role == "assistant" and index == len(messages) - 1: + # Continued turn: the template gives a post-query assistant turn a + # block, then the model resumes the content. Match it, minus the terminator/cue. + reasoning = message.get("reasoning_content", "") + if not isinstance(reasoning, str): + raise APIError(400, "`reasoning_content` must be a string.", + f"messages.{index}.reasoning_content") + parts.append(f"<|im_start|>assistant\n\n{reasoning.strip()}\n\n\n" + f"{text.strip()}") + continue + if role == "tool": + # No tool role in this template: the result rides in as a user turn. + parts.append("<|im_start|>user\n" + + _fallback_tool_result(message, index) + "<|im_end|>\n") + continue + if role == "assistant" and _TOOL_FALLBACK: + text += _fallback_tool_calls(message.get("tool_calls"), index) parts.append(f"<|im_start|>{role}\n{text}<|im_end|>\n") - parts.append("<|im_start|>assistant\n") - parts.append("\n" if enable_thinking else "\n\n\n\n") + if add_generation_prompt: + parts.append("<|im_start|>assistant\n") + parts.append("\n" if enable_thinking else "\n\n\n\n") return "".join(parts) @@ -1372,8 +1578,14 @@ def parse_qwen38_tool_calls(reply, tools=None): def render_chat_qwen38(messages, enable_thinking=True, reasoning_effort=None, tools=None, - tool_choice=None): - """Text-only Qwen3.8 chat-template subset with native reasoning hints.""" + tool_choice=None, add_generation_prompt=True): + """Text-only Qwen3.8 chat-template subset with native reasoning hints. + + add_generation_prompt=False continues a trailing assistant turn. ChatML closes every + turn with <|im_end|>, so the open-turn shape is the past-turn render of that last message + MINUS its terminator, and no generation cue after it -- the position the model occupies + while writing an assistant turn. (GLM has no per-turn terminator, so there suppressing the + cue is enough; here the terminator has to be dropped too.)""" if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") if tool_choice in ("none",): @@ -1465,22 +1677,30 @@ def render_chat_qwen38(messages, enable_thinking=True, reasoning_effort=None, to rendered = f"\n{reasoning.strip()}\n\n\n{text}" if calls: rendered += _qwen38_tool_calls(calls, bool(text.strip()), index) - parts.append(f"<|im_start|>assistant\n{rendered}<|im_end|>\n") + # A continued turn is the last message rendered open: no <|im_end|>, no cue. + terminator = "" if (not add_generation_prompt and index == len(messages) - 1) \ + else "<|im_end|>\n" + parts.append(f"<|im_start|>assistant\n{rendered}{terminator}") continue parts.append(f"<|im_start|>{role}\n{text}<|im_end|>\n") - parts.append("<|im_start|>assistant\n") - parts.append("\n" if enable_thinking else "\n\n\n\n") + if add_generation_prompt: + parts.append("<|im_start|>assistant\n") + parts.append("\n" if enable_thinking else "\n\n\n\n") return "".join(parts) def render_chat_inkling(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None, audio_out=None): + tool_choice=None, audio_out=None, add_generation_prompt=True): """Text-only subset of Inkling's chat_template.jinja: role tokens with <|content_text|> parts and <|end_message|> terminators, an assistant <|content_model_end_sampling|> after each prior model turn, the thinking-effort hint appended after the messages (the template's fallback - branch), then <|message_model|> as the generation prompt.""" + branch), then <|message_model|> as the generation prompt. + + add_generation_prompt=False continues a trailing assistant turn: the last model turn is + rendered open -- without its <|end_message|> and the <|content_model_end_sampling|> that + close it, and with no cue. Those two markers are the terminator to drop here.""" if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") if tools or (tool_choice not in (None, "none")): @@ -1516,6 +1736,8 @@ def render_chat_inkling(messages, enable_thinking=False, reasoning_effort=None, if not effort_emitted and role not in ("system", "developer"): prompt.append(effort_str) effort_emitted = True + open_turn = (not add_generation_prompt and role == "assistant" + and index == len(messages) - 1) raw = message.get("content") if audio_out is not None and role == "user" and isinstance(raw, list): # multipart user content: text runs and audio clips become separate @@ -1530,26 +1752,34 @@ def render_chat_inkling(messages, enable_thinking=False, reasoning_effort=None, + "<|audio|>" * val + "<|audio_end|><|end_message|>") else: text = content_text(raw, f"messages.{index}.content") if raw is not None else "" - prompt.append(f"{rtok}<|content_text|>{text}<|end_message|>") - if role == "assistant": + # A continued turn is the last model message rendered open: no <|end_message|>. + terminator = "" if open_turn else "<|end_message|>" + prompt.append(f"{rtok}<|content_text|>{text}{terminator}") + if role == "assistant" and not open_turn: prompt.append("<|content_model_end_sampling|>") if not effort_emitted: # all-system edge case: fallback prompt.append(effort_str) - prompt.append("<|message_model|>") # add_generation_prompt - # Thinking off: prefill the content channel. Without this the model can still - # sample <|content_thinking|> as its first token (the effort hint is only a - # soft signal), open a reasoning block, and burn the whole token budget before - # reaching <|content_text|> — which the splitter then strips to an empty - # answer. Ending the prompt at <|message_model|><|content_text|> forces content - # mode; it is exactly the sequence every non-thinking turn is trained on. - if eff == 0.0: - prompt.append("<|content_text|>") + if add_generation_prompt: + prompt.append("<|message_model|>") # generation cue + # Thinking off: prefill the content channel. Without this the model can still + # sample <|content_thinking|> as its first token (the effort hint is only a + # soft signal), open a reasoning block, and burn the whole token budget before + # reaching <|content_text|> — which the splitter then strips to an empty + # answer. Ending the prompt at <|message_model|><|content_text|> forces content + # mode; it is exactly the sequence every non-thinking turn is trained on. + if eff == 0.0: + prompt.append("<|content_text|>") return "".join(prompt) def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): - """Render the text-only subset of the official GLM-5.2 chat template.""" + tool_choice=None, add_generation_prompt=True): + """Render the text-only subset of the official GLM-5.2 chat template. + + add_generation_prompt=False continues a trailing assistant turn. GLM has no per-turn + terminator (the next role token ends a turn), so the loop already renders that last message + as a past turn -- <|assistant|>{content} -- and suppressing the cue leaves + the prompt open on it, exactly as on glm53. Nothing to strip, unlike the ChatML families.""" if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") prompt = ["[gMASK]"] @@ -1583,7 +1813,7 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No "user query.\n\nYou are provided with function signatures within " "XML tags:\n\n") for tool in tools: - fn = tool.get("function", tool) if isinstance(tool, dict) else {} + fn = _tool_function(tool) clean = {k: v for k, v in fn.items() if k not in ("defer_loading", "strict")} prompt.append(json.dumps(clean, ensure_ascii=False) + "\n") prompt.append("\n\nFor each function call, output the function name and arguments " @@ -1622,8 +1852,17 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No args = json.loads(args) except (json.JSONDecodeError, TypeError): args = {} + if not isinstance(args, dict): + # `arguments` that is valid JSON but not an object ("[1,2]", + # "5", a bare list) reached .items() and raised + # AttributeError, which do_POST answers with HTTP 500. The + # same field is already tolerated when it does not parse at + # all, and every sibling renderer renders the call without + # arguments instead of failing; this is the one branch that + # was never completed. + args = {} prompt.append(BOX_START + (fn.get("name") or "")) - for key, value in (args or {}).items(): + for key, value in args.items(): prompt.append(f"{key}" + (value if isinstance(value, str) else json.dumps(value, ensure_ascii=False)) + "") @@ -1636,8 +1875,9 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No raise APIError(400, f"Unsupported message role: {role!r}.", f"messages.{index}.role", "unsupported_role") prev_tool = (role == "tool") - prompt.append("<|assistant|>" if enable_thinking else - "<|assistant|>") + if add_generation_prompt: + prompt.append("<|assistant|>" if enable_thinking else + "<|assistant|>") return "".join(prompt) @@ -1899,8 +2139,7 @@ def _glm53_tool_block(tools): somiglia a quello dell'addestramento non e' quello dell'addestramento.""" body = "".join(f"\n{_glm53_tool_json(tool)}\n\n" for tool in tools - if not (isinstance(tool, dict) - and (tool.get("function", tool) or {}).get("defer_loading"))) + if not _tool_function(tool).get("defer_loading")) return GLM53_TOOL_PREAMBLE + body + GLM53_TOOL_EPILOGUE @@ -1919,8 +2158,10 @@ def _glm53_tool_calls(calls): arguments = json.loads(arguments) except ValueError: arguments = {} + if not isinstance(arguments, dict): + arguments = {} # same gap as render_chat above pieces = [f"{name}"] - for key, value in (arguments or {}).items(): + for key, value in arguments.items(): rendered = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False) pieces.append(f"{key}{rendered}") pieces.append("") @@ -1929,7 +2170,7 @@ def _glm53_tool_calls(calls): def render_chat_glm53(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): + tool_choice=None, add_generation_prompt=True): """Render the text-only subset of the official GLM-5.3-Flash chat template. Not a variant of the GLM-5.2 renderer above, and the differences are not @@ -1944,7 +2185,7 @@ def render_chat_glm53(messages, enable_thinking=False, reasoning_effort=None, to so the existing parser needs nothing added for this family. The whole thing is pinned byte for byte against chat_template.jinja rendered - with jinja2 (tests/test_glm53_chat_template.py). Getting the prompt nearly + with jinja2 (tests/glm53_chat_template_harness.py). Getting the prompt nearly right is the failure mode worth guarding: the model answers either way. """ if not isinstance(messages, list) or not messages: @@ -2033,7 +2274,14 @@ def render_chat_glm53(messages, enable_thinking=False, reasoning_effort=None, to # su cui il modello e' addestrato" perche' il template lo scrive davanti a un # TURNO PASSATO senza ragionamento. E' vero per un turno passato e falso per il # prompt di generazione: la posizione da cui il modello scrive non e' mai quella. - prompt.append("<|assistant|>") + # + # add_generation_prompt=False non e' una forma nostra: e' l'altro ramo di questo stesso + # `if` nel template. Il prompt finisce allora sull'ultimo turno assistant reso come + # turno PASSATO -- seguito dal contenuto -- e il modello lo prosegue + # invece di aprirne uno nuovo. Chi non chiede la prosecuzione non vede differenza: + # il ramo True e' invariato, byte per byte, ed e' quello che il test confronta. + if add_generation_prompt: + prompt.append("<|assistant|>") return "".join(prompt) @@ -2071,7 +2319,7 @@ def _dsv41_tools_block(tools): """V4.1 tool-declaration block, rendered by the vendored reference template.""" schemas = [] for tool in (tools or []): - fn = tool.get("function", tool) if isinstance(tool, dict) else {} + fn = _tool_function(tool) # Gateway-side scrub: OpenAI clients attach routing hints the model # schema must not carry. clean = {k: v for k, v in fn.items() if k not in ("defer_loading", "strict")} @@ -2139,13 +2387,18 @@ def _dsv41_merge_turns(messages): def render_chat_dsv41(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None): + tool_choice=None, add_generation_prompt=True): """encoding.py _encode_messages_text for one turn. Tool use follows the checkpoint's own DSML format (encoding/encoding.py, vendored in v41_dsml.py): schemas are declared at the end of the system message, assistant tool calls are <|DSML| calls> blocks, and tool results are blocks merged into the following user turn. + + add_generation_prompt=False continues a trailing assistant turn: the last turn is rendered + open -- its block and content as a PAST turn, but without the closing + <|end▁of▁sentence|> and with no cue appended. That is the same drop-the-terminator move as + deepseek_v4, whose EOS this shares; the model resumes from the content it was handed. """ if not isinstance(messages, list) or not messages: raise APIError(400, "`messages` must be a non-empty array.", "messages") @@ -2205,27 +2458,157 @@ def render_chat_dsv41(messages, enable_thinking=False, reasoning_effort=None, to prompt.append(turn["content"]) if turn.get("tool_calls"): prompt.append(v41_dsml.render_tool_calls(turn["tool_calls"])) - prompt.append(DSV41_EOS) + # A continued turn is the last message rendered open: no EOS, no cue below. + if add_generation_prompt or index != len(turns) - 1: + prompt.append(DSV41_EOS) # the generation cue, exactly as render_message appends it after a user turn - prompt.append(DSV41_ASSISTANT) - prompt.append("" if enable_thinking and len(turns) - 1 >= last_user else "") + if add_generation_prompt: + prompt.append(DSV41_ASSISTANT) + prompt.append("" if enable_thinking and len(turns) - 1 >= last_user else "") return "".join(prompt) +# ---- continuing an unfinished assistant turn (COLI_CONTINUE_ASSISTANT) ---------------- +# A trailing `assistant` message means "continue writing this turn", not "here is a turn I +# already finished". The official template says exactly that, and says it in one place -- +# {%- if add_generation_prompt -%}<|assistant|>{{- '' -}}{%- endif -%} +# -- whose False branch every renderer in this file hard-codes to True. With the cue +# suppressed the prompt ends mid-turn, on the shape the template writes in front of a PAST +# assistant turn, which is a position the model saw all through training. +# +# That distinction is what makes this safe on GLM-5.3 specifically. #1327 measured that a +# CLOSED, EMPTY at the end of a prompt is out of distribution and the model +# keeps reasoning through it. The position here is a different one: followed +# by real content, i.e. the past-turn shape, which is why a continuation must carry text. +# +# llama.cpp needs no switch for this because it runs the checkpoint's jinja at request time, +# so `add_generation_prompt=False` costs it nothing. This gateway renders by hand, on purpose +# and for speed (tests/glm53_chat_template_harness.py says why), and the bill for that choice is +# exactly here: one template flag, one open-turn shape to derive per renderer. Each string +# renderer derives its own, pinned byte-for-byte against the checkpoint's template; +# CONTINUATION_FAMILIES is the set that has done so. Kimi K3 differs in WHERE its shape lives: +# its prompt is framed engine-side (render_chat_kimi hands a K3CHAT1 record to kimi_k3.c, which +# assembles the XTML tokens), so its open turn is a `C` record here plus a branch in that C path, +# pinned by tests/test_k3_chat_tools.c against the tiny tokenizer rather than by a template diff. + +# Families whose renderer implements the add_generation_prompt=False (open-turn) branch. A +# trailing assistant turn on a family NOT in this set falls through to the ordinary render +# (the cue is appended, exactly as before this existed) rather than erroring -- continuation is +# on by default, and a family without its open-turn shape yet must not start rejecting requests +# nobody opted into. Each renderer adds itself here in the same commit that derives its shape. +CONTINUATION_FAMILIES = {"glm53", "qwen38", "qwen36", "glm", "olmoe", "deepseek_v4", "inkling", + "kimi", "deepseek_v41"} + + +def resolve_generation_prompt(messages, body): + """Does this prompt end on a generation cue, or on an assistant turn to continue? + + Returns True for the ordinary case (append the cue) and False for a continuation, which is + the template's `add_generation_prompt=False`. + + Continuation is ON by default. A message list ending in a non-empty assistant turn already + says "continue me" -- the same contract as Anthropic's API -- and no OpenAI-compatible + client sends a trailing assistant turn by accident. It is deliberately NOT a request field: + a client would have to know colibri specifically to send one, and the clients that most + want this -- anything pointed at an OpenAI- or Anthropic-compatible URL -- send a message + list and nothing else. + + COLI_CONTINUE_ASSISTANT=0 is the off-switch, for a deployment that wants the old behaviour + (fold the trailing turn into a completed one and append a fresh cue). It is the only value + that turns this off; anything else, including unset, leaves it on. + + A family whose renderer has no open-turn shape yet (ARCH not in CONTINUATION_FAMILIES) + falls through to the ordinary render rather than erroring: continuation defaults on, so a + family added before its open-turn shape must not start rejecting trailing-assistant + requests that worked before. Every shipped family is in the set today, Kimi K3 included -- + its open turn is framed in kimi_k3.c (a `C` record), not derived in the renderer here. + """ + continuing = os.environ.get("COLI_CONTINUE_ASSISTANT", "1") != "0" + last = messages[-1] if isinstance(messages, list) and messages else None + if not (isinstance(last, dict) and last.get("role") == "assistant"): + return True + where = f"messages.{len(messages) - 1}" + if not continuing: + return True + if ARCH not in CONTINUATION_FAMILIES: + return True # open-turn shape not derived for this family yet -- render as before + if body.get("tools") or body.get("functions"): + raise APIError(400, "A continued assistant turn cannot be combined with `tools`: " + "the tool-call parsers read an assistant turn from its start, and a " + "continuation can end anywhere -- including inside a " + "block.", "tools", "unsupported_parameter") + if last.get("tool_calls"): + raise APIError(400, "A continued `assistant` message cannot carry `tool_calls`.", + f"{where}.tool_calls", "unsupported_value") + if len(messages) < 2: + raise APIError(400, "A continued `assistant` turn needs a preceding turn to " + "continue from.", "messages") + raw = last.get("content") + if isinstance(raw, list): # multimodal parts: only the text counts + text = "".join(part.get("text", "") for part in raw + if isinstance(part, dict) and part.get("type") == "text") + elif raw is None: + text = "" + elif isinstance(raw, str): + text = raw + else: + raise APIError(400, "Message content must be a string or an array of blocks.", + f"{where}.content") + if not text.strip(): + raise APIError(400, "A continued `assistant` turn needs text to continue. An empty " + "one ends the prompt on a closed, empty block, which " + "is the out-of-distribution position #1327 removed -- the model " + "reasons straight through it instead of answering.", + f"{where}.content", "invalid_value") + if text != text.rstrip(): + raise APIError(400, "A continued `assistant` turn cannot end with whitespace: the " + "template strips it, so the model would resume from different bytes " + "than the ones sent. Put the space at the start of what you expect " + "back instead.", f"{where}.content", "invalid_value") + return False + + def render_chat_for_arch(messages, enable_thinking=False, reasoning_effort=None, tools=None, - tool_choice=None, audio_out=None): - """Render a chat request with the active engine's native prompt contract.""" + tool_choice=None, audio_out=None, add_generation_prompt=True): + """Render a chat request with the active engine's native prompt contract. + + `add_generation_prompt=False` (a continued assistant turn) is implemented for the families + in CONTINUATION_FAMILIES. resolve_generation_prompt() passes any other family through with + the cue appended, so it never reaches here with the flag False; this stays as the backstop, + because silently appending a cue to a continuation is the exact failure this exists to remove. + """ + if not add_generation_prompt and ARCH not in CONTINUATION_FAMILIES: + raise APIError(400, f"Continuing an assistant turn is not implemented for {ARCH!r}.", + "messages", "unsupported_parameter") if ARCH == "inkling": return render_chat_inkling(messages, enable_thinking, reasoning_effort, tools, - tool_choice, audio_out=audio_out) - renderer = (render_chat_glm53 if ARCH == "glm53" else - render_chat_kimi if ARCH == "kimi" else - render_chat_qwen if ARCH == "qwen36" else - render_chat_qwen38 if ARCH == "qwen38" else - render_chat_v4 if ARCH == "deepseek_v4" else - render_chat_dsv41 if ARCH == "deepseek_v41" else - render_chat_olmoe if ARCH == "olmoe" else render_chat) - return renderer(messages, enable_thinking, reasoning_effort, tools, tool_choice) + tool_choice, audio_out=audio_out, + add_generation_prompt=add_generation_prompt) + if ARCH == "glm53": + return render_chat_glm53(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + if ARCH == "qwen38": + return render_chat_qwen38(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + if ARCH == "qwen36": + return render_chat_qwen(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + if ARCH == "glm": + return render_chat(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + if ARCH == "olmoe": + return render_chat_olmoe(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + if ARCH == "deepseek_v4": + return render_chat_v4(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + if ARCH == "kimi": + return render_chat_kimi(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt=add_generation_prompt) + if ARCH == "deepseek_v41": + return render_chat_dsv41(messages, enable_thinking, reasoning_effort, tools, + tool_choice, add_generation_prompt) + return render_chat(messages, enable_thinking, reasoning_effort, tools, tool_choice) # ---- Anthropic Messages API (#343) -------------------------------------------------------- @@ -2237,7 +2620,7 @@ def render_chat_for_arch(messages, enable_thinking=False, reasoning_effort=None, ANTHROPIC_LOCAL_SIGNATURE = "colibri-local" # opaque compatibility metadata, not a crypto proof -def starts_in_reasoning(enable_thinking): +def starts_in_reasoning(enable_thinking, add_generation_prompt=True): """Se l'uscita del modello comincia DENTRO al blocco di ragionamento. Dipende da come il prompt lo ha lasciato, e ogni famiglia lo lascia come @@ -2250,8 +2633,20 @@ def starts_in_reasoning(enable_thinking): ha un interruttore, render_chat_glm53 apre SEMPRE, e "thinking spento" vuol dire solo effort Low. L'uscita comincia dentro al blocco in ogni caso; partire in modalita' testo perche' il client ha detto False e' - esattamente il ragionamento incollato davanti alla risposta di #1278.""" - return enable_thinking or ARCH == "glm53" + esattamente il ragionamento incollato davanti alla risposta di #1278. + + Il turno proseguito (add_generation_prompt=False) e' il terzo stato, e non + lo dice l'interruttore: il prompt finisce sull'ultimo turno assistant reso + come turno PASSATO, quindi GIA' CHIUSO seguito dal + contenuto, col ragionamento acceso o spento che sia. Il modello riprende in + modalita' testo; se lo splitter parte in modalita' ragionamento aspetta un + che e' gia' passato, e archivia come ragionamento tutta la + risposta -- content vuoto, reasoning_content pieno, stop pulito. Misurato + su glm53 int4, CPU: 10 e 109 caratteri di ragionamento contro + zero di risposta, col prompt corretto sul filo. Vale anche per glm53: la + regola di famiglia sopra dice dove comincia un turno NUOVO, e il turno + proseguito non ne apre nessuno.""" + return (enable_thinking or ARCH == "glm53") and add_generation_prompt class ThinkingStreamSplit: @@ -2304,10 +2699,12 @@ def finish(self): close = finish # interface parity with InklingStreamSplit in the streaming path -def split_thinking_reply(text, enable_thinking=True): +def split_thinking_reply(text, enable_thinking=True, add_generation_prompt=True): """Return the marker-free (thinking, answer) portions of one GLM reply.""" thinking, answer = [], [] - split = ThinkingStreamSplit(thinking.append, answer.append, initial_thinking=starts_in_reasoning(enable_thinking)) + split = ThinkingStreamSplit(thinking.append, answer.append, + initial_thinking=starts_in_reasoning(enable_thinking, + add_generation_prompt)) split.feed(text) split.finish() return "".join(thinking), "".join(answer) @@ -2483,6 +2880,11 @@ def anthropic_tools(body): DEFAULT_CHAT_STOP_SEQUENCES = ("<|user|>", "<|observation|>") +# Seconds to wait for the engine to exit on its own after stdin EOF (its +# atexit teardown writes HEAT_FILE). EOF is only observed between turns, +# so an in-flight generation delays exit; override for impatient scripts. +_ENGINE_DRAIN_S = float(os.environ.get("COLI_ENGINE_DRAIN_S", "30")) + def parse_stop_sequences(body): value = body.get("stop") @@ -2690,8 +3092,8 @@ def generation_options(body, limit): raise APIError(400, "Log probabilities are not supported yet.", "logprobs", "unsupported_parameter") if body.get("frequency_penalty", 0) or body.get("presence_penalty", 0): raise APIError(400, "Token penalties are not supported yet.", None, "unsupported_parameter") - if body.get("seed") is not None: - raise APIError(400, "Per-request seeds are not supported yet.", "seed", "unsupported_parameter") + # `seed` is accepted for request-shape compatibility and silently discarded: + # this server puts no per-request seed on the wire, at any temperature. # response_format -> optional per-request grammar for the engine's grammar-forced # draft source (#70/#148). NEVER a sampling constraint: drafts are verified, so a # schema the engine cannot compile degrades to "no speedup", not to an error and @@ -2797,7 +3199,7 @@ def model_arch(model): return resolve_model(model).descriptor.id -def cap_for_arch(arch, cap, env=None): +def cap_for_arch(arch, cap, env=None, model=None): """Cap-sentinel shim (#379): CURRENT-STATE CALIBRATION, not durable core. An absent cap (None) means different things across today's engines -- @@ -2836,6 +3238,22 @@ def cap_for_arch(arch, cap, env=None): planned = 0 if planned >= 1: return planned + if arch == "deepseek_v41" and model is not None: + # V4.1 only reads its argv cap, not RAM_GB. Without --auto-tier the + # legacy eight slots silently discarded both --ram and RAM_GB (#1666). + from resource_plan import build_plan + settings = env if env is not None else os.environ + ram = settings.get("RAM_GB", "0") + limits = family_by_id(arch).limits + plan = build_plan(model, ram_gb=0 if ram == "auto" else float(ram), + context=int(settings.get(limits.context_env, limits.default_context)), + gpu_indices=[]) + slots = plan["tiers"]["ram"]["cache_slots_per_layer"] + if slots < 1: + raise ValueError("DeepSeek V4.1 RAM budget cannot hold one expert slot per layer") + print(f"[v41] RAM plan: {slots} expert cache slots/layer; --cap overrides", + file=sys.stderr) + return slots return family_by_id(arch).limits.implicit_cap @@ -2954,6 +3372,35 @@ class JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure): return None # never let process bookkeeping break starting the engine +def _write_all(stream, data, frame): + """Write every byte of `data` to `stream`, looping on short writes. + + The production engine stdin is a raw, unbuffered pipe (bufsize=0 -> + io.FileIO), whose write() is a single os.write() and may transfer fewer + bytes than it was given (a signal landing mid-write, a full pipe buffer + on a large IMAGE frame). Discarding the return value would leave the + tail of a frame unsent and desynchronize the engine's stdin framing, so + the remainder is re-offered until it is all consumed. + + Neither `None` nor 0 is progress. `RawIOBase.write` answers `None` when + the stream is non-blocking and could not take a single byte, and 0 says + the same thing with a count; re-offering the buffer after either would + spin forever, so both fail closed as the named engine-write error a + broken pipe raises.""" + written = 0 + total = len(data) + view = memoryview(data) + while written < total: + sent = stream.write(view[written:]) + # None is RawIOBase's "not one byte went out", not an uncounted + # full write, so it fails closed exactly as a zero count does. + if sent is None or sent <= 0: + raise RuntimeError( + f"failed to write {frame} to the engine " + f"(stdin took {written} of {total} bytes)") + written += sent + + class Engine: # cap=None = "not explicitly set": a glm-arch model's engine resolves the # 0 sentinel (8 historically, 1 on Metal+darwin+fast SSD -- colibri.c @@ -2976,12 +3423,18 @@ def __init__(self, executable, model, cap=None, max_tokens=1024, env=None, kv_sl child_env = dict(env or os.environ, SNAP=str(model), SERVE="1", SERVE_BATCH="1", NGEN=str(max_tokens), KV_SLOTS=str(kv_slots)) tune_child_env(child_env, arch) - resolved_cap = cap_for_arch(arch, cap, child_env) + resolved_cap = cap_for_arch(arch, cap, child_env, model=model) child_env.pop("COLI_PROFILE_CAP", None) child_env.pop("COLI_PLAN_CAP", None) + # Own process group on Windows: a CTRL_BREAK sent to the serve + # process group (the graceful stop, handled as SIGBREAK above) must + # not reach the engine — the C runtime's default would kill it + # before its stdin-EOF teardown (atexit -> HEAT_FILE save) can run. + spawn_flags = subprocess.CREATE_NEW_PROCESS_GROUP if sys.platform == "win32" else 0 self.process = subprocess.Popen( [str(executable), str(resolved_cap)], env=child_env, stdin=subprocess.PIPE, stdout=subprocess.PIPE, bufsize=0, + creationflags=spawn_flags, ) # Keep the job handle on the instance: KILL_ON_JOB_CLOSE fires when the # LAST handle closes, so this reference is what ties the engine (and the @@ -3030,6 +3483,32 @@ def _fail_pending(self, error): for events in requests: events.put(("error", error)) + def _write_frame(self, request_id, data, frame): + """Checked server->engine protocol write for CANCEL/STOP: the write + and its flush happen under one write_lock acquisition. Any failure + here -- an OSError from the pipe itself, or _write_all's own + fail-closed RuntimeError on a None/zero-progress write -- drops this + request's pending-map entry: the dispatcher only does that on this + id's own DONE/ERROR frame, and neither arrives when the write that + would have solicited one never reached the engine. An OSError is + additionally re-raised as a named RuntimeError rather than left as + itself: BrokenPipeError is a ConnectionError subclass, so an + unwrapped failure here would fall into do_POST's client-hangup + handler (`except ConnectionError: pass`) and the client would see a + silent connection close instead of the 500 engine_error the failure + actually is. _write_all's own RuntimeError is already the named + error this raises for an OSError, so it is re-raised as-is.""" + try: + with self.write_lock: + _write_all(self.process.stdin, data, frame) + self.process.stdin.flush() + except Exception as error: + with self.pending_lock: + self.pending.pop(request_id, None) + if isinstance(error, OSError): + raise RuntimeError(f"failed to write {frame} to the engine ({error})") from error + raise + def _read_exact(self, size): chunks = [] remaining = size @@ -3250,14 +3729,21 @@ def decode_tool(data): # annunciato subito prima del SUBMIT a cui appartengono. Deve # partire dentro lo stesso lock, o un'altra richiesta potrebbe # infilarsi in mezzo e prendersi l'immagine di questa. - if image is not None: - patches, grid_h, grid_w = image - blob = patches.tobytes() if hasattr(patches, "tobytes") else patches - self.process.stdin.write( - f"IMAGE {request_id} {len(blob)} {grid_h} {grid_w}\n".encode() - + blob + b"\n") - self.process.stdin.write(header + payload + xpayload + b"\n") - self.process.stdin.flush() + try: + if image is not None: + patches, grid_h, grid_w = image + blob = patches.tobytes() if hasattr(patches, "tobytes") else patches + try: + _write_all( + self.process.stdin, + f"IMAGE {request_id} {len(blob)} {grid_h} {grid_w}\n".encode() + + blob + b"\n", "IMAGE") + except OSError as error: + raise RuntimeError(f"failed to write IMAGE to the engine ({error})") from error + _write_all(self.process.stdin, header + payload + xpayload + b"\n", "SUBMIT") + self.process.stdin.flush() + except OSError as error: + raise RuntimeError(f"failed to write SUBMIT to the engine ({error})") from error except Exception: with self.pending_lock: self.pending.pop(request_id, None) @@ -3297,9 +3783,7 @@ def _accept(info): # DONE frame; ClientCancelled is raised when it arrives. if not cancel_sent and not stop_sent and cancelled and cancelled(): cancel_sent = True - with self.write_lock: - self.process.stdin.write(f"CANCEL {request_id}\n".encode()) - self.process.stdin.flush() + self._write_frame(request_id, f"CANCEL {request_id}\n".encode(), "CANCEL") continue if kind == "accept": if accepted: @@ -3311,17 +3795,13 @@ def _accept(info): decode(value) if stopped and stopped(): stop_sent = True - with self.write_lock: - self.process.stdin.write(f"STOP {request_id}\n".encode()) - self.process.stdin.flush() + self._write_frame(request_id, f"STOP {request_id}\n".encode(), "STOP") elif cancelled and cancelled(): # Same admission-holding rule as the idle branch above: # send CANCEL, then keep consuming frames until the # engine acknowledges with ERROR CANCELLED or DONE. cancel_sent = True - with self.write_lock: - self.process.stdin.write(f"CANCEL {request_id}\n".encode()) - self.process.stdin.flush() + self._write_frame(request_id, f"CANCEL {request_id}\n".encode(), "CANCEL") elif kind == "echo": # Lettura del prefill: arriva PRIMA di ogni DATA e non e' testo # generato, quindi non passa da decode() e non entra nella @@ -3335,14 +3815,10 @@ def _accept(info): decode_tool(value) if stopped and stopped(): stop_sent = True - with self.write_lock: - self.process.stdin.write(f"STOP {request_id}\n".encode()) - self.process.stdin.flush() + self._write_frame(request_id, f"STOP {request_id}\n".encode(), "STOP") elif cancelled and cancelled(): cancel_sent = True - with self.write_lock: - self.process.stdin.write(f"CANCEL {request_id}\n".encode()) - self.process.stdin.flush() + self._write_frame(request_id, f"CANCEL {request_id}\n".encode(), "CANCEL") elif kind == "done": _accept({"prompt_tokens": None}) if cancel_sent: @@ -3370,21 +3846,38 @@ def close(self): self.closed = True self._fail_pending(RuntimeError("colibri engine is shutting down")) if self.process.poll() is None: - self.process.terminate() + # Graceful drain first: the engine's serve loop reads requests + # from stdin, and EOF there is the one portable path to its + # atexit teardown (qt_shutdown -> HEAT_FILE save). EOF only + # lands between turns, so the drain wait must be generous. + # poll() (not the absence of TimeoutExpired) decides whether + # the hard-stop ladder below still needs to run: wait() may + # simply return None for a process (or test double) that only + # "terminates" when asked. + try: + self.process.stdin.close() + except (OSError, ValueError, AttributeError): + pass try: - self.process.wait(timeout=5) + self.process.wait(timeout=_ENGINE_DRAIN_S) except subprocess.TimeoutExpired: - # A large resident cache (e.g. 111 GB at --memory-gb 126) can - # take longer than the grace period to unmap and free on - # SIGTERM. SIGKILL cannot be caught, so the process is already - # on its way out; a second timeout only means the reap has not - # landed yet. Teardown is best-effort: never raise from here, or - # a completed measurement is lost to a shutdown that succeeded. - self.process.kill() + pass + if self.process.poll() is None: + self.process.terminate() try: self.process.wait(timeout=5) except subprocess.TimeoutExpired: - pass + # A large resident cache (e.g. 111 GB at --memory-gb 126) can + # take longer than the grace period to unmap and free on + # SIGTERM. SIGKILL cannot be caught, so the process is already + # on its way out; a second timeout only means the reap has not + # landed yet. Teardown is best-effort: never raise from here, or + # a completed measurement is lost to a shutdown that succeeded. + self.process.kill() + try: + self.process.wait(timeout=5) + except subprocess.TimeoutExpired: + pass if self.dispatcher is not threading.current_thread(): self.dispatcher.join(timeout=5) @@ -3447,6 +3940,27 @@ def __init__(self, address, engine, model_id, api_key=None, max_tokens=1024, self._conn_by_ip = {} self._conn_owner = {} + def generate(self, prompt, max_tokens, temperature, top_p, on_text, *args, **kwargs): + started = time.monotonic() + first_output = False + + def measured(callback): + def feed(text): + nonlocal first_output + if text and not first_output: + first_output = True + self.scheduler.observe_timing("first_output_seconds", time.monotonic() - started) + return callback(text) + return feed + + if kwargs.get("on_tool") is not None: + kwargs["on_tool"] = measured(kwargs["on_tool"]) + try: + return self.engine.generate(prompt, max_tokens, temperature, top_p, + measured(on_text), *args, **kwargs) + finally: + self.scheduler.observe_timing("engine_call_seconds", time.monotonic() - started) + def process_request(self, request, client_address): """Refuse past the caps instead of spawning an unbounded thread.""" peer = client_address[0] if client_address else "?" @@ -3791,6 +4305,16 @@ def do_GET(self): try: self._check_host() path = urlsplit(self.path).path + if path == "/metrics": + self.require_auth() + data = self.server.scheduler.prometheus().encode("utf-8") + self.send_response(200) + self.send_header("Content-Type", "text/plain; version=0.0.4; charset=utf-8") + self.send_header("Content-Length", str(len(data))) + self.send_header("Cache-Control", "no-store") + self.end_headers() + self.wfile.write(data) + return if path == "/health": # Liveness is always public; hardware/scheduler internals only when a # request is authed (or no key set), so a configured key isn't leaked @@ -3799,6 +4323,7 @@ def do_GET(self): if self._is_authed(): payload["scheduler"] = self.server.scheduler.snapshot() payload["kv_slots"] = self.server.kv_slots + payload["continue_assistant"] = os.environ.get("COLI_CONTINUE_ASSISTANT", "1") != "0" and ARCH in CONTINUATION_FAMILIES tiers = getattr(self.server.engine, "tiers", None) if self.server.engine else None if tiers: payload["tiers"] = tiers hwinfo = getattr(self.server.engine, "hwinfo", None) if self.server.engine else None @@ -3861,14 +4386,19 @@ def do_POST(self): self._check_host() self.require_auth() body = self.read_json() - self.check_model(body) path = urlsplit(self.path).path + # A client written for Jev sends "jev-latest": on that route the + # served model answers whatever name was asked for. + if path != "/v1/systemone": + self.check_model(body) if path == "/v1/chat/completions": self.chat_completion(body, request_id) elif path == "/v1/completions": self.completion(body, request_id) elif path == "/v1/brio": self.brio(body, request_id) + elif path == "/v1/systemone": + self.systemone(body, request_id) elif path == "/v1/messages": self.anthropic_messages(body, request_id) else: @@ -3914,11 +4444,11 @@ def do_POST(self): # malformato perche' non lo scrive il modello. Prima queste due forme # esistevano solo come script di misura: chi integrava doveva riscriverle. @staticmethod - def _brio_options(options, where): + def _brio_options(options, where, limit=64): if not isinstance(options, list) or not options: raise APIError(400, f"`{where}` must be a non-empty array of strings.", where) - if len(options) > 64: - raise APIError(400, f"`{where}` accepts at most 64 entries.", where) + if len(options) > limit: + raise APIError(400, f"`{where}` accepts at most {limit} entries.", where) seen = set() for option in options: if not isinstance(option, str) or not option.strip(): @@ -3930,7 +4460,11 @@ def _brio_options(options, where): raise APIError(400, f"`{where}` needs at least two options to choose between.", where) return options - def brio(self, body, request_id): + def brio(self, body, request_id, send=True): + # `send=False` returns the result instead of writing it: /v1/systemone + # builds a `questions` request and re-shapes the answer. `_max_options` + # is that caller's word too (Jev allows 255 labels); clamped. + option_limit = min(int(body.get("_max_options", 64) or 64), 255) forms = [k for k in ("options", "questions", "schema") if body.get(k) is not None] if len(forms) != 1: raise APIError(400, "Provide exactly one of `options`, `questions` or `schema`.", @@ -3960,7 +4494,8 @@ def brio(self, body, request_id): if per not in ("mean", "sum"): raise APIError(400, "`normalize` must be \"mean\" or \"sum\".", "normalize") questions.append((text, self._brio_options(entry.get("options"), - f"questions[{i}].options"), per)) + f"questions[{i}].options", + option_limit), per)) else: raw = body["schema"] if not isinstance(raw, dict) or not raw: @@ -4037,7 +4572,7 @@ def score(text, pin): def on_accept(value): accepted.update(value) - self.server.engine.generate( + self.server.generate( text, 0, 0.0, 1.0, lambda _chunk: None, cache_slot, self.client_disconnected, logprobs=1, pin=pin, on_echo=echoes.append, on_accept=on_accept) @@ -4086,7 +4621,18 @@ def choose(prefix, choices, norm): # le domande (o tutte le caselle) condividono. Con un livello solo # la domanda si rilegge una volta per opzione; con due, 176 token # invece di 496 su quattro item (misurato). - if state_prefix and form != "options": + # + # Vale anche per la forma `options`: dentro una singola richiesta lo + # stato si legge comunque una volta (lo snapshot dello stato viene + # ripristinato quando `choose` fotografa il prefisso completo), ma + # il punto di ritorno sullo stato condiviso serve TRA richieste. La + # pagina web manda una domanda per richiesta sullo stesso documento; + # senza questa fotografia ogni domanda rifarebbe il prefill di tutto + # il documento, buttando via il "read once" che e' il senso della + # modalita. Con essa, ogni domanda successiva paga solo i propri + # token. Il costo e' uno snapshot in piu' su una richiesta one-shot, + # riusato o sfrattato. + if state_prefix: n_state, _ = score(state_prefix, True) prompt_max = max(prompt_max, n_state) @@ -4136,9 +4682,143 @@ def choose(prefix, choices, norm): "read_tokens": read_total, "total_tokens": prompt_max + read_total}, }) - self.send_json(200, result, request_id, - {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000)), - "x-colibri-elapsed-ms": str(round((time.time() - started) * 1000))}) + headers = {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000)), + "x-colibri-elapsed-ms": str(round((time.time() - started) * 1000))} + if not send: + result["_headers"] = headers + return result + self.send_json(200, result, request_id, headers) + + # ------------------------------------------------------------ Jev-compatible + # + # POST /v1/systemone speaks the request and the reply of TypeSafe's Jev + # API (docs.typesafe.ai/api): a client written for it points at colibri + # and changes the base URL, nothing else. The three primitives map onto + # the `questions` form of /v1/brio, the same channel: the state is + # photographed once and every question pays only its own tokens. + # + # noul -> one yes/no question. `noul` is the probability of yes. The + # optional criteria (what true and false mean) go into the + # question text. + # choice -> the labels of `criteria` are the options; their descriptions + # go into the question text, because a label alone ("billing") + # does not say what it means. `confidence` follows their + # documented formula, (n * peak - 1) / (n - 1). + # score -> the levels of `criteria` are the options "1".."n"; `score` + # is the expected value under the distribution, `legend` the + # levels by number, `confidence` as for choice. + # + # What differs, stated rather than hidden: `model` echoes the served + # model, not "jev-latest"; `usage.output_tokens` counts the option tokens + # READ, since this engine generates nothing; validation errors are 422 as + # theirs are, with this server's error envelope. docs/brio.md has the + # mapping table. + _SYSTEMONE_MAX_QUESTIONS = 64 + + @staticmethod + def _systemone_text(value, where): + """Jev's EntryType: a string, or JSON given as an object or an array.""" + if value is None: + return None + if isinstance(value, str): + return value.strip() or None + if isinstance(value, (dict, list)): + return json.dumps(value, ensure_ascii=False, indent=2) + raise APIError(422, f"`{where}` must be a string, an object or an array.", where) + + @staticmethod + def _systemone_confidence(probabilities): + """(n * peak - 1) / (n - 1): 1 when all the mass is on one label, 0 when flat.""" + values = list(probabilities) + n = len(values) + if n < 2: + return 1.0 + return round(max(0.0, (n * max(values) - 1.0) / (n - 1)), 6) + + def systemone(self, body, request_id): + state = self._systemone_text(body.get("state"), "state") + if state is None: + raise APIError(422, "`state` is required: the content the questions are about.", "state") + raw = body.get("questions") + if not isinstance(raw, dict) or not raw: + raise APIError(422, "`questions` must be a non-empty object of id: question.", "questions") + if len(raw) > self._SYSTEMONE_MAX_QUESTIONS: + raise APIError(422, f"`questions` accepts at most {self._SYSTEMONE_MAX_QUESTIONS} entries.", + "questions") + plan = [] # (id, kind, text, options, levels) + for qid, question in raw.items(): + where = f"questions.{qid}" + if not isinstance(qid, str) or not qid.strip(): + raise APIError(422, "Every question id must be a non-empty string.", "questions") + if not isinstance(question, dict): + raise APIError(422, f"`{where}` must be an object.", where) + kind = question.get("type") + instructions = self._systemone_text(question.get("instructions"), f"{where}.instructions") + criteria = question.get("criteria") + if kind == "noul": + if criteria is not None and not isinstance(criteria, dict): + raise APIError(422, f"`{where}.criteria` must be an object with `true` and/or `false`.", + f"{where}.criteria") + yes = self._systemone_text((criteria or {}).get("true"), f"{where}.criteria.true") + no = self._systemone_text((criteria or {}).get("false"), f"{where}.criteria.false") + text = instructions or "Is this true?" + if yes: + text += f"\nyes: {yes}" + if no: + text += f"\nno: {no}" + plan.append((qid, "noul", text + "\nAnswer yes or no.", ["yes", "no"], None)) + elif kind == "choice": + if not isinstance(criteria, dict) or not criteria: + raise APIError(422, f"`{where}.criteria` must be a non-empty object of label: description.", + f"{where}.criteria") + if len(criteria) > 255: + raise APIError(422, f"`{where}.criteria` accepts at most 255 labels.", f"{where}.criteria") + labels, lines = [], [] + for label, description in criteria.items(): + if not isinstance(label, str) or not label.strip(): + raise APIError(422, f"Every label of `{where}.criteria` must be a non-empty string.", + f"{where}.criteria") + labels.append(label) + text = self._systemone_text(description, f"{where}.criteria.{label}") + lines.append(f"- {label}: {text}" if text else f"- {label}") + if len(labels) < 2: + raise APIError(422, f"`{where}.criteria` needs at least two labels.", f"{where}.criteria") + text = (instructions or "Which of the following applies?") + "\nOptions:\n" + "\n".join(lines) + plan.append((qid, "choice", text + "\nAnswer with one of the options.", labels, None)) + elif kind == "score": + if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10: + raise APIError(422, f"`{where}.criteria` must be an array of 2 to 10 level descriptions.", + f"{where}.criteria") + levels = [self._systemone_text(c, f"{where}.criteria[{i}]") or f"level {i + 1}" + for i, c in enumerate(criteria)] + text = (instructions or "Rate this on the scale below.") + "\nScale:\n" + \ + "\n".join(f"{i + 1}: {d}" for i, d in enumerate(levels)) + plan.append((qid, "score", text + "\nAnswer with the number.", + [str(i + 1) for i in range(len(levels))], levels)) + else: + raise APIError(422, f"`{where}.type` must be \"noul\", \"choice\" or \"score\".", f"{where}.type") + inner = {"state": state, "_max_options": 255, + "questions": [{"question": text, "options": options} for _, _, text, options, _ in plan]} + result = self.brio(inner, request_id, send=False) + answers = {} + for (qid, kind, _, options, levels), got in zip(plan, result["answers"]): + p = {c["option"]: c["p"] for c in got["choices"]} + if kind == "noul": + answers[qid] = {"type": "noul", "noul": round(p.get("yes", 0.0), 6)} + elif kind == "choice": + answers[qid] = {"type": "choice", "choice": got["answer"], + "probabilities": {o: round(p[o], 6) for o in options}, + "confidence": self._systemone_confidence(p.values())} + else: + answers[qid] = {"type": "score", + "score": round(sum(int(k) * v for k, v in p.items()), 6), + "legend": {str(i + 1): d for i, d in enumerate(levels)}, + "probabilities": {o: round(p[o], 6) for o in options}, + "confidence": self._systemone_confidence(p.values())} + reply = {"model": self.server.model_id, "answers": answers, + "usage": {"input_tokens": result["usage"]["prompt_tokens"], + "output_tokens": result["usage"]["read_tokens"]}} + self.send_json(200, reply, request_id, result.get("_headers")) def _fail(self, error, request_id): """Report an error, unless the response is already on the wire. Once a streaming 200 @@ -4157,7 +4837,8 @@ def error_body(self, error): return {"type": "error", "error": {"type": error.error_type, "message": error.message}} def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=None, - enable_thinking=False, audio=None, image=None): + enable_thinking=False, audio=None, image=None, + add_generation_prompt=True): # COLI_DEBUG tees the engine transaction to stderr: 1 = decoded output stream only, # 2 = both sides (rendered prompt + output). render_chat already folds prior turns and # tool results into `prompt`, so level 2 is the full conversation the engine saw. @@ -4205,7 +4886,8 @@ def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=Non completion_id = id_prefix + uuid.uuid4().hex created = int(time.time()) - with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission: + with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission, \ + contextlib.ExitStack() as stream_cleanup: queue_wait, cache_slot = admission queue_headers = {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000))} if not stream: @@ -4217,7 +4899,7 @@ def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=Non def generation_stopped(): return stop_filter.stopped() or sideband.stopped() - stats = self.server.engine.generate( + stats = self.server.generate( prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot, self.client_disconnected, grammar=grammar, stopped=generation_stopped, **({"on_tool": sideband.feed} if sideband.enabled else {}), @@ -4233,7 +4915,8 @@ def generation_stopped(): # #597 item 4: GLM emits reasoning then then the answer. Route the # reasoning to reasoning_content instead of dumping it (or the raw ) # into the visible answer / tool-call parser. - reasoning, text = split_thinking_reply(text, enable_thinking) + reasoning, text = split_thinking_reply(text, enable_thinking, + add_generation_prompt) length_finish = "length" if stats["length_limited"] else "stop" if chat and tools: content, calls = parse_arch_tool_calls(text, tools, sideband.reply()) @@ -4356,6 +5039,8 @@ def start_stream(_accept_info=None): "logprobs": None, "finish_reason": None}]) ka_thread[0] = threading.Thread(target=_keepalive, daemon=True) ka_thread[0].start() + stream_cleanup.callback(ka_thread[0].join, timeout=2) + stream_cleanup.callback(ka_stop.set) if chat and tools: # Suppress tool-call markers from the streamed content and parse the authoritative # calls from the FULL reply after generation. Hold back a marker-length tail so a @@ -4388,7 +5073,8 @@ def feed_content(chunk): # answer text only (post-) # #597: keep GLM reasoning out of the tool-call buffer — a think splitter sends it # to reasoning_content and passes only the answer text on to feed_content/parser. think = (ThinkingStreamSplit(emit_reasoning, feed_content, - initial_thinking=starts_in_reasoning(enable_thinking)) + initial_thinking=starts_in_reasoning( + enable_thinking, add_generation_prompt)) if glm_think else None) def emit_tools(chunk): if dbg_echo: @@ -4399,7 +5085,7 @@ def emit_tools(chunk): def generation_stopped(): return stop_filter.stopped() or sideband.stopped() - stats = self.server.engine.generate( + stats = self.server.generate( prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot, self.client_disconnected, grammar=grammar, stopped=generation_stopped, **({"on_tool": sideband.feed} if sideband.enabled else {}), @@ -4423,8 +5109,10 @@ def generation_stopped(): if splitter is not None: # inkling content/marker splitter content_split = splitter elif glm_think: # GLM reasoning → reasoning_content - content_split = ThinkingStreamSplit(emit_reasoning, emit, - initial_thinking=starts_in_reasoning(enable_thinking)) + content_split = ThinkingStreamSplit( + emit_reasoning, emit, + initial_thinking=starts_in_reasoning(enable_thinking, + add_generation_prompt)) else: content_split = None def emit_plain(chunk): @@ -4432,7 +5120,7 @@ def emit_plain(chunk): sys.stderr.write(chunk); sys.stderr.flush() (content_split.feed if content_split else emit)(chunk) stop_filter = StopFilter(stop_sequences, emit_plain, ignore_leading_stop) - stats = self.server.engine.generate( + stats = self.server.generate( prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot, self.client_disconnected, grammar=grammar, stopped=stop_filter.stopped, on_accept=start_stream, **({"audio": audio} if audio else {}), @@ -4539,10 +5227,13 @@ def chat_completion(self, body, request_id): raise APIError(400, "one image per request for now; the engine " "holds a single pending image.", "messages") image = images[0] if images else None + add_generation_prompt = resolve_generation_prompt(messages, body) prompt = render_chat_for_arch(messages, enable_thinking, reasoning_effort, - tools, tool_choice, audio_out=audio_clips) + tools, tool_choice, audio_out=audio_clips, + add_generation_prompt=add_generation_prompt) self.generation(body, prompt, request_id, True, tools, tool_choice, enable_thinking=enable_thinking, + add_generation_prompt=add_generation_prompt, audio=b"".join(audio_clips) if audio_clips else None, image=image) @@ -4581,12 +5272,16 @@ def anthropic_messages(self, body, request_id): if tool_choice == "none": tools = None default_effort = "xhigh" if ARCH == "qwen38" and thinking is None else "high" + add_generation_prompt = resolve_generation_prompt(messages, body) prompt = render_chat_for_arch(messages, enable_thinking, default_effort if enable_thinking else None, - tools, tool_choice) - self.anthropic_generation(translated, prompt, request_id, tools, enable_thinking) + tools, tool_choice, + add_generation_prompt=add_generation_prompt) + self.anthropic_generation(translated, prompt, request_id, tools, enable_thinking, + add_generation_prompt) - def anthropic_generation(self, body, prompt, request_id, tools, enable_thinking): + def anthropic_generation(self, body, prompt, request_id, tools, enable_thinking, + add_generation_prompt=True): maximum, temperature, top_p, grammar, _stop_sequences = generation_options( body, self.server.max_tokens) # Same policy as /v1/chat/completions: `body` is the translated OpenAI-shaped @@ -4618,7 +5313,8 @@ def blocks_and_stop(text, stats, tool_reply=None): if ARCH == "inkling": text, reasoning = split_inkling(text) elif enable_thinking: - reasoning, text = split_thinking_reply(text) + reasoning, text = split_thinking_reply(text, enable_thinking, + add_generation_prompt) if enable_thinking: content.append({"type": "thinking", "thinking": reasoning, "signature": ANTHROPIC_LOCAL_SIGNATURE}) @@ -4638,7 +5334,8 @@ def blocks_and_stop(text, stats, tool_reply=None): reason = "tool_calls" if calls else ("length" if stats["length_limited"] else "stop") return content, self.ANTHROPIC_STOP[reason] - with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission: + with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission, \ + contextlib.ExitStack() as stream_cleanup: queue_wait, cache_slot = admission queue_headers = {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000))} if not stream: @@ -4650,7 +5347,7 @@ def blocks_and_stop(text, stats, tool_reply=None): def generation_stopped(): return stop_filter.stopped() or sideband.stopped() - stats = self.server.engine.generate( + stats = self.server.generate( prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot, self.client_disconnected, grammar=grammar, stopped=generation_stopped, **({"on_tool": sideband.feed} if sideband.enabled else {})) @@ -4719,6 +5416,8 @@ def keepalive(): "content_block": {"type": "text", "text": ""}}) ka_thread = threading.Thread(target=keepalive, daemon=True) ka_thread.start() + stream_cleanup.callback(ka_thread.join, timeout=2) + stream_cleanup.callback(ka_stop.set) raw = [] sideband = ToolSideband(ARCH == "kimi" and bool(tools), stop_sequences, @@ -4780,7 +5479,8 @@ def close_thinking(): # lo splitter serve pure col ragionamento "spento", o il # pensiero finisce incollato davanti alla risposta. split = (ThinkingStreamSplit(emit_thinking, emit_answer, close_thinking) - if starts_in_reasoning(enable_thinking) else None) + if starts_in_reasoning(enable_thinking, add_generation_prompt) + else None) def on_text(chunk): raw.append(chunk) @@ -4791,7 +5491,7 @@ def on_text(chunk): def generation_stopped(): return stop_filter.stopped() or sideband.stopped() - stats = self.server.engine.generate( + stats = self.server.generate( prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot, lambda: not connected[0], grammar=grammar, stopped=generation_stopped, **({"on_tool": sideband.feed} if sideband.enabled else {})) @@ -4887,6 +5587,16 @@ def serve(model, host="127.0.0.1", port=8000, model_id=None, api_key=None, server.engine = runtime print(f"OpenAI-compatible API listening on http://{host}:{port}/v1", file=sys.stderr) signal.signal(signal.SIGTERM, lambda *_: threading.Thread(target=server.shutdown, daemon=True).start()) + # On Windows SIGTERM is never delivered (os.kill is TerminateProcess); + # CTRL_BREAK — the one console signal a controller CAN target at this + # process group — arrives as SIGBREAK. Without this handler it kills + # the serve loop outright, skipping the finally that drains the + # engine (stdin EOF -> atexit -> HEAT_FILE save). The engine child + # runs in its own process group (see Engine.__init__) and does not + # receive this event. + if hasattr(signal, "SIGBREAK"): + signal.signal(signal.SIGBREAK, + lambda *_: threading.Thread(target=server.shutdown, daemon=True).start()) try: server.serve_forever() except KeyboardInterrupt: diff --git a/c/oracle.h b/c/oracle.h new file mode 100644 index 000000000..7790cb92a --- /dev/null +++ b/c/oracle.h @@ -0,0 +1,97 @@ +/* Validation for the GLM reference file; no model or Python dependency. */ +#ifndef COLI_ORACLE_H +#define COLI_ORACLE_H +#include +#include +#include "json.h" + +typedef struct { + int *prompt, *full, *tf; + int np, nfull; +} OracleRef; + +static void oracle_ref_free(OracleRef *r) { + free(r->prompt); free(r->full); free(r->tf); + memset(r, 0, sizeof(*r)); +} + +static int *oracle_read_ids(jval *root, const char *key, int vocab, int *n) { + jval *a=json_get(root,key); + *n=0; + if (!a || a->t!=J_ARR || a->len<1) { + fprintf(stderr,"[ORACLE] %s must be a nonempty token array\n",key); + return NULL; + } + int *ids=malloc((size_t)a->len*sizeof(*ids)); + if (!ids) { fprintf(stderr,"[ORACLE] out of memory reading %s\n",key); return NULL; } + for (int i=0; ilen; i++) { + jval *v=a->kids[i]; + if (v->t!=J_NUM || !isfinite(v->num) || v->num<0 || v->num>=vocab || + v->num!=floor(v->num)) { + fprintf(stderr,"[ORACLE] %s[%d] must be an integer token in [0,%d)\n",key,i,vocab); + free(ids); return NULL; + } + ids[i]=(int)v->num; + } + *n=a->len; + return ids; +} + +static int oracle_ref_parse(const char *text, int vocab, int teacher_forcing, OracleRef *r) { + memset(r,0,sizeof(*r)); + jval *root=json_parse_checked(text); + if (!root || root->t!=J_OBJ || vocab<1) { + fprintf(stderr,"[ORACLE] invalid reference JSON or vocabulary\n"); + json_free(root); return 0; + } + /* Duplicate keys could otherwise silently select a stale prediction array. */ + const char *keys[]={"prompt_ids","full_ids","tf_pred"}; + for (int k=0; k<3; k++) { + int count=0; + for (int i=0; ilen; i++) if (!strcmp(root->keys[i],keys[k])) count++; + if (count>1) { + fprintf(stderr,"[ORACLE] duplicate reference field %s\n",keys[k]); + goto fail; + } + } + r->prompt=oracle_read_ids(root,"prompt_ids",vocab,&r->np); + r->full=oracle_read_ids(root,"full_ids",vocab,&r->nfull); + if (!r->prompt || !r->full) goto fail; + if (r->nfullnp || (!teacher_forcing && r->nfull==r->np) || + memcmp(r->prompt,r->full,(size_t)r->np*sizeof(int))) { + fprintf(stderr,"[ORACLE] full_ids must start with prompt_ids and include the compared tokens\n"); + goto fail; + } + if (teacher_forcing) { + int ntf=0; + r->tf=oracle_read_ids(root,"tf_pred",vocab,&ntf); + if (!r->tf) goto fail; + if (ntf!=r->nfull) { + fprintf(stderr,"[ORACLE] tf_pred length %d != full_ids length %d\n",ntf,r->nfull); + goto fail; + } + } + json_free(root); return 1; +fail: + json_free(root); oracle_ref_free(r); return 0; +} + +static int oracle_logits_finite(const float *lo, int vocab) { + for (int i=0; i'9') return 0; + char *end; + errno=0; + long value=strtol(text,&end,10); + if (errno==ERANGE || *end || value<0 || value>=total) return 0; + *allowed=(int)value; + return 1; +} +#endif diff --git a/c/qgemv.h b/c/qgemv.h new file mode 100644 index 000000000..62f53d7f1 --- /dev/null +++ b/c/qgemv.h @@ -0,0 +1,147 @@ +#ifndef COLIBRI_QGEMV_H +#define COLIBRI_QGEMV_H +/* Plain (non-group-scaled) int8 GEMV: one f32 scale per output row, q[O,I] + * int8 row-major. Its own header so tests/test_qgemv.c can link the very + * kernel the engine runs; qwen36.c carries a main and cannot be linked into + * a test. + * + * This is qwen36's dense-projection kernel (lm_head and any non-expert + * quantized matmul); the group-scaled sibling used for expert matmuls lives + * in gsgemv.h. Its float operations must stay in exactly this order -- float + * addition is not associative and the engine's token stream is required to + * be byte-identical to the reference. tests/test_qgemv.c holds the + * pre-restructure kernel verbatim and compares raw float bits, so any + * reassociation fails there rather than surfacing as drifted text much + * later. + * + * The SSE4.1 tier below is the exception to that byte-identical rule, by + * design: it is new code with no pre-existing output to match, and it is + * checked by tests/test_qgemv.c against a tolerance, not memcmp. It routes + * its FMA and float loads through sse41_kernels.h -- the same shared + * primitives header olmoe.c uses -- rather than inlining its own copy. */ +#include +#include +#include +#if defined(__ARM_NEON) +#include +#endif +#if (defined(__AVX2__) && defined(__FMA__)) || defined(__SSE4_1__) +#include +#endif +#if defined(__SSE4_1__) +#include "sse41_kernels.h" +#endif + +#if defined(__ARM_NEON) +static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) { + int32x4_t acc = vdupq_n_s32(0); + int8x16_t va = vld1q_s8(a), vb = vld1q_s8(b); +#if defined(__ARM_FEATURE_DOTPROD) + acc = vdotq_s32(acc, va, vb); +#else + acc = vpadalq_s16(acc, vmull_s8(vget_low_s8(va), vget_low_s8(vb))); + acc = vpadalq_s16(acc, vmull_s8(vget_high_s8(va), vget_high_s8(vb))); +#endif + return vaddvq_s32(acc); +} +#endif + +static void matmul_q(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) { +#if defined(__ARM_NEON) + /* IDOT is opt-in, not default-on: this path quantizes the ACTIVATIONS to + * Q8_0 per 16-element block, which the scalar path does not, so the two are + * not numerically equivalent. olmoe shipped it default-on and it cost + * token-exactness end to end (#1044, fixed in af48fe8 by making it opt-in); + * qwen36 inherited the same default from the same family of kernels. The + * tiny-oracle gate would not have caught it -- that job runs on x86. */ + static int idot = -1; + if (idot < 0) { const char *e = getenv("IDOT"); idot = (e && atoi(e)); } + if (idot && I % 16 == 0 && I <= 4096) { + int nb = I / 16; int8_t xi[4096]; float xs[256]; + for (int b = 0; b < nb; b++) { + const float *xb = x + b*16; + float am = 0.f; for (int i = 0; i < 16; i++) { float a = fabsf(xb[i]); if (a > am) am = a; } + float s = am/127.f; if (s < 1e-12f) s = 1e-12f; + xs[b] = s; float inv = 1.f/s; + for (int i = 0; i < 16; i++) xi[b*16+i] = (int8_t)lrintf(xb[i]*inv); + } + #pragma omp parallel for schedule(static) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + float acc = 0.f; + for (int b = 0; b < nb; b++) acc += xs[b]*(float)dot_i8_16(xi+b*16, w+b*16); + y[o] = acc * scale[o]; + } + return; + } +#endif +#if defined(__AVX2__) && defined(__FMA__) + /* Hand-vectorized int8->f32 GEMV (gcc does not auto-vectorize the + * convert+accumulate chain). 32 weights per iteration, FMA accumulate. */ + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps(); + __m256 a2 = _mm256_setzero_ps(), a3 = _mm256_setzero_ps(); + int i = 0; + for (; i + 32 <= I; i += 32) { + __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); + __m128i b1 = _mm_loadu_si128((const __m128i*)(w + i + 16)); + a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0); + a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1); + a2 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+16), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a2); + a3 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+24), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a3); + } + a0 = _mm256_add_ps(_mm256_add_ps(a0,a1), _mm256_add_ps(a2,a3)); + __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1)); + s = _mm_add_ps(s, _mm_movehl_ps(s,s)); + s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1)); + float acc = _mm_cvtss_f32(s); + for (; i < I; i++) acc += x[i] * (float)w[i]; + y[o] = acc * scale[o]; + } +#elif defined(__SSE4_1__) + /* SSE4.1 tier: one rung below the AVX2 branch above, same shape -- + * single output row per iteration, no row-interleaving and no runtime + * gate (matmul_q has no group scale to gate on, unlike matmul_q_gs's + * SSE4.1 branch in gsgemv.h). 128-bit lanes instead of 256-bit: four + * accumulators of 4 floats each, 16 weights per iteration where AVX2 + * manages 32. The multiply-add and the float loads route through + * sse41_kernels.h (COLIBRI_FMA / colibri_sse41_loadu_ps), the same + * shared primitives olmoe.c's canary uses, instead of an inline copy of + * the intrinsics. This tier is new code with no pre-existing + * byte-identical output to match, so tests/test_qgemv.c checks it + * against a tolerance, not memcmp -- unlike the AVX2 and scalar tiers + * above/below. */ + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + __m128 a0 = _mm_setzero_ps(), a1 = _mm_setzero_ps(); + __m128 a2 = _mm_setzero_ps(), a3 = _mm_setzero_ps(); + int i = 0; + for (; i + 16 <= I; i += 16) { + __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); + a0 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b0)), a0); + a1 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+4), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,4))), a1); + a2 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+8), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,8))), a2); + a3 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+12), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,12))), a3); + } + a0 = _mm_add_ps(_mm_add_ps(a0,a1), _mm_add_ps(a2,a3)); + a0 = _mm_add_ps(a0, _mm_movehl_ps(a0,a0)); + a0 = _mm_add_ss(a0, _mm_shuffle_ps(a0,a0,1)); + float acc = _mm_cvtss_f32(a0); + for (; i < I; i++) acc += x[i] * (float)w[i]; + y[o] = acc * scale[o]; + } +#else + #pragma omp parallel for schedule(static) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + float acc = 0.f; + for (int i = 0; i < I; i++) acc += x[i] * (float)w[i]; + y[o] = acc * scale[o]; + } +#endif +} + +#endif /* COLIBRI_QGEMV_H */ diff --git a/c/quant.h b/c/quant.h index f53ad4360..a62c36ffb 100644 --- a/c/quant.h +++ b/c/quant.h @@ -14,33 +14,10 @@ #include #endif -/* ---- SIMD includes -------------------------------------------------------- */ -#ifdef __AVX2__ -#include -static inline float hsum256(__m256 v){ - __m128 lo=_mm256_castps256_ps128(v), hi=_mm256_extractf128_ps(v,1); - lo=_mm_add_ps(lo,hi); __m128 sh=_mm_movehl_ps(lo,lo); lo=_mm_add_ps(lo,sh); - sh=_mm_shuffle_ps(lo,lo,1); lo=_mm_add_ss(lo,sh); return _mm_cvtss_f32(lo); -} -static inline int hsum256_i32(__m256i v){ - __m128i lo=_mm256_castsi256_si128(v), hi=_mm256_extracti128_si256(v,1); - lo=_mm_add_epi32(lo,hi); lo=_mm_hadd_epi32(lo,lo); lo=_mm_hadd_epi32(lo,lo); - return _mm_cvtsi128_si32(lo); -} -#endif -#if defined(__AVXVNNI__) && defined(__AVX2__) -static inline int hsum128_i32(__m128i v){ - v=_mm_hadd_epi32(v,v); v=_mm_hadd_epi32(v,v); return _mm_cvtsi128_si32(v); -} -#endif -#ifdef __ARM_NEON -#include -#endif -#ifdef __VSX__ -#include -#undef vector -#undef pixel -#undef bool +#include "idot.h" /* SIMD prelude + the integer dot kernels, shared with qwen36 */ + +#if defined(__SSE4_1__) +#include "sse41_kernels.h" #endif /* ---- AVX-512 int4->float accumulator -------------------------------------- */ @@ -168,8 +145,17 @@ static void matmul_i4(float *y, const float *x, const uint8_t *q4, const float * static void matmul_i4_grouped(float *y, const float *x, const uint8_t *q4, const float *scale, int S, int I, int O, int gs){ int rb=(I+1)/2; int ng=(I+gs-1)/gs; + int o0=0; +#if defined(__SSE4_1__) && !defined(__AVX2__) + /* Even group sizes keep every group start on a low-nibble boundary. */ + if(!(gs&1)){ + o0=O&~3; + if(o0) matmul_i4_grouped_sse41_rows4(y,x,q4,scale,S,I,O,gs,rb,ng,o0); + if(o0==O) return; + } +#endif #pragma omp parallel for schedule(static) - for(int o=0;oI) blen=I-base; + float acc0=0,acc1=0,acc2=0,acc3=0; + for(int i=base;iamax)amax=a; } - float s=amax/127.f; if(s<1e-12f) s=1e-12f; float inv=1.f/s; - for(int i=0;i bit-identico. Stessa struttura - * dei 4 accumulatori del ramo NEON piu' sotto. - * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer - * adds are associative, so the result is bit-identical (mirrors the NEON path). */ - __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128(); - for(;i+64<=I;i+=64){ - __m128i w0=_mm_loadu_si128((const __m128i*)(w+i)), x0=_mm_loadu_si128((const __m128i*)(x+i)); - __m128i w1=_mm_loadu_si128((const __m128i*)(w+i+16)), x1=_mm_loadu_si128((const __m128i*)(x+i+16)); - __m128i w2=_mm_loadu_si128((const __m128i*)(w+i+32)), x2=_mm_loadu_si128((const __m128i*)(x+i+32)); - __m128i w3=_mm_loadu_si128((const __m128i*)(w+i+48)), x3=_mm_loadu_si128((const __m128i*)(x+i+48)); - a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0)); - a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1)); - a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2)); - a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3)); - } - __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3)); - for(;i+16<=I;i+=16){ - __m128i wv=_mm_loadu_si128((const __m128i*)(w+i)); - __m128i xv=_mm_loadu_si128((const __m128i*)(x+i)); - acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),_mm_sign_epi8(xv,wv)); - } - sum=hsum128_i32(acc); -#elif defined(__AVX2__) - __m256i acc=_mm256_setzero_si256(); const __m256i ones=_mm256_set1_epi16(1); - for(;i+32<=I;i+=32){ - __m256i wv=_mm256_loadu_si256((const __m256i*)(w+i)); - __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i)); - __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv)); - acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones)); - } - sum=hsum256_i32(acc); -#elif defined(__ARM_NEON) -#if defined(__ARM_FEATURE_DOTPROD) - int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); - for(;i+64<=I;i+=64){ - a0=vdotq_s32(a0,vld1q_s8(w+i), vld1q_s8(x+i)); - a1=vdotq_s32(a1,vld1q_s8(w+i+16),vld1q_s8(x+i+16)); - a2=vdotq_s32(a2,vld1q_s8(w+i+32),vld1q_s8(x+i+32)); - a3=vdotq_s32(a3,vld1q_s8(w+i+48),vld1q_s8(x+i+48)); - } - int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); - for(;i+16<=I;i+=16) acc=vdotq_s32(acc,vld1q_s8(w+i),vld1q_s8(x+i)); - sum=vaddvq_s32(acc); -#else - int32x4_t acc=vdupq_n_s32(0); - for(;i+16<=I;i+=16){ - int8x16_t wv=vld1q_s8(w+i), xv=vld1q_s8(x+i); - int16x8_t p=vmull_s8(vget_low_s8(wv),vget_low_s8(xv)); - p=vmlal_s8(p,vget_high_s8(wv),vget_high_s8(xv)); - acc=vpadalq_s16(acc,p); - } - sum=vaddvq_s32(acc); -#endif -#elif defined(__VSX__) - __vector signed int acc=vec_splats(0); - const __vector signed char vz=vec_splats((signed char)0); - for(;i+16<=I;i+=16){ - __vector signed char wv=vec_xl(0,(const signed char*)(w+i)); - __vector signed char xv=vec_xl(0,(const signed char*)(x+i)); - __vector __bool char neg=vec_cmplt(wv,vz); - __vector signed char xs=vec_sel(xv,vec_sub(vz,xv),neg); - __vector unsigned char wa=(__vector unsigned char)vec_sel(wv,vec_sub(vz,wv),neg); - acc=vec_msum(xs,wa,acc); - } - sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3); -#endif - for(;i>1))); - __m256i lo=_mm256_and_si256(by,m4v), hi=_mm256_and_si256(_mm256_srli_epi16(by,4),m4v); - __m256i z0=_mm256_unpacklo_epi8(lo,hi), z1=_mm256_unpackhi_epi8(lo,hi); - __m512i wv=_mm512_sub_epi8(_mm512_inserti64x4(_mm512_castsi256_si512(z0),z1,1),b8v); - __m512i xv=_mm512_permutexvar_epi64(xidx,_mm512_loadu_si512((const void*)(x+i))); - __mmask64 neg=_mm512_movepi8_mask(wv); - __m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv); - acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs); - } - sum=_mm512_reduce_add_epi32(acc); -#elif defined(__AVXVNNI__) && defined(__AVX2__) - /* 4 accumulatori indipendenti (64 elementi = 32 byte packed/iter): un solo acc - * incatena i vpdpbusd (latenza-bound ~5c). Somme intere associative -> bit-identico. - * Stessa struttura dei 4 accumulatori del ramo NEON piu' sotto. - * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer - * adds are associative, so the result is bit-identical (mirrors the NEON path). */ - const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8); - __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128(); - for(;i+64<=I;i+=64){ - __m128i by0=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); /* elem i..i+31 */ - __m128i by1=_mm_loadu_si128((const __m128i*)(w4+(i>>1)+16)); /* elem i+32..i+63 */ - __m128i lo0=_mm_and_si128(by0,m4), hi0=_mm_and_si128(_mm_srli_epi16(by0,4),m4); - __m128i lo1=_mm_and_si128(by1,m4), hi1=_mm_and_si128(_mm_srli_epi16(by1,4),m4); - __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo0,hi0),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo0,hi0),b8); - __m128i w2=_mm_sub_epi8(_mm_unpacklo_epi8(lo1,hi1),b8), w3=_mm_sub_epi8(_mm_unpackhi_epi8(lo1,hi1),b8); - __m128i x0=_mm_loadu_si128((const __m128i*)(x+i)), x1=_mm_loadu_si128((const __m128i*)(x+i+16)); - __m128i x2=_mm_loadu_si128((const __m128i*)(x+i+32)), x3=_mm_loadu_si128((const __m128i*)(x+i+48)); - a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0)); - a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1)); - a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2)); - a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3)); - } - __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3)); - for(;i+32<=I;i+=32){ /* 32-nibble remainder: 2 dpbusd, same unpack */ - __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); - __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4); - __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo,hi),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo,hi),b8); - __m128i x0=_mm_loadu_si128((const __m128i*)(x+i)); - __m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16)); - acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0)); - acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1)); - } - sum=hsum128_i32(acc); -#elif defined(__AVX2__) - const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi8(8); - const __m256i ones=_mm256_set1_epi16(1); - __m256i acc=_mm256_setzero_si256(); - for(;i+32<=I;i+=32){ - __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); - __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4); - __m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi); - __m256i wv=_mm256_sub_epi8(_mm256_set_m128i(n1,n0),b8); - __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i)); - __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv)); - acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones)); - } - sum=hsum256_i32(acc); -#elif defined(__ARM_NEON) - const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8); -#if defined(__ARM_FEATURE_DOTPROD) - int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); - for(;i+64<=I;i+=64){ - uint8x16_t byA=vld1q_u8(w4+(i>>1)), byB=vld1q_u8(w4+(i>>1)+16); - uint8x16x2_t zA=vzipq_u8(vandq_u8(byA,m4q), vshrq_n_u8(byA,4)); - uint8x16x2_t zB=vzipq_u8(vandq_u8(byB,m4q), vshrq_n_u8(byB,4)); - a0=vdotq_s32(a0,vsubq_s8(vreinterpretq_s8_u8(zA.val[0]),b8q),vld1q_s8(x+i)); - a1=vdotq_s32(a1,vsubq_s8(vreinterpretq_s8_u8(zA.val[1]),b8q),vld1q_s8(x+i+16)); - a2=vdotq_s32(a2,vsubq_s8(vreinterpretq_s8_u8(zB.val[0]),b8q),vld1q_s8(x+i+32)); - a3=vdotq_s32(a3,vsubq_s8(vreinterpretq_s8_u8(zB.val[1]),b8q),vld1q_s8(x+i+48)); - } - int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); - for(;i+32<=I;i+=32){ - uint8x16_t by=vld1q_u8(w4+(i>>1)); - uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4)); - acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q),vld1q_s8(x+i)); - acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q),vld1q_s8(x+i+16)); - } - sum=vaddvq_s32(acc); -#else - int32x4_t acc=vdupq_n_s32(0); - for(;i+32<=I;i+=32){ - uint8x16_t by=vld1q_u8(w4+(i>>1)); - uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4)); - int8x16_t w0=vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q); - int8x16_t w1=vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q); - int8x16_t x0=vld1q_s8(x+i), x1=vld1q_s8(x+i+16); - int16x8_t p=vmull_s8(vget_low_s8(w0),vget_low_s8(x0)); - p=vmlal_s8(p,vget_high_s8(w0),vget_high_s8(x0)); - acc=vpadalq_s16(acc,p); - p=vmull_s8(vget_low_s8(w1),vget_low_s8(x1)); - p=vmlal_s8(p,vget_high_s8(w1),vget_high_s8(x1)); - acc=vpadalq_s16(acc,p); - } - sum=vaddvq_s32(acc); -#endif -#elif defined(__VSX__) - const __vector unsigned char m4v=vec_splats((unsigned char)0x0F); - const __vector unsigned char sh4=vec_splats((unsigned char)4); - const __vector signed char b8v=vec_splats((signed char)8); - const __vector signed char vz=vec_splats((signed char)0); - __vector signed int acc=vec_splats(0); - for(;i+32<=I;i+=32){ - __vector unsigned char by=vec_xl(0,w4+(i>>1)); - __vector unsigned char lo=vec_and(by,m4v), hi=vec_sr(by,sh4); - __vector signed char w0=vec_sub((__vector signed char)vec_mergeh(lo,hi),b8v); - __vector signed char w1=vec_sub((__vector signed char)vec_mergel(lo,hi),b8v); - __vector signed char x0=vec_xl(0,(const signed char*)(x+i)); - __vector signed char x1=vec_xl(0,(const signed char*)(x+i+16)); - __vector __bool char n0=vec_cmplt(w0,vz), n1=vec_cmplt(w1,vz); - acc=vec_msum(vec_sel(x0,vec_sub(vz,x0),n0), - (__vector unsigned char)vec_sel(w0,vec_sub(vz,w0),n0),acc); - acc=vec_msum(vec_sel(x1,vec_sub(vz,x1),n1), - (__vector unsigned char)vec_sel(w1,vec_sub(vz,w1),n1),acc); - } - sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3); #endif - for(;i+1>1]; sum+=((int)(b&0xF)-8)*x[i]+((int)(b>>4)-8)*x[i+1]; } - if(i>1]; sum+=((int)(b&0xF)-8)*x[i]; } - return sum; -} -/* ---- ARM i8mm SMMLA tiled kernels ---------------------------------------- */ -#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8) -static inline int32x4_t mm_tile16(int32x4_t acc, int8x16_t wo, int8x16_t wo1, - int8x16_t xs, int8x16_t xs1){ - acc=vmmlaq_s32(acc, vcombine_s8(vget_low_s8(wo), vget_low_s8(wo1)), - vcombine_s8(vget_low_s8(xs), vget_low_s8(xs1))); - return vmmlaq_s32(acc, vcombine_s8(vget_high_s8(wo), vget_high_s8(wo1)), - vcombine_s8(vget_high_s8(xs), vget_high_s8(xs1))); -} -static void matmul_q_idot_mm(float *y, const int8_t *xq, const float *sx, const int8_t *q, - const float *scale, int S, int I, int O){ - #pragma omp parallel for schedule(static) - for(int o=0;o<(O&~1);o+=2){ - const int8_t *wo=q+(int64_t)o*I, *wo1=q+(int64_t)(o+1)*I; - float sc0=scale[o], sc1=scale[o+1]; - for(int s=0;s<(S&~1);s+=2){ - const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I; - int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0; - for(;i+64<=I;i+=64){ - a0=mm_tile16(a0,vld1q_s8(wo+i), vld1q_s8(wo1+i), vld1q_s8(xs+i), vld1q_s8(xs1+i)); - a1=mm_tile16(a1,vld1q_s8(wo+i+16),vld1q_s8(wo1+i+16),vld1q_s8(xs+i+16),vld1q_s8(xs1+i+16)); - a2=mm_tile16(a2,vld1q_s8(wo+i+32),vld1q_s8(wo1+i+32),vld1q_s8(xs+i+32),vld1q_s8(xs1+i+32)); - a3=mm_tile16(a3,vld1q_s8(wo+i+48),vld1q_s8(wo1+i+48),vld1q_s8(xs+i+48),vld1q_s8(xs1+i+48)); - } - for(;i+16<=I;i+=16) - a0=mm_tile16(a0,vld1q_s8(wo+i),vld1q_s8(wo1+i),vld1q_s8(xs+i),vld1q_s8(xs1+i)); - int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); - int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1); - int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3); - for(;i>1)), byo1=vld1q_u8(wo1+(i>>1)); - uint8x16_t cyo=vld1q_u8(wo+(i>>1)+16), cyo1=vld1q_u8(wo1+(i>>1)+16); - uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4)); - uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4)); - uint8x16x2_t ko =vzipq_u8(vandq_u8(cyo, m4q), vshrq_n_u8(cyo, 4)); - uint8x16x2_t ko1=vzipq_u8(vandq_u8(cyo1,m4q), vshrq_n_u8(cyo1,4)); - a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q), - vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q), - vld1q_s8(xs+i), vld1q_s8(xs1+i)); - a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q), - vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q), - vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16)); - a2=mm_tile16(a2, vsubq_s8(vreinterpretq_s8_u8(ko.val[0]),b8q), - vsubq_s8(vreinterpretq_s8_u8(ko1.val[0]),b8q), - vld1q_s8(xs+i+32), vld1q_s8(xs1+i+32)); - a3=mm_tile16(a3, vsubq_s8(vreinterpretq_s8_u8(ko.val[1]),b8q), - vsubq_s8(vreinterpretq_s8_u8(ko1.val[1]),b8q), - vld1q_s8(xs+i+48), vld1q_s8(xs1+i+48)); - } - for(;i+32<=I;i+=32){ - uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1)); - uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4)); - uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4)); - a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q), - vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q), - vld1q_s8(xs+i), vld1q_s8(xs1+i)); - a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q), - vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q), - vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16)); - } - int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3)); - int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1); - int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3); - for(;i+1>1], bo1=wo1[i>>1]; - int a0=(int)(bo&0xF)-8, a1=(int)(bo>>4)-8, b0=(int)(bo1&0xF)-8, b1=(int)(bo1>>4)-8; - int u0=xs[i],u1=xs[i+1],v0=xs1[i],v1=xs1[i+1]; - d00+=a0*u0+a1*u1; d01+=a0*v0+a1*v1; d10+=b0*u0+b1*u1; d11+=b0*v0+b1*v1; } - if(i>1], bo1=wo1[i>>1]; - int a0=(int)(bo&0xF)-8, b0=(int)(bo1&0xF)-8; - d00+=a0*xs[i]; d01+=a0*xs1[i]; d10+=b0*xs[i]; d11+=b0*xs1[i]; } - y[(int64_t)s*O+o] =(float)d00*sc0*sx[s]; - y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s]; - y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1]; - y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1]; - } - if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I; - y[(int64_t)s*O+o] =(float)dot_i4i8(wo, xs,I)*sc0*sx[s]; - y[(int64_t)s*O+(o+1)]=(float)dot_i4i8(wo1,xs,I)*sc1*sx[s]; } - } - if(O&1){ int o=O-1; const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o]; - #pragma omp parallel for schedule(static) - for(int s=0;s=2){ matmul_q_idot_mm(y,xq,sx,q,scale,S,I,O); return; } -#endif - #pragma omp parallel for schedule(static) - for(int o=0;o=2){ matmul_i4_idot_mm(y,xq,sx,q4,scale,S,I,O); return; } -#endif - #pragma omp parallel for schedule(static) - for(int o=0;oplanar repack. */ -static void planarize_i4_row(uint8_t *row, int I){ - uint8_t tmp[32]; - int nb=I/64; - for(int b=0;b>1]>>((src_lo&1)*4))&0xF; - uint8_t nib_hi=(blk[src_hi>>1]>>((src_hi&1)*4))&0xF; - tmp[k]=(uint8_t)(nib_lo|(nib_hi<<4)); - } - memcpy(blk,tmp,32); - } -} -static void planarize_i4(uint8_t *q4, int O, int I){ - int rb=(I+1)/2; - #pragma omp parallel for schedule(static) - for(int o=0;o>1))); - __m256i b1=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)+32)); - a0=coli_dpbusd256(a0,_mm256_and_si256(b0,m4), - _mm256_loadu_si256((const __m256i*)(x+i))); - a1=coli_dpbusd256(a1,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4), - _mm256_loadu_si256((const __m256i*)(x+i+32))); - a2=coli_dpbusd256(a2,_mm256_and_si256(b1,m4), - _mm256_loadu_si256((const __m256i*)(x+i+64))); - a3=coli_dpbusd256(a3,_mm256_and_si256(_mm256_srli_epi16(b1,4),m4), - _mm256_loadu_si256((const __m256i*)(x+i+96))); - } - __m256i acc=_mm256_add_epi32(_mm256_add_epi32(a0,a1),_mm256_add_epi32(a2,a3)); - for(;i+64<=I;i+=64){ - __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1))); - acc=coli_dpbusd256(acc,_mm256_and_si256(b0,m4), - _mm256_loadu_si256((const __m256i*)(x+i))); - acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4), - _mm256_loadu_si256((const __m256i*)(x+i+32))); - } - sum=hsum256_i32(acc); -#elif defined(__AVX2__) - const __m256i m4=_mm256_set1_epi8(0x0F); - const __m256i ones=_mm256_set1_epi16(1); - __m256i acc=_mm256_setzero_si256(); - for(;i+64<=I;i+=64){ - __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1))); - /* maddubs(u8, s8): u<=15, |x|<=127 -> coppia <= 3810, int16 sicuro */ - __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(b0,m4), - _mm256_loadu_si256((const __m256i*)(x+i))); - __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(b0,4),m4), - _mm256_loadu_si256((const __m256i*)(x+i+32))); - acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p0,ones)); - acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p1,ones)); - } - sum=hsum256_i32(acc); -#elif defined(__ARM_NEON) - int32x4_t acc=vdupq_n_s32(0); - for(;i+64<=I;i+=64){ - uint8x16_t b0=vld1q_u8(w4+(i>>1)), b1=vld1q_u8(w4+(i>>1)+16); - int8x16_t lo0=vreinterpretq_s8_u8(vandq_u8(b0,vdupq_n_u8(0x0F))); - int8x16_t lo1=vreinterpretq_s8_u8(vandq_u8(b1,vdupq_n_u8(0x0F))); - int8x16_t hi0=vreinterpretq_s8_u8(vshrq_n_u8(b0,4)); - int8x16_t hi1=vreinterpretq_s8_u8(vshrq_n_u8(b1,4)); -#if defined(__ARM_FEATURE_DOTPROD) - acc=vdotq_s32(acc,lo0,vld1q_s8(x+i)); - acc=vdotq_s32(acc,lo1,vld1q_s8(x+i+16)); - acc=vdotq_s32(acc,hi0,vld1q_s8(x+i+32)); - acc=vdotq_s32(acc,hi1,vld1q_s8(x+i+48)); -#else - int8x16_t xs0=vld1q_s8(x+i), xs1=vld1q_s8(x+i+16); - int8x16_t xs2=vld1q_s8(x+i+32), xs3=vld1q_s8(x+i+48); - int16x8_t m; - m=vmull_s8(vget_low_s8(lo0),vget_low_s8(xs0)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_high_s8(lo0),vget_high_s8(xs0)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_low_s8(lo1),vget_low_s8(xs1)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_high_s8(lo1),vget_high_s8(xs1)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_low_s8(hi0),vget_low_s8(xs2)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_high_s8(hi0),vget_high_s8(xs2)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_low_s8(hi1),vget_low_s8(xs3)); acc=vpadalq_s16(acc,m); - m=vmull_s8(vget_high_s8(hi1),vget_high_s8(xs3)); acc=vpadalq_s16(acc,m); -#endif - } - sum=vaddvq_s32(acc); -#endif - for(;i+64<=I;i+=64){ /* fallback scalare sui blocchi planari */ - const uint8_t *blk=w4+(i>>1); - for(int k=0;k<32;k++){ - sum+=(int32_t)(blk[k]&0xF)*x[i+k]; - sum+=(int32_t)(blk[k]>>4)*x[i+k+32]; - } - } - for(;i>1]; - sum+=(int32_t)(byte&0xF)*x[i]; - if(i+1>4)*x[i+1]; - } - return sum; -} - -/* ---- K1b (OPT-IN, IDOT_GS=1): IDOT planare A GRUPPI (fmt=4, gs%64==0) ----- - * Con gs=64 il gruppo di scala COINCIDE col blocco-piano da 64 elementi: il - * dot unsigned del blocco (2 dpbusd) -> int32 di gruppo, meno 8*somma(x) del - * gruppo, per la scala f32 del gruppo. Attivazioni int8 (stessa famiglia - * qrow_i8 del resto dell'IDOT): NON bit-identico al kernel f32 a gruppi -- - * per questo e' dietro flag, in attesa dell'ablazione. xsg = somme int32 - * per (riga, gruppo), calcolate dal chiamante in una passata esatta. - * EN: grouped planar IDOT, opt-in. With gs=64 the scale group IS the plane - * block; per-group unsigned dot minus 8*group-sum, times the group scale. - * int8 activations: not bit-identical to the f32 grouped kernel, hence the - * flag until the ablation blesses a default. */ -static void matmul_i4p_grouped_idot(float *y, const int8_t *xq, const float *sx, - const int32_t *xsg, const uint8_t *q4, - const float *scale, int S, int I, int O, int gs){ - int rb=(I+1)/2, ng=(I+gs-1)/gs, bpg=gs/64; /* blocchi-piano per gruppo */ - #pragma omp parallel for schedule(static) - for(int o=0;o>1); - const int8_t *xb=xr+base; -#if defined(coli_dpbusd256) - const __m256i m4=_mm256_set1_epi8(0x0F); - __m256i bb=_mm256_loadu_si256((const __m256i*)blk); - __m256i acc=_mm256_setzero_si256(); - acc=coli_dpbusd256(acc,_mm256_and_si256(bb,m4), - _mm256_loadu_si256((const __m256i*)xb)); - acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(bb,4),m4), - _mm256_loadu_si256((const __m256i*)(xb+32))); - d+=hsum256_i32(acc); -#elif defined(__AVX2__) - const __m256i m4=_mm256_set1_epi8(0x0F); - const __m256i ones=_mm256_set1_epi16(1); - __m256i bb=_mm256_loadu_si256((const __m256i*)blk); - __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(bb,m4), - _mm256_loadu_si256((const __m256i*)xb)); - __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(bb,4),m4), - _mm256_loadu_si256((const __m256i*)(xb+32))); - __m256i acc=_mm256_add_epi32(_mm256_madd_epi16(p0,ones), - _mm256_madd_epi16(p1,ones)); - d+=hsum256_i32(acc); -#else - for(int k=0;k<32;k++){ - d+=(int32_t)(blk[k]&0xF)*xb[k]; - d+=(int32_t)(blk[k]>>4)*xb[k+32]; - } -#endif - } - a=fmaf((float)(d-8*xg[g]),scl[g],a); - } - if(g*gs>1]; - d+=(int32_t)((i&1)?(byte>>4):(byte&0xF))*xr[i]; - } - a=fmaf((float)(d-8*xg[g]),scl[g],a); - } - y[(int64_t)s*O+o]=a*sx[s]; - } - } -} - -/* matmul IDOT planare (fmt=2): y = (dot_u - 8*xsum[s]) * scale[o] * sx[s]. - * Bit-identico a matmul_i4_idot: somme intere, identita' esatta. */ -static void matmul_i4p_idot(float *y, const int8_t *xq, const float *sx, const int32_t *xsum, - const uint8_t *q4, const float *scale, int S, int I, int O){ - int rb=(I+1)/2; - #pragma omp parallel for schedule(static) - for(int o=0;o - * bit-identico al path per-riga per associativita'. - * EN: 1x4 register tile — the weight block's load+mask cost is paid - * once per 4 activation rows. Integer sums: bit-identical to the - * per-row path by associativity. */ - const __m256i m4t=_mm256_set1_epi8(0x0F); - for(;s+4<=S;s+=4){ - const int8_t *x0=xq+(int64_t)s*I, *x1=x0+I, *x2=x1+I, *x3=x2+I; - __m256i a0=_mm256_setzero_si256(), a1=_mm256_setzero_si256(); - __m256i a2=_mm256_setzero_si256(), a3=_mm256_setzero_si256(); - int i=0; - for(;i+64<=I;i+=64){ - __m256i b =_mm256_loadu_si256((const __m256i*)(w+(i>>1))); - __m256i lo=_mm256_and_si256(b,m4t); - __m256i hi=_mm256_and_si256(_mm256_srli_epi16(b,4),m4t); - a0=coli_dpbusd256(a0,lo,_mm256_loadu_si256((const __m256i*)(x0+i))); - a0=coli_dpbusd256(a0,hi,_mm256_loadu_si256((const __m256i*)(x0+i+32))); - a1=coli_dpbusd256(a1,lo,_mm256_loadu_si256((const __m256i*)(x1+i))); - a1=coli_dpbusd256(a1,hi,_mm256_loadu_si256((const __m256i*)(x1+i+32))); - a2=coli_dpbusd256(a2,lo,_mm256_loadu_si256((const __m256i*)(x2+i))); - a2=coli_dpbusd256(a2,hi,_mm256_loadu_si256((const __m256i*)(x2+i+32))); - a3=coli_dpbusd256(a3,lo,_mm256_loadu_si256((const __m256i*)(x3+i))); - a3=coli_dpbusd256(a3,hi,_mm256_loadu_si256((const __m256i*)(x3+i+32))); - } - int32_t d0=hsum256_i32(a0), d1=hsum256_i32(a1); - int32_t d2=hsum256_i32(a2), d3=hsum256_i32(a3); - /* coda a coppie, unsigned: stessa identita' -8*xsum del per-riga */ - for(;i>1]; - d0+=(int32_t)(byte&0xF)*x0[i]; d1+=(int32_t)(byte&0xF)*x1[i]; - d2+=(int32_t)(byte&0xF)*x2[i]; d3+=(int32_t)(byte&0xF)*x3[i]; - if(i+1>4)*x0[i+1]; d1+=(int32_t)(byte>>4)*x1[i+1]; - d2+=(int32_t)(byte>>4)*x2[i+1]; d3+=(int32_t)(byte>>4)*x3[i+1]; - } - } - y[(int64_t)(s+0)*O+o]=(float)(d0-8*xsum[s+0])*sc*sx[s+0]; - y[(int64_t)(s+1)*O+o]=(float)(d1-8*xsum[s+1])*sc*sx[s+1]; - y[(int64_t)(s+2)*O+o]=(float)(d2-8*xsum[s+2])*sc*sx[s+2]; - y[(int64_t)(s+3)*O+o]=(float)(d3-8*xsum[s+3])*sc*sx[s+3]; - } -#endif - for(;s bit-identico. */ diff --git a/c/qwen36.c b/c/qwen36.c index 1b74becfe..e7515014e 100644 --- a/c/qwen36.c +++ b/c/qwen36.c @@ -67,6 +67,7 @@ static int qwen36_max_ctx(void) { #include "json.h" /* tokenizer.json parsing (reuse minimal parser) */ #include "qwen36_tier.h" /* optional CUDA VRAM expert tier */ #include "expert_ffn.h" /* routed experts: planar int4 kernel + layer runner */ +#include "idot.h" /* integer dot kernels for the dense trunk (COLI_DENSE_IDOT, COLI_DENSE_BITS) */ #ifdef COLI_SEGMENT_ADAPTER #include "segment_runtime.h" #include "segment_adapters.h" @@ -188,7 +189,15 @@ static void build_byte_sym(void){ g_unmap[cp]=(short)b; /* reverse: mapped codepoint -> original byte */ } } -static void push_id(int **ids,int *n,int *cap,int v){ if(*n==*cap){*cap*=2; *ids=realloc(*ids,*cap*sizeof(int));} (*ids)[(*n)++]=v; } +static void push_id(int **ids,int *n,int *cap,int v){ + if(*n==*cap){ + *cap*=2; + int *tmp=realloc(*ids,*cap*sizeof(int)); + if(!tmp){ fprintf(stderr,"qwen36: OOM reallocating token id buffer (%d entries)\n",*cap); exit(1); } + *ids=tmp; + } + (*ids)[(*n)++]=v; +} static int try_special(const char *s,int i,int n,int *id_out){ int best_len=0,best_id=-1; @@ -227,7 +236,22 @@ static int pretok_end(const char *s,int i,int n){ if(s[i]==' '&&i+10) return nl_end; + /* rule6: \s+(?!\S) -- a run followed by a non-space keeps its last char + * for the next piece, which then takes it as " x" or " ." (HF: " <" is + * " " then " <", not " " then "<"). A single char cannot back off and + * falls to rule7, \s+, which takes it whole. */ + if(ki) return last; + return k; + } return i+adv; } static void bpe_piece(const char *piece,int len,int **ids,int *n,int *cap){ @@ -236,7 +260,13 @@ static void bpe_piece(const char *piece,int len,int **ids,int *n,int *cap){ for(int b=0;b1){ int best=-1,besti=-1; @@ -256,22 +286,39 @@ static void bpe_piece(const char *piece,int len,int **ids,int *n,int *cap){ for(int k=0;k` + * together with the `.` before it, so the special was never at a piece start and + * got encoded as text (#1653: `X.<|im_end|>` was 7 tokens instead of 3, and + * every chat turn ending in punctuation paid +4). qwen38.c already splits this + * way; this is the same shape. */ +static int next_special(const char *s,int i,int n){ + for(int k=i;k0) return k; } + return n; +} static void encode_text(const char *text,int **out_ids,int *out_n){ int cap=1024,n=0; int *ids=malloc(cap*sizeof(int)); int tlen=(int)strlen(text); int i=0; while(i0){ push_id(&ids,&n,&cap,sid); i+=L; continue; } - int j=pretok_end(text,i,tlen); if(j<=i) j=i+utf8_adv(text,i,tlen); - if(j>tlen) j=tlen; - bpe_piece(text+i,j-i,&ids,&n,&cap); - i=j; + /* Ordinary text runs to the next added token, and the pre-tokenizer + * sees that boundary as the end of its input, exactly as HF's does. */ + int end=next_special(text,i+1,tlen); + while(iend) j=end; + bpe_piece(text+i,j-i,&ids,&n,&cap); + i=j; + } } *out_ids=ids; *out_n=n; } -/* Load Qwen tokenizer.json and build an id->piece table. Only needs the - * "model.vocab" map (piece string -> id); merges are irrelevant for decoding. */ +/* Load Qwen tokenizer.json and build an id->piece table from model.vocab + * and added_tokens. Merges are irrelevant for decoding. */ static void load_tokenizer(const char *path){ FILE *f = fopen(path, "rb"); if (!f) { fprintf(stderr, "[tok] cannot open %s\n", path); return; } @@ -285,18 +332,43 @@ static void load_tokenizer(const char *path){ jval *vocab = json_get(model, "vocab"); if (!vocab) vocab = json_get(model, "tokens"); if (!vocab) { fprintf(stderr, "[tok] no model.vocab/tokens in %s\n", path); free(buf); return; } + jval *adds = json_get(root, "added_tokens"); + int mx = 0; if (vocab->t == J_OBJ){ for (int i=0;ilen;i++){ int id=(int)vocab->kids[i]->num; if(id>mx)mx=id; } } else { mx = vocab->len - 1; } + if (adds && adds->t==J_ARR){ + for (int k=0;klen;k++){ + jval *t = adds->kids[k]; + int id = (int)jnum(t,"id"); + if (id > mx) mx = id; + } + } + g_tok = calloc((size_t)mx+1, sizeof(char*)); if (vocab->t == J_OBJ){ for (int i=0;ilen;i++){ int id=(int)vocab->kids[i]->num; if(id>=0 && id<=mx) g_tok[id]=strdup(vocab->keys[i]); } } else { for (int i=0;ilen;i++){ if(vocab->kids[i] && vocab->kids[i]->t==J_STR) g_tok[i]=strdup(vocab->kids[i]->str); } } + if (adds && adds->t==J_ARR){ + for (int k=0;klen;k++){ + jval *t = adds->kids[k]; + const char *c = jstr(t,"content"); + int id = (int)jnum(t,"id"); + /* Only the non-special ones: , , , + * are text the gateway parses. Special tokens + * (<|im_start|>, <|endoftext|>, ...) keep decoding to nothing, + * as reference decoding does with skip_special_tokens. */ + jval *sp = json_get(t,"special"); + if (sp && sp->t==J_BOOL && sp->boolean) continue; + if (c && id>=0 && id<=mx && !g_tok[id]) + g_tok[id]=strdup(c); + } + } g_tok_n = mx+1; /* ---- encoder tables (text -> ids) ---- */ @@ -333,7 +405,6 @@ static void load_tokenizer(const char *path){ smap_put(&g_merge, key, r); } } - jval *adds = json_get(root, "added_tokens"); if (adds && adds->t==J_ARR && g_nspecial==0){ g_nspecial = adds->len; g_sp_str = malloc(g_nspecial*sizeof(char*)); @@ -439,7 +510,7 @@ static void sse_chunk(const char *json){ * <0xXX> byte-fallback tokens emit the raw byte directly. */ static void decode_id_to_bytes(int id, unsigned char *out, int *outn){ *outn = 0; - if (!g_tok || id<0 || id>=g_tok_n) return; + if (!g_tok || id<0 || id>=g_tok_n || !g_tok[id]) return; const unsigned char *pc = (const unsigned char*)g_tok[id]; /* byte-fallback token: <0xXX> -> raw byte */ if (pc[0]=='<' && pc[1]=='0' && pc[2]=='x' && pc[5]=='>'){ @@ -613,18 +684,47 @@ typedef struct { /* Gated DeltaNet (linear_attention) dims, read from qwen36_meta.json. */ int dn_vheads, dn_kheads, dn_kdim, dn_vdim, dn_convk, dn_conv_dim; int expert_gs; /* expert scale group size along input dim; 0 = per-row */ + /* Mixed expert layout (convert_qwen36.py --down-bits): gate/up stay int4 + * (ebits, expert_gs), down_proj is int8 with its own group size. One slab + * per expert, [gate int4 packed | up int4 packed | down int8], 2*inter*hidden + * bytes -- told apart from int4 (1.5x) and int8 (3x) by size, like today. */ + int expert_down_bits, expert_down_gs; } Cfg; +/* ---- Dense int8: a dense matrix that is quantized to int8 during load + * (load_tq, below matmul_d) instead of loaded as f32 and quantized in a + * separate pass afterward -- so the f32 staging buffer for THIS matrix alone + * is what's briefly resident, not every dense matrix in the model at once. + * `w` is the f32 copy: kept (and `q`/`sc` left NULL) when COLI_DENSE_I8=0, + * the reference/parity path; freed once `q`/`sc` are populated otherwise + * (COLI_KEEP_F32=1 keeps it alongside them, for debugging). matmul_d + * dispatches on q!=NULL directly -- no pointer-keyed scan. */ +/* q/sc: int8 rows with one scale per row (the classic copy, what the VRAM + * tier uploads). q4/sg: the same matrix as int4 planar blocks of 64 with one + * scale per group (COLI_DENSE_BITS=4), the layout the K1b grouped kernel + * reads; ng = I/64 groups per row. */ +typedef struct { const float *w; int8_t *q; float *sc; int I, O; uint8_t *q4; float *sg; int ng; } QW; +static void qw_free(QW *w) { + free((void*)w->w); free(w->q); free(w->sc); free(w->q4); free(w->sg); + w->w = NULL; w->q = NULL; w->sc = NULL; w->q4 = NULL; w->sg = NULL; w->ng = 0; +} + /* ---------- per-layer dense weights ---------- */ typedef struct { - float *in_ln, *post_ln, *q, *k, *v, *o, *qn, *kn, *gate, *gate_bias; - float *sh_g, *sh_u, *sh_d, *sh_gate; /* shared expert (dense f32) + shared_expert_gate */ + float *in_ln, *post_ln, *qn, *kn, *gate_bias; + QW q, k, v, o, gate; + QW sh_g, sh_u, sh_d; float *sh_gate; /* shared expert (dense, int8-during-load) + shared_expert_gate */ /* Gated DeltaNet (linear_attention) dense weights (f16->f32 via st_read_f32). */ - float *dn_qkv, *dn_z, *dn_b, *dn_a; /* in_proj_qkv/z/b/a */ + QW dn_qkv, dn_z; float *dn_b, *dn_a; /* in_proj_qkv/z (int8-during-load), b/a (not dense-matmul'd) */ float *dn_conv; /* conv1d.weight [conv_dim, convk] (groups=conv_dim) */ float *dn_dtbias, *dn_alog; /* dt_bias[vh], A_log[vh] */ float *dn_norm; /* RMSNormGated weight [vdim] */ - float *dn_out; /* out_proj [hidden, value_dim] */ + QW dn_out; /* out_proj [hidden, value_dim] */ + /* VRAM copies the tier placed (qt_dense handle + 1, 0 = stays on the CPU): + * the DeltaNet out_proj, the attention q/k/v/o and the shared expert's + * three matrices. Offered per layer as "dnout", "attnproj", "shexp"; + * see trunk_offer_dense / trunk_place_dense. */ + int qth_dnout, qth_q, qth_k, qth_v, qth_o, qth_shg, qth_shu, qth_shd; } Layer; /* ---------- LRU expert cache (int8 weights + per-row float scales) ---------- */ @@ -638,17 +738,27 @@ typedef struct { int n, cap; } LCache; +/* CACHE_ROUTE telemetry (docs/CACHE_ROUTE.md): the lever changes which experts + * run, so it carries its own meters. Only touched when the lever is on. */ +typedef struct { + uint64_t slots, swaps, swaps_vram; /* chosen slots; not in the true top-K; of those, VRAM-resident */ + uint64_t agree_hit, agree_tot; /* |chosen ∩ true top-K| summed, K summed */ + double kl_sum; uint64_t kl_n; /* mean KL(true top-K mass || chosen mass) */ +} RouteStats; + typedef struct { Cfg c; shards S; int quant_bits; - float *embed, *lm_head, *final_norm; + float *embed, *final_norm; + QW lm_head; Layer *L; LCache *cache; /* [n_layers] */ int *active_of; /* [n_layers] original->active idx (Phase 2: identity for all layers) */ float **DN_rec; /* [n_layers] recurrent state S[h]=[kdim,vdim] for DeltaNet layers (NULL for attn) */ float **DN_conv; /* [n_layers] conv ring [conv_dim, convk-1] for DeltaNet layers (NULL for attn) */ uint64_t clock, hits, miss; + RouteStats route; /* CACHE_ROUTE / ROUTE_AGREE meters */ /* Telemetria per la dashboard (Brain/Profile): tempo di lettura esperti * accumulato dall'avvio, e bitmap degli esperti toccati nel turno. */ double t_disk; @@ -680,6 +790,13 @@ static volatile unsigned pilot_r = 0, pilot_w = 0; static Model *pilot_m = NULL; static int g_pilot = 0; static int g_wide = 1; +/* CACHE_ROUTE family, same names and defaults as the GLM engine (docs/CACHE_ROUTE.md). */ +static int g_cache_route = 0; +static int g_route_j = 2; +static int g_route_m = 12; +static float g_route_p = 0.f; +static float g_route_alpha = 1.f; +static int g_route_agree = 0; static void pilot_prefetch(Model *m, int lnext, const float *x, int S); static void *pilot_worker(void *arg); @@ -810,85 +927,11 @@ static void matmul(float *y, const float *x, const float *W, int S, int I, int O } } -/* y[1,O] = x[1,I] @ W^T with W quantized: q[O,I] int8 + scale per row. */ -#if defined(__ARM_NEON) -#include -static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) { - int32x4_t acc = vdupq_n_s32(0); - int8x16_t va = vld1q_s8(a), vb = vld1q_s8(b); -#if defined(__ARM_FEATURE_DOTPROD) - acc = vdotq_s32(acc, va, vb); -#else - acc = vpadalq_s16(acc, vmull_s8(vget_low_s8(va), vget_low_s8(vb))); - acc = vpadalq_s16(acc, vmull_s8(vget_high_s8(va), vget_high_s8(vb))); -#endif - return vaddvq_s32(acc); -} -#endif -static void matmul_q(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) { -#if defined(__ARM_NEON) - /* IDOT is opt-in, not default-on: this path quantizes the ACTIVATIONS to - * Q8_0 per 16-element block, which the scalar path does not, so the two are - * not numerically equivalent. olmoe shipped it default-on and it cost - * token-exactness end to end (#1044, fixed in af48fe8 by making it opt-in); - * qwen36 inherited the same default from the same family of kernels. The - * tiny-oracle gate would not have caught it -- that job runs on x86. */ - static int idot = -1; - if (idot < 0) { const char *e = getenv("IDOT"); idot = (e && atoi(e)); } - if (idot && I % 16 == 0 && I <= 4096) { - int nb = I / 16; int8_t xi[4096]; float xs[256]; - for (int b = 0; b < nb; b++) { - const float *xb = x + b*16; - float am = 0.f; for (int i = 0; i < 16; i++) { float a = fabsf(xb[i]); if (a > am) am = a; } - float s = am/127.f; if (s < 1e-12f) s = 1e-12f; - xs[b] = s; float inv = 1.f/s; - for (int i = 0; i < 16; i++) xi[b*16+i] = (int8_t)lrintf(xb[i]*inv); - } - #pragma omp parallel for schedule(static) - for (int o = 0; o < O; o++) { - const int8_t *w = q + (int64_t)o * I; - float acc = 0.f; - for (int b = 0; b < nb; b++) acc += xs[b]*(float)dot_i8_16(xi+b*16, w+b*16); - y[o] = acc * scale[o]; - } - return; - } -#endif -#if defined(__AVX2__) && defined(__FMA__) - /* Hand-vectorized int8->f32 GEMV (gcc does not auto-vectorize the - * convert+accumulate chain). 32 weights per iteration, FMA accumulate. */ - #pragma omp parallel for schedule(static) if(O >= 256) - for (int o = 0; o < O; o++) { - const int8_t *w = q + (int64_t)o * I; - __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps(); - __m256 a2 = _mm256_setzero_ps(), a3 = _mm256_setzero_ps(); - int i = 0; - for (; i + 32 <= I; i += 32) { - __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); - __m128i b1 = _mm_loadu_si128((const __m128i*)(w + i + 16)); - a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0); - a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1); - a2 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+16), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a2); - a3 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+24), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a3); - } - a0 = _mm256_add_ps(_mm256_add_ps(a0,a1), _mm256_add_ps(a2,a3)); - __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1)); - s = _mm_add_ps(s, _mm_movehl_ps(s,s)); - s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1)); - float acc = _mm_cvtss_f32(s); - for (; i < I; i++) acc += x[i] * (float)w[i]; - y[o] = acc * scale[o]; - } -#else - #pragma omp parallel for schedule(static) - for (int o = 0; o < O; o++) { - const int8_t *w = q + (int64_t)o * I; - float acc = 0.f; - for (int i = 0; i < I; i++) acc += x[i] * (float)w[i]; - y[o] = acc * scale[o]; - } -#endif -} +/* y[1,O] = x[1,I] @ W^T with W quantized: q[O,I] int8 + scale per row. + * matmul_q lives in qgemv.h so tests/test_qgemv.c can link the exact kernel + * the engine runs (qwen36.c has a main() and cannot itself be linked into a + * test binary). */ +#include "qgemv.h" /* Multi-row dense-int8 prefill kernel. matmul_q() above is deliberately kept * as the S=1 decode implementation: its four AVX accumulators stay in @@ -978,8 +1021,14 @@ static void matmul_q_batch(float *y, const float *x, const int8_t *q, } /* Group-scaled int8 GEMV: one f32 scale per `gs` input elements per row - * (gs64 expert containers). Row layout of `scale`: [O][I/gs] row-major. */ + * (gs64 expert containers). Row layout of `scale`: [O][I/gs] row-major. + * matmul_q_gs lives in gsgemv.h so tests/test_gsgemv.c can link the exact + * kernel the engine runs. */ static int g_expert_gs = 0; /* set from qwen36_meta.json (expert_gs) at load */ +#include "gsgemv.h" + +static int g_expert_mixed = 0; /* mixed layout on disk (int4 gate/up, int8 down) */ +static int g_expert_down_gs = 0; /* down_proj's group size in the mixed layout (0 = per row) */ /* 1 = expert container packs int4 (tier fmt=4); 0 = int8 per-row (tier fmt=1). * Same signal main's nbytes probe and tier_warmstart receive; the decode path * needs it to offer int8 experts (#1391): on an int8 container e->g4 is NULL. */ @@ -992,6 +1041,12 @@ static int g_expert_is_int4 = 1; * int8 copy) and with QWEN_EXPERT_KERNEL=0, which keeps the historical * unpack-to-int8 path for A/Bs. Decided once from the container itself. */ static int container_layer_is_int4(Model *m, int layer); +/* The routed experts take the same integer path as the dense trunk + * (expert_ffn.h mode 1: activation to int8 once per row, dpbusd against the + * planar nibbles). Measured on the 35B: expert compute 22.7 to 15.9 + * ms/token for +0.1% perplexity. QWEN_EXPERT_ACT=f32 restores mode 0, f32 + * activations and the bit-identical contract with the pair kernels. */ +static int xf_act_mode(void){ static int v=-1; if(v<0){ const char *e=getenv("QWEN_EXPERT_ACT"); v=(e&&!strcmp(e,"f32"))?0:1; } return v; } static int xf_mode(Model *m) { static int v = -1; if (v >= 0) return v; @@ -1030,61 +1085,93 @@ static void tier_offer_slot(int layer, int eid, const Slot *s) { qt_note(layer, eid, (const uint8_t *)s->g, (const uint8_t *)s->u, (const uint8_t *)s->d, s->gs, s->us, s->ds); } -static void matmul_q_gs(float *y, const float *x, const int8_t *q, const float *scale, - int I, int O, int gs) { - int ng = (I + gs - 1) / gs; -#if defined(__AVX2__) && defined(__FMA__) - if ((gs & 31) == 0) { - #pragma omp parallel for schedule(static) if(O >= 256) - for (int o = 0; o < O; o++) { - const int8_t *w = q + (int64_t)o * I; - const float *sc = scale + (int64_t)o * ng; - float acc = 0.f; - for (int gi = 0; gi < ng; gi++) { - __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps(); - int base = gi * gs, end = base + gs; if (end > I) end = I; - for (int i = base; i + 16 <= end; i += 16) { - __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); - a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0); - a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1); - } - a0 = _mm256_add_ps(a0, a1); - __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1)); - s = _mm_add_ps(s, _mm_movehl_ps(s,s)); - s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1)); - acc += _mm_cvtss_f32(s) * sc[gi]; - } - y[o] = acc; - } - return; - } -#endif - #pragma omp parallel for schedule(static) if(O >= 256) - for (int o = 0; o < O; o++) { - const int8_t *w = q + (int64_t)o * I; - const float *sc = scale + (int64_t)o * ng; - float acc = 0.f; - for (int gi = 0; gi < ng; gi++) { - int base = gi * gs, end = base + gs; if (end > I) end = I; - float part = 0.f; - for (int i = base; i < end; i++) part += x[i] * (float)w[i]; - acc += part * sc[gi]; - } - y[o] = acc; - } -} /* Expert-GEMV dispatch: per-row scales (classic) or grouped (gs64 container). */ static void matmul_qe(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) { if (g_expert_gs) matmul_q_gs(y, x, q, scale, I, O, g_expert_gs); else matmul_q(y, x, q, scale, I, O); } +/* down_proj: in the mixed layout it carries its own scale layout (int8, per + * row or expert_down_gs), everywhere else it is matmul_qe. */ +static void matmul_qd(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) { + if (!g_expert_mixed) { matmul_qe(y, x, q, scale, I, O); return; } + if (g_expert_down_gs) matmul_q_gs(y, x, q, scale, I, O, g_expert_down_gs); + else matmul_q(y, x, q, scale, I, O); +} /* ---- Dense int8: per-row quantized copies of the large f32 matrices. - * matmul_d dispatches via pointer lookup to matmul_q; COLI_DENSE_I8=0 falls - * back to f32 (reference path for parity tests). ~4x less memory traffic. */ -#define QDW_MAX 1024 -static struct { const float *w; int8_t *q; float *sc; int I, O; } g_qdw[QDW_MAX]; -static int g_qdw_n = 0; + * matmul_d dispatches directly off QW.q (no pointer-keyed scan -- see QW, + * above Layer); COLI_DENSE_I8=0 falls back to f32 (QW.w, reference path for + * parity tests). ~4x less memory traffic. */ +/* ---- Dense trunk, integer dot products ------------------------------------ + * + * Every dense GEMV of a token (DeltaNet projections and out_proj, attention + * q/k/v/o, the shared expert, lm_head) used to multiply int8 weights by f32 + * activations: each weight byte converted to f32 and fed to an FMA, eight + * weights per instruction. Measured on lm_head (248320 x 2048 int8, 508 MB) + * that runs at 29 GB/s on a 16-core AVX-512 host whose memory bus does 80: + * the kernel, not the bus, was the limit, and the dense part of a token is + * 1.9 GB of int8 on the 35B, three times the routed experts. + * + * The activation is now quantized to int8 once per call (one scale, + * amax/127, the qrow_i8 contract the expert IDOT already uses) and the dot + * is integer: 32 weights per instruction on AVX2 (maddubs), 64 on AVX-512 + * VNNI, exact int32 sums scaled once per output. Not bit-identical to the + * f32 path (the activation is rounded): measured on the 35B, +1.0% + * perplexity on 4 x 512 tokens, lm_head 12.6 to 10.2 ms/token, decode + * 6.71 to 7.35 tok/s alone and 8.23 with the experts' int8 activations. + * COLI_DENSE_IDOT=0 restores the f32-activation kernel. + * + * COLI_DENSE_BITS=4 additionally stores the dense matrices as int4 in blocks + * of 64 with one scale per block, the K1b planar layout, halving the bytes + * the token reads; lm_head alone goes from 508 to 254 MB. It implies the + * integer dot (that layout has no f32 kernel). Same gate: measured. */ +static int dense_idot_on(void){ static int v=-1; if(v<0){ const char *e=getenv("COLI_DENSE_IDOT"); v=!(e&&*e=='0'); } return v; } +static int dense_bits(void){ static int v=-1; if(v<0){ const char *e=getenv("COLI_DENSE_BITS"); v=(e&&atoi(e)==4)?4:8; } return v; } + +/* f32 rows -> int4 in blocks of 64 with one f32 scale per block, packed as the + * K1b planar layout (unsigned nibbles v+8, block b: lo nibbles = elements + * b*64..b*64+31, hi = b*64+32..b*64+63). The quantizer is the symmetric + * absmax/7 the expert containers use. */ +static void pack_int4_g64_planar(const float *w, uint8_t *q4, float *sg, int O, int I){ + int rb = I / 2, ng = I / 64; + #pragma omp parallel for schedule(static) + for (int o = 0; o < O; o++) { + const float *wr = w + (int64_t)o * I; + uint8_t *row = q4 + (int64_t)o * rb; + float *sr = sg + (int64_t)o * ng; + for (int g = 0; g < ng; g++) { + const float *blk = wr + g * 64; + float amax = 0.f; + for (int k = 0; k < 64; k++) { float a = fabsf(blk[k]); if (a > amax) amax = a; } + /* absmax/7 is where the search starts; the scale is then refined + * by least squares against the rounded codes (s = w.q / q.q) and + * the candidate with the smallest squared error wins. Three + * rounds: measured on the 35B this recovers a third of the + * perplexity absmax alone loses at 4 bits. */ + float best_s = amax / 7.f; if (best_s < 1e-8f) best_s = 1e-8f; + int best_q[64]; double best_err = 1e30; + float s = best_s; + for (int round = 0; round < 4; round++) { + int q[64]; double err = 0, wq = 0, qq = 0; + float inv = 1.f / s; + for (int k = 0; k < 64; k++) { + int v = (int)lrintf(blk[k] * inv); if (v > 7) v = 7; if (v < -8) v = -8; + q[k] = v; double d = (double)blk[k] - (double)v * s; err += d * d; + wq += (double)blk[k] * v; qq += (double)v * v; + } + if (err < best_err) { best_err = err; best_s = s; memcpy(best_q, q, sizeof q); } + if (qq <= 0) break; + float ns = (float)(wq / qq); /* least-squares scale for these codes */ + if (ns <= 0.f || ns == s) break; + s = ns; + } + sr[g] = best_s; + uint8_t *dst = row + g * 32; + for (int k = 0; k < 32; k++) + dst[k] = (uint8_t)((best_q[k] + 8) | ((best_q[k + 32] + 8) << 4)); + } + } +} #ifdef COLI_QWEN_BATCH_TEST static uint64_t g_qwen_matmul_d_calls; #endif @@ -1094,10 +1181,36 @@ static int dense_i8_on(void){ static int v=-1; if(v<0){ const char *e=getenv("CO * the time. */ static int kv_prefix_off(void){ const char *e=getenv("COLI_KV_PREFIX"); return e && *e=='0'; } static int dense_batch_on(void){ const char *e=getenv("QWEN_DENSE_BATCH"); return !(e&&*e=='0'); } -static void qdw_register(const float *W, int I, int O){ - if (!W || !dense_i8_on() || g_qdw_n >= QDW_MAX) return; +/* COLI_DENSE_INT4 names which components take the int4 copy when + * COLI_DENSE_BITS=4: a comma list of lmhead, dnproj, dnout, attn, shexp, + * router; unset means all of them. The parts of the trunk pay 4 bits + * differently (measured: the attention projections and the DeltaNet input + * projections cost the most perplexity, lm_head the least per byte saved), + * so the default is chosen per component from the numbers, not for the + * whole trunk at once. */ +static int dense_int4_wanted(const char *tag){ + if (dense_bits() != 4 || !tag) return 0; + const char *e = getenv("COLI_DENSE_INT4"); + if (!e || !*e) return 1; + size_t n = strlen(tag); + for (const char *p = e; *p; ) { + while (*p == ',' || *p == ' ') p++; + const char *q = p; while (*q && *q != ',' && *q != ' ') q++; + if ((size_t)(q - p) == n && !strncmp(p, tag, n)) return 1; + p = q; + } + return 0; +} +/* Per-row max-abs / round-clamp int8 quantization of an in-memory f32 matrix + * W [O][I] row-major -- same math load_tq runs during streamed load, factored + * out so it can also run on a caller-owned buffer directly (tests that + * synthesize weights in memory, without a shard file to load from). Does not + * touch out->w -- the caller sets that (or leaves it, e.g. load_tq frees it + * right after). `tag` selects the int4 planar copy per dense_int4_wanted + * (NULL/COLI_DENSE_BITS!=4: skipped, out->q4 stays NULL). */ +static void qw_quantize(const float *W, int I, int O, const char *tag, QW *out) { int8_t *q = malloc((size_t)O*I); float *sc = malloc((size_t)O*sizeof(float)); - if (!q || !sc) { free(q); free(sc); return; } + if (!q || !sc) { fprintf(stderr, "OOM qw_quantize\n"); exit(1); } #pragma omp parallel for schedule(static) for (int o = 0; o < O; o++) { const float *r = W + (int64_t)o*I; float am = 0.f; @@ -1106,20 +1219,65 @@ static void qdw_register(const float *W, int I, int O){ int8_t *d = q + (int64_t)o*I; for (int i = 0; i < I; i++) { int v = (int)lrintf(r[i]*inv); if (v>127) v=127; if (v<-127) v=-127; d[i] = (int8_t)v; } } - g_qdw[g_qdw_n].w=W; g_qdw[g_qdw_n].q=q; g_qdw[g_qdw_n].sc=sc; g_qdw[g_qdw_n].I=I; g_qdw[g_qdw_n].O=O; g_qdw_n++; + out->q = q; out->sc = sc; out->I = I; out->O = O; + out->q4 = NULL; out->sg = NULL; out->ng = 0; + if (dense_int4_wanted(tag) && I % 64 == 0) { + uint8_t *q4 = malloc((size_t)O*(I/2)); float *sg = malloc((size_t)O*(I/64)*sizeof(float)); + if (q4 && sg) { + pack_int4_g64_planar(W, q4, sg, O, I); + out->q4 = q4; out->sg = sg; out->ng = I/64; + } else { free(q4); free(sg); } + } } -static void matmul_d(float *y, const float *x, const float *W, int S, int I, int O){ +static void matmul_d(float *y, const float *x, const QW *w, int S, int I, int O){ #ifdef COLI_QWEN_BATCH_TEST g_qwen_matmul_d_calls++; #endif - for (int i = 0; i < g_qdw_n; i++) if (g_qdw[i].w == W && g_qdw[i].I == I) { + if (w->q) { + if (w->q4 || dense_idot_on()) { + /* integer dot: the activation rows to int8 once, then the K1b + * grouped kernel (int4 planar) or the per-row int8 kernel */ + int ng = I / 64; + int8_t *xq = malloc((size_t)S * I); + float *sx = malloc((size_t)S * sizeof(float)); + int32_t *xsg = w->q4 ? malloc((size_t)S * ng * sizeof(int32_t)) : NULL; + if (xq && sx && (!w->q4 || xsg)) { + for (int s = 0; s < S; s++) + sx[s] = dense_act_i8(x + (int64_t)s * I, I, xq + (int64_t)s * I, xsg ? xsg + (int64_t)s * ng : NULL); + if (w->q4) matmul_i4p_grouped_idot(y, xq, sx, xsg, w->q4, w->sg, S, I, O, 64); + else matmul_q_idot(y, xq, sx, w->q, w->sc, S, I, O); + free(xq); free(sx); free(xsg); + return; + } + free(xq); free(sx); free(xsg); /* out of memory: the f32 path below */ + } if (S > 1 && dense_batch_on()) - matmul_q_batch(y, x, g_qdw[i].q, g_qdw[i].sc, S, I, O); + matmul_q_batch(y, x, w->q, w->sc, S, I, O); else - for (int s = 0; s < S; s++) matmul_q(y+(int64_t)s*O, x+(int64_t)s*I, g_qdw[i].q, g_qdw[i].sc, I, O); + for (int s = 0; s < S; s++) matmul_q(y+(int64_t)s*O, x+(int64_t)s*I, w->q, w->sc, I, O); return; } - matmul(y, x, W, S, I, O); + matmul(y, x, w->w, S, I, O); +} +/* A dense matrix the tier placed in VRAM (handle+1 kept in the Layer, 0 = CPU): + * a device matmul, or 0 and the caller runs matmul_d as before. The + * tier turns a failing handle off itself, so the fallback is permanent. */ +static inline int qtd_batch(int hp1, float *y, const float *x, int S, int I, int O){ + return hp1 > 0 && qt_dense_matmul_batch(hp1 - 1, y, x, S, I, O); +} +static inline int qtd(int hp1, float *y, const float *x, int I, int O){ + return hp1 > 0 && qt_dense_matmul(hp1 - 1, y, x, I, O); +} +/* Bytes of w's dense-i8 copy (int8 rows + per-row scales), 0 when there is + * none (COLI_DENSE_I8=0): nothing to offer, the CPU path stands. */ +static size_t qdw_bytes(const QW *w){ + return w->q ? (size_t)w->I * w->O + (size_t)w->O * sizeof(float) : 0; +} +/* Upload w's dense-i8 copy to `dev`; handle+1, or 0 when it stays on the CPU. */ +static int qdw_place(const QW *w, int dev){ + if (dev == QT_PLACE_CPU || !w->q) return 0; + int h = qt_dense_init(w->q, w->sc, w->I, w->O, dev); + return h >= 0 ? h + 1 : 0; } /* rmsnorm over a row of length D (in-place capable: out may == x). @@ -1266,6 +1424,7 @@ static void load_meta(Cfg *c, const char *snap) { G("q_head_dim", q_head_dim); G("k_head_dim", k_head_dim); G("v_head_dim", v_head_dim); G("o_in", o_in); G("rope_dim", rope_dim); G("qk_rope_head_dim", rope_dim); G("expert_gs", expert_gs); + G("expert_down_bits", expert_down_bits); G("expert_down_gs", expert_down_gs); G("num_experts", n_experts); G("topk", topk); G("moe_inter", inter); G("shared_inter", shared_inter); G("n_group", n_group); G("topk_group", topk_group); @@ -1337,6 +1496,27 @@ static float *load_t_n(Model *m, const char *name, int64_t want) { return p; } +/* Dense matrix load, quantized to int8 (+ int4 planar per `tag`, see + * dense_int4_wanted) DURING loading rather than in a separate pass over the + * whole model afterward (see QW, above Layer): reads `name` (I*O elements, + * same size discipline as load_t_n), and when `quantize` && COLI_DENSE_I8 is + * on, quantizes it via qw_quantize and frees the f32 staging buffer right + * away -- so at most one dense matrix's f32 copy is ever resident at a time, + * not the whole model's. `quantize` is false for loaders that never ran + * through the old post-hoc qdw_register pass either (the Segment/Edge + * adapters build partial or auxiliary models straight off + * model_init_range/load_t_n, never main()'s dense-i8 block) -- passing it + * through keeps their f32-only behavior exactly as it was; `tag` is unused + * on that path. */ +static void load_tq(Model *m, const char *name, int I, int O, int quantize, const char *tag, QW *out) { + float *p = load_t_n(m, name, (int64_t)I * O); + out->w = p; out->q = NULL; out->sc = NULL; out->I = I; out->O = O; + out->q4 = NULL; out->sg = NULL; out->ng = 0; + if (!quantize || !dense_i8_on()) return; + qw_quantize(p, I, O, tag, out); + if (getenv("COLI_KEEP_F32")) out->w = p; else { free(p); out->w = NULL; } +} + static void model_init_range(Model *m, const char *snap, int cap, int bits, int layer_begin, int layer_end, int load_boundaries, int allocate_state) { @@ -1362,9 +1542,17 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, exit(1); } double t0 = now_s(); + /* Quantize during load only for the full-model path (main()'s static Model, + * load_boundaries=1): the Segment/Edge adapters build partial or auxiliary + * models straight off this same loop and never ran the old post-hoc + * qdw_register pass either, so gating on load_boundaries keeps their + * f32-only numerics exactly as they were. */ + int quantize_dense = load_boundaries && dense_i8_on(); + int qcount = 0; double qfreed = 0; if (load_boundaries) { - m->embed = load_t_n(m, "model.embed_tokens.weight", (int64_t)c->vocab * c->hidden); - m->lm_head = load_t_n(m, "lm_head.weight", (int64_t)c->vocab * c->hidden); + m->embed = load_t_n(m, "model.embed_tokens.weight", (int64_t)c->vocab * c->hidden); + load_tq(m, "lm_head.weight", c->hidden, c->vocab, quantize_dense, "lmhead", &m->lm_head); + if (m->lm_head.q) { qcount++; qfreed += (double)c->hidden * c->vocab * sizeof(float); } m->final_norm = load_t_n(m, "model.norm.weight", c->hidden); } m->L = calloc((size_t)c->n_layers, sizeof(Layer)); @@ -1374,15 +1562,19 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, m->active_of = malloc((size_t)c->n_layers * sizeof(int)); for (int i = 0; i < c->n_layers; i++) m->active_of[i] = i; char nm[256]; + int q_out = c->q_heads * c->q_head_dim, kv_out = c->kv_heads * c->k_head_dim; + #define QCOUNT(field) do { if ((field).q) { qcount++; qfreed += (double)(field).I * (field).O * sizeof(float); } } while (0) for (int i = layer_begin; i < layer_end; i++) { int ai = m->active_of[i]; /* == i for Phase 2 */ Layer *l = &m->L[i]; - /* input/post layernorms + MoE exist for every layer */ + /* input/post layernorms exist for every layer */ #define LD(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,ai); l->field = load_t_n(m,nm,(want)) LD(in_ln, "input_layernorm.weight", c->hidden); LD(post_ln,"post_attention_layernorm.weight", c->hidden); - LD(gate, "mlp.gate.weight", (int64_t)c->n_experts * c->hidden); #undef LD + snprintf(nm,sizeof(nm),"model.layers.%d.mlp.gate.weight", ai); + load_tq(m, nm, c->hidden, c->n_experts, quantize_dense, "router", &l->gate); + QCOUNT(l->gate); /* q/k norms are per-head [head_dim]; only on attention layers, load if present */ if (c->has_qk_norm) { snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.q_norm.weight", ai); @@ -1394,42 +1586,51 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits, snprintf(nm,sizeof(nm),"model.layers.%d.mlp.gate.e_score_correction_bias", ai); if (st_has(&m->S, nm)) { l->gate_bias = falloc(c->n_experts); st_read_f32(&m->S, nm, l->gate_bias, 0); } else l->gate_bias = NULL; - /* shared expert (dense f32) */ - #define LD2(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert." suffix,ai); l->field = load_t_n(m,nm,(want)) - LD2(sh_g, "gate_proj.weight", (int64_t)c->shared_inter * c->hidden); - LD2(sh_u, "up_proj.weight", (int64_t)c->shared_inter * c->hidden); - LD2(sh_d, "down_proj.weight", (int64_t)c->hidden * c->shared_inter); - #undef LD2 + /* shared expert (dense, int8-during-load) */ + snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert.gate_proj.weight", ai); + load_tq(m, nm, c->hidden, c->shared_inter, quantize_dense, "shexp", &l->sh_g); QCOUNT(l->sh_g); + snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert.up_proj.weight", ai); + load_tq(m, nm, c->hidden, c->shared_inter, quantize_dense, "shexp", &l->sh_u); QCOUNT(l->sh_u); + snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert.down_proj.weight", ai); + load_tq(m, nm, c->shared_inter, c->hidden, quantize_dense, "shexp", &l->sh_d); QCOUNT(l->sh_d); /* shared_expert_gate: Linear(hidden -> 1), sigmoid-gated shared expert */ snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert_gate.weight", ai); l->sh_gate = st_has(&m->S, nm) ? load_t_n(m, nm, c->hidden) : NULL; if (c->is_attn[i]) { - /* Gated Attention (full_attention) layer */ - #define LD3(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d.self_attn." suffix,ai); l->field = load_t_n(m,nm,(want)) - LD3(q, "q_proj.weight", (int64_t)c->q_heads * c->q_head_dim * c->hidden); - LD3(k, "k_proj.weight", (int64_t)c->kv_heads * c->k_head_dim * c->hidden); - LD3(v, "v_proj.weight", (int64_t)c->kv_heads * c->v_head_dim * c->hidden); - LD3(o, "o_proj.weight", (int64_t)c->hidden * c->o_in); - #undef LD3 - l->dn_qkv=l->dn_z=l->dn_b=l->dn_a=l->dn_conv=NULL; - l->dn_dtbias=l->dn_alog=l->dn_norm=l->dn_out=NULL; + /* Gated Attention (full_attention) layer, dense projections int8-during-load */ + snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.q_proj.weight", ai); + load_tq(m, nm, c->hidden, q_out, quantize_dense, "attn", &l->q); QCOUNT(l->q); + snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.k_proj.weight", ai); + load_tq(m, nm, c->hidden, kv_out, quantize_dense, "attn", &l->k); QCOUNT(l->k); + snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.v_proj.weight", ai); + load_tq(m, nm, c->hidden, kv_out, quantize_dense, "attn", &l->v); QCOUNT(l->v); + snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.o_proj.weight", ai); + load_tq(m, nm, c->o_in, c->hidden, quantize_dense, "attn", &l->o); QCOUNT(l->o); + l->dn_qkv=l->dn_z=(QW){0}; l->dn_b=l->dn_a=l->dn_conv=NULL; + l->dn_dtbias=l->dn_alog=l->dn_norm=NULL; l->dn_out=(QW){0}; } else { /* Gated DeltaNet (linear_attention) layer */ - l->q=l->k=l->v=l->o=NULL; + l->q=l->k=l->v=l->o=(QW){0}; #define LD4(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn." suffix,ai); l->field = load_t_n(m,nm,(want)) int64_t vdim_tot = (int64_t)c->dn_vheads * c->dn_vdim; - LD4(dn_qkv, "in_proj_qkv.weight", (int64_t)c->dn_conv_dim * c->hidden); - LD4(dn_z, "in_proj_z.weight", vdim_tot * c->hidden); + snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn.in_proj_qkv.weight", ai); + load_tq(m, nm, c->hidden, c->dn_conv_dim, quantize_dense, "dnproj", &l->dn_qkv); QCOUNT(l->dn_qkv); + snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn.in_proj_z.weight", ai); + load_tq(m, nm, c->hidden, (int)vdim_tot, quantize_dense, "dnproj", &l->dn_z); QCOUNT(l->dn_z); LD4(dn_b, "in_proj_b.weight", (int64_t)c->dn_vheads * c->hidden); LD4(dn_a, "in_proj_a.weight", (int64_t)c->dn_vheads * c->hidden); LD4(dn_conv,"conv1d.weight", (int64_t)c->dn_conv_dim * c->dn_convk); LD4(dn_dtbias, "dt_bias", c->dn_vheads); LD4(dn_alog,"A_log", c->dn_vheads); LD4(dn_norm, "norm.weight", c->dn_vdim); - LD4(dn_out, "out_proj.weight", (int64_t)c->hidden * vdim_tot); #undef LD4 + snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn.out_proj.weight", ai); + load_tq(m, nm, (int)vdim_tot, c->hidden, quantize_dense, "dnout", &l->dn_out); QCOUNT(l->dn_out); } } + #undef QCOUNT + if (quantize_dense) + fprintf(stderr, "[dense-i8] %d matrices quantized during load, %.1f GB f32 freed\n", qcount, qfreed/1073741824.0); m->cache = calloc((size_t)c->n_layers, sizeof(LCache)); for (int i = layer_begin; i < layer_end; i++) { m->cache[i].cap = cap; @@ -1474,7 +1675,8 @@ static void model_init(Model *m, const char *snap, int cap, int bits) { /* scale counts per expert matrix: per-row (gs=0) or grouped along input dim */ static int64_t scale_count_gu(const Cfg *c){ return c->expert_gs ? (int64_t)c->inter * ((c->hidden + c->expert_gs - 1) / c->expert_gs) : c->inter; } -static int64_t scale_count_d (const Cfg *c){ return c->expert_gs ? (int64_t)c->hidden * ((c->inter + c->expert_gs - 1) / c->expert_gs) : c->hidden; } +static int down_gs_of(const Cfg *c){ return c->expert_down_bits ? c->expert_down_gs : c->expert_gs; } +static int64_t scale_count_d (const Cfg *c){ int gs = down_gs_of(c); return gs ? (int64_t)c->hidden * ((c->inter + gs - 1) / gs) : c->hidden; } static void slot_ensure_allocated(Model *m, Slot *s) { if (s->g || s->pw) return; @@ -1572,9 +1774,10 @@ static void load_expert_merged(Model *m, int layer, int eid, Slot *s) { int64_t want_w = ng + ng + nd; int64_t want_s = 2*scale_count_gu(cc) + scale_count_d(cc); st_tensor *tw = st_find(&m->S, nm), *ts = st_find(&m->S, qsnm); - if (!tw || (tw->nbytes != want_w && tw->nbytes != want_w / 2)) { - fprintf(stderr, "%s: expert weight is %lld bytes — expected %lld (int8) or %lld (int4)\n", - nm, (long long)(tw ? tw->nbytes : -1), (long long)want_w, (long long)(want_w / 2)); exit(1); } + int64_t want_mixed = ng + nd; /* gate|up packed int4 (ng bytes) + down int8 (nd bytes) */ + if (!tw || (tw->nbytes != want_w && tw->nbytes != want_w / 2 && tw->nbytes != want_mixed)) { + fprintf(stderr, "%s: expert weight is %lld bytes — expected %lld (int8), %lld (int4) or %lld (int4 gate/up + int8 down)\n", + nm, (long long)(tw ? tw->nbytes : -1), (long long)want_w, (long long)(want_w / 2), (long long)want_mixed); exit(1); } if (!ts || ts->numel != want_s) { fprintf(stderr, "%s: scale array is %lld elems — expected %lld (refusing)\n", qsnm, (long long)(ts ? ts->numel : -1), (long long)want_s); exit(1); } @@ -1584,6 +1787,23 @@ static void load_expert_merged(Model *m, int layer, int eid, Slot *s) { rest of the MoE path (matmul_q) is unchanged. Nibble convention (must match c/tools/convert_qwen36.py pack_int4): LOW nibble = element 2k, HIGH nibble = 2k+1; each nibble is signed 4-bit (sign-extend if bit3 set). */ + if (tw->nbytes == want_mixed) { + /* mixed layout: unpack gate|up (2*ng int4 elements in ng bytes) into the + * slot's g|u block, copy down's int8 rows behind them. No packed copy is + * kept: the tier does not take this layout yet (main refuses it). */ + static int noted_m = 0; + if (!noted_m) { fprintf(stderr, "[qwen36] mixed expert layout detected (int4 gate/up, int8 down) — unpacking gate/up to int8 in slot\n"); noted_m = 1; } + uint8_t *raw = (uint8_t *)malloc((size_t)want_mixed); + if (!raw) { fprintf(stderr, "OOM reading mixed expert %s\n", nm); exit(1); } + st_read_raw(&m->S, nm, raw, 1); + unpack_int4_to_int8(s->g, raw, ng + ng); /* 2*ng elements from ng bytes */ + memcpy(s->d, raw + ng, (size_t)nd); + free(raw); + s->is_int4 = 0; + free(s->g4); free(s->u4); free(s->d4); s->g4 = s->u4 = s->d4 = NULL; + st_read_f32(&m->S, qsnm, s->gs, 0); + return; + } if (tw->nbytes == want_w / 2) { static int noted = 0; if (!noted) { fprintf(stderr, "[qwen36] int4 packed weights detected — %s\n", s->pw ? "kept int4, repacked planar for expert_ffn.h" : "unpacking to int8 in slot"); noted = 1; } @@ -1656,19 +1876,15 @@ static void slot_ensure_int8(Model *m, Slot *s) { int64_t ng = (int64_t)c->inter * c->hidden, nd = (int64_t)c->hidden * c->inter; int8_t *w = malloc((size_t)(ng + ng + nd)); if (!w) { fprintf(stderr, "OOM slot_ensure_int8\n"); exit(1); } - const uint8_t *src4[3] = { s->g4, s->u4, s->d4 }; - int64_t lens[3] = { ng, ng, nd }; - int8_t *dst = w; - for (int t = 0; t < 3; t++) { - const uint8_t *p = src4[t]; - for (int64_t i = 0; i < lens[t]; i += 2) { - uint8_t b = p[i >> 1]; - int8_t lo = (int8_t)(b & 0xF); if (lo & 8) lo -= 16; - int8_t hi = (int8_t)((b >> 4) & 0xF); if (hi & 8) hi -= 16; - dst[i] = lo; dst[i + 1] = hi; - } - dst += lens[t]; - } + /* #1271's unpack_int4_to_int8 (branchless, AVX2/NEON/scalar -- see its + * definition above load_expert_merged, which already uses it for the + * container's int4 read) instead of this function's own separate scalar + * copy of the same nibble-unpack math. g4/u4/d4 are three independently + * malloc'd packed buffers here (unlike load_expert_merged's single + * contiguous `raw`), so one call per segment. */ + unpack_int4_to_int8(w, s->g4, ng); + unpack_int4_to_int8(w + ng, s->u4, ng); + unpack_int4_to_int8(w + ng + ng, s->d4, nd); s->g = w; s->u = w + ng; s->d = w + ng + ng; } @@ -1879,9 +2095,11 @@ static void attention(Model *m, Layer *l, int layer, float *x, int S, int pos_ba float *q = falloc((int64_t)S*q_out); float *k = falloc((int64_t)S*kv_out); float *vv= falloc((int64_t)S*kv_out); - matmul_d(q, x, l->q, S, D, q_out); - matmul_d(k, x, l->k, S, D, kv_out); - matmul_d(vv, x, l->v, S, D, kv_out); + /* The projections the tier placed answer from VRAM for the whole batch, + * with one backend call per matrix; unavailable handles use CPU matmul. */ + if (!qtd_batch(l->qth_q, q, x, S, D, q_out)) matmul_d(q, x, &l->q, S, D, q_out); + if (!qtd_batch(l->qth_k, k, x, S, D, kv_out)) matmul_d(k, x, &l->k, S, D, kv_out); + if (!qtd_batch(l->qth_v, vv, x, S, D, kv_out)) matmul_d(vv, x, &l->v, S, D, kv_out); /* split q into query (first hd) and gate (next gate_dim), both per head */ float *query = falloc((int64_t)S*H*hd); float *gate = falloc((int64_t)S*H*gate_dim); @@ -1943,7 +2161,7 @@ static void attention(Model *m, Layer *l, int layer, float *x, int S, int pos_ba float g = gate_dim ? gate[o] : 0.f; ag[o] = ctx[o] * (1.f / (1.f + expf(-g))); } - matmul_d(out, ag, l->o, S, H*hd, D); + if (!qtd_batch(l->qth_o, out, ag, S, H*hd, D)) matmul_d(out, ag, &l->o, S, H*hd, D); free(q); free(k); free(vv); free(query); free(gate); free(ctx); free(ag); } @@ -1976,10 +2194,10 @@ static void qwen_shared_experts_cpu(Model *m, Layer *l, const float *x, int S, if (B == 1) { for (int s=0;ssh_g,1,D,I); - matmul_d(u,xs,l->sh_u,1,D,I); + if(!qtd(l->qth_shg,g,xs,D,I)) matmul_d(g,xs,&l->sh_g,1,D,I); + if(!qtd(l->qth_shu,u,xs,D,I)) matmul_d(u,xs,&l->sh_u,1,D,I); for(int i=0;ish_d,1,I,D); + if(!qtd(l->qth_shd,hh,g,I,D)) matmul_d(hh,g,&l->sh_d,1,I,D); float sgate=1.f; if(l->sh_gate){float sg=0.f;for(int i=0;ish_gate[i];sgate=1.f/(1.f+expf(-sg));} float *os=out+(int64_t)s*D; @@ -1990,10 +2208,10 @@ static void qwen_shared_experts_cpu(Model *m, Layer *l, const float *x, int S, float *bh=falloc((int64_t)B*D); for(int base=0;basesh_g,rows,D,I); - matmul_d(bu,x+(int64_t)base*D,l->sh_u,rows,D,I); + matmul_d(bg,x+(int64_t)base*D,&l->sh_g,rows,D,I); + matmul_d(bu,x+(int64_t)base*D,&l->sh_u,rows,D,I); for(int64_t q=0;q<(int64_t)rows*I;q++){float sv=bg[q];bg[q]=(sv/(1.f+expf(-sv)))*bu[q];} - matmul_d(bh,bg,l->sh_d,rows,I,D); + matmul_d(bh,bg,&l->sh_d,rows,I,D); for(int s=0;s ROUTE_RANK_MAX) K = ROUTE_RANK_MAX; + if (J < 0) J = 0; if (J > K) J = K; + int cap = (P > 0.f && P < 1.f) ? (M > 4*K ? M : 4*K) : (M > K ? M : K); + if (cap > E) cap = E; + if (cap > ROUTE_RANK_MAX) cap = ROUTE_RANK_MAX; + int rank[ROUTE_RANK_MAX]; float rw[ROUTE_RANK_MAX]; int8_t rl[ROUTE_RANK_MAX]; + int n = 0; + for (int r = 0; r < cap; r++) { + int best = -1; float bv = -1e30f; + for (int e = 0; e < E; e++) { + if (keep && !keep[e]) continue; + int taken = 0; for (int j = 0; j < n; j++) if (rank[j] == e) { taken = 1; break; } + if (!taken && pr[e] > bv) { bv = pr[e]; best = e; } + } + if (best < 0) break; + rank[n] = best; rw[n] = bv; n++; + } + int Kt = K < n ? K : n; /* the true top-K is rank[0..Kt) */ + int win = n; + if (P > 0.f && P < 1.f) { /* cumulative-mass window: grow past K until P of the ranked mass */ + float tot = 1e-20f; for (int r = 0; r < n; r++) tot += rw[r] > 0 ? rw[r] : 0; + float cum = 0; win = Kt; + for (int r = 0; r < n; r++) { cum += rw[r] > 0 ? rw[r] : 0; win = r + 1; if (cum >= P * tot) break; } + if (win < Kt) win = Kt; + } + for (int r = 0; r < n; r++) rl[r] = (r >= J && r < win) ? (int8_t)lvl(ctx, rank[r]) : 0; + int chosen = 0, pos[ROUTE_RANK_MAX]; uint8_t used[ROUTE_RANK_MAX] = {0}; + for (int r = 0; r < J && r < n && chosen < K; r++) { pos[chosen++] = r; used[r] = 1; } + for (int level = 2; level >= 1; level--) + for (int r = J; r < win && chosen < K; r++) + if (!used[r] && rl[r] == level) { pos[chosen++] = r; used[r] = 1; } + for (int r = 0; r < n && chosen < K; r++) + if (!used[r]) { pos[chosen++] = r; used[r] = 1; } + for (int kk = 0; kk < chosen; kk++) { + int r = pos[kk]; idx[kk] = rank[r]; val[kk] = rw[r]; + if (r >= Kt && alpha > 0.f && alpha < 1.f) val[kk] *= alpha; + } + for (int kk = chosen; kk < K; kk++) { idx[kk] = -1; val[kk] = 0.f; } /* fewer eligible than K: keep mask */ + if (!st) return; + st->slots += (uint64_t)chosen; st->agree_tot += (uint64_t)chosen; + float tsum = 1e-20f, csum = 1e-20f; + for (int t = 0; t < Kt; t++) tsum += rw[t] > 0 ? rw[t] : 0; + for (int kk = 0; kk < chosen; kk++) { + csum += val[kk] > 0 ? val[kk] : 0; + if (pos[kk] < Kt) st->agree_hit++; + else { st->swaps++; if (rl[pos[kk]] == 2) st->swaps_vram++; } + } + double kl = 0; /* KL(true top-K mass || chosen mass), as the GLM meter */ + for (int t = 0; t < Kt; t++) { + double pt = (rw[t] > 0 ? rw[t] : 0) / tsum; if (pt <= 0) continue; + double pc = 1e-12; + for (int kk = 0; kk < chosen; kk++) if (pos[kk] == t) { pc = (val[kk] > 0 ? val[kk] : 0) / csum; break; } + kl += pt * log(pt / pc); + } + st->kl_sum += kl; st->kl_n++; +} + +/* Residency levels for route_select: 2 = in the VRAM tier, 1 = in this + * layer's RAM cache (pinned or LRU), 0 = would be read from disk. */ +typedef struct { Model *m; int layer; } RouteCtx; +static int route_level(void *vctx, int e) { + RouteCtx *rc = (RouteCtx *)vctx; + if (qt_is_resident(rc->layer, e)) return 2; + pthread_mutex_lock(&g_pilot_mx); + Slot *s = slot_indexed(rc->m, rc->layer, e); + pthread_mutex_unlock(&g_pilot_mx); + return s ? 1 : 0; +} + +static void route_footer(FILE *f, const Model *m) { + if (g_cache_route && m->route.slots) + fprintf(f, "CACHE_ROUTE J=%d M=%d P=%.2f alpha=%.2f | swap %.1f%% (%llu/%llu, %llu to VRAM)\n", + g_route_j, g_route_m, g_route_p, g_route_alpha, + 100.0*m->route.swaps/m->route.slots, (unsigned long long)m->route.swaps, + (unsigned long long)m->route.slots, (unsigned long long)m->route.swaps_vram); + if (m->route.agree_tot) + fprintf(f, "route_agree %.1f%% | route_kl %.4f\n", 100.0*m->route.agree_hit/m->route.agree_tot, + m->route.kl_n ? m->route.kl_sum/(double)m->route.kl_n : 0.0); +} + static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { Cfg *c = &m->c; int D = c->hidden, E = c->n_experts, K = c->topk, I = c->inter; float *logits = falloc((int64_t)S*E); double _tr = tm_now(); - matmul_d(logits, x, l->gate, S, D, E); + matmul_d(logits, x, &l->gate, S, D, E); tm_add(S, 4, tm_now()-_tr); if (c->has_bias && l->gate_bias) { for (int s = 0; s < S; s++) { float *pr = logits + (int64_t)s*E; for (int e = 0; e < E; e++) pr[e] += l->gate_bias[e]; } @@ -2102,14 +2417,23 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { for (int e = 0; e < Ec; e++) keep[e] = 1; } int idx[256]; float val[256]; - for (int kk = 0; kk < K; kk++) { - int best = -1; float bv = -1e30f; - for (int e = 0; e < E; e++) { - if (!keep[e]) continue; - int taken = 0; for (int j = 0; j < kk; j++) if (idx[j]==e){taken=1;break;} - if (!taken && pr[e] > bv) { bv = pr[e]; best = e; } + if (g_cache_route) { + RouteCtx rc = { m, layer }; + route_select(pr, keep, E, K, g_route_j, g_route_m, g_route_p, g_route_alpha, + route_level, &rc, idx, val, &m->route); + } else { + for (int kk = 0; kk < K; kk++) { + int best = -1; float bv = -1e30f; + for (int e = 0; e < E; e++) { + if (!keep[e]) continue; + int taken = 0; for (int j = 0; j < kk; j++) if (idx[j]==e){taken=1;break;} + if (!taken && pr[e] > bv) { bv = pr[e]; best = e; } + } + idx[kk] = best; val[kk] = bv; + } + if (g_route_agree) { /* plain routing: full agreement by construction */ + m->route.agree_hit += (uint64_t)K; m->route.agree_tot += (uint64_t)K; m->route.kl_n++; } - idx[kk] = best; val[kk] = bv; } if (m->resident_collecting) { for (int kk = 0; kk < K; kk++) if (idx[kk] >= 0) m->seen[(int64_t)layer * E + idx[kk]] = 1; @@ -2146,7 +2470,7 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { matmul_qe(g, xs, e->g, e->gs, D, I); matmul_qe(u, xs, e->u, e->us, D, I); for (int i = 0; i < I; i++) { float gv = g[i]; g[i] = (gv / (1.f + expf(-gv))) * u[i]; } - matmul_qe(hh, g, e->d, e->ds, I, D); + matmul_qd(hh, g, e->d, e->ds, I, D); float w = val[kk]; float *os = out + (int64_t)s*D; for (int d = 0; d < D; d++) os[d] += w * hh[d]; } @@ -2155,10 +2479,10 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { { double _ts2 = tm_now(); int Ish = c->shared_inter; - matmul_d(sh, xs, l->sh_g, 1, D, Ish); - matmul_d(shu, xs, l->sh_u, 1, D, Ish); + if (!qtd(l->qth_shg, sh, xs, D, Ish)) matmul_d(sh, xs, &l->sh_g, 1, D, Ish); + if (!qtd(l->qth_shu, shu, xs, D, Ish)) matmul_d(shu, xs, &l->sh_u, 1, D, Ish); for (int i = 0; i < Ish; i++) { float sv = sh[i]; sh[i] = (sv / (1.f + expf(-sv))) * shu[i]; } - matmul_d(shd, sh, l->sh_d, 1, Ish, D); + if (!qtd(l->qth_shd, shd, sh, Ish, D)) matmul_d(shd, sh, &l->sh_d, 1, Ish, D); float sgate = 1.f; if (l->sh_gate) { float sg = 0.f; const float *wg = l->sh_gate; @@ -2170,7 +2494,10 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { tm_add(S, 3, tm_now()-_ts2); } double _q2 = tm_now(); - qt_take(qmask, val, K, out + (int64_t)s*D); + if(!qt_take(qmask, val, K, out + (int64_t)s*D)){ + fprintf(stderr,"qwen36: CUDA expert collection failed at layer %d; stopping inference\n",layer); + exit(1); + } if (tm_on() && S==1) { extern double g_qt_iss, g_qt_cpu, g_qt_tak; g_qt_iss += _q1-_q0; g_qt_cpu += _q2-_q1; g_qt_tak += tm_now()-_q2; @@ -2182,7 +2509,7 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { matmul_qe(g, xs, e->g, e->gs, D, I); matmul_qe(u, xs, e->u, e->us, D, I); for (int i = 0; i < I; i++) { float gv = g[i]; g[i] = (gv / (1.f + expf(-gv))) * u[i]; } - matmul_qe(hh, g, e->d, e->ds, I, D); + matmul_qd(hh, g, e->d, e->ds, I, D); float w = val[kk]; float *os = out + (int64_t)s*D; for (int d = 0; d < D; d++) os[d] += w * hh[d]; @@ -2209,6 +2536,14 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) { * split conv_out -> q_in/k_in/v_in; repeat_interleave q,k by rep; l2norm * (q scaled by 1/sqrt(kdim)); recurrence S[h]*=exp(g); kv=k@S; delta=(v-kv)*beta; * S+=k (x) delta; out=q@S; per-head Gated RMSNorm (plain weight) -> out_proj. */ +/* Bound both the host result block and device input/output staging. */ +static int dnproj_batch_rows(int S, int H, int O) { + int64_t rows = (32LL << 20) / (((int64_t)H + O) * sizeof(float)); + if (rows < 1) rows = 1; + if (rows > 256) rows = 256; + return S < rows ? S : (int)rows; +} + static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_base, float *out) { (void)pos_base; Cfg *c = &m->c; @@ -2220,12 +2555,12 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas float scale = 1.f / sqrtf((float)kdim); int H = c->hidden; - /* qkv and z live in ONE buffer: the fused GPU projection writes - * [conv_dim ++ value_dim] in a single GEMV, and the CPU fallback fills the - * same two regions. Either way the code below reads qkv/z unchanged. */ - float *qkvz = falloc((int64_t)conv_dim + value_dim); - float *qkv = qkvz; - float *z = qkvz + conv_dim; + /* Input projections have no recurrent dependency. Keep qkv ++ z for a + * bounded block, then consume rows in order through conv and recurrence. */ + int proj_dim = conv_dim + value_dim; + int B = qt_dnproj_ready(layer) ? dnproj_batch_rows(S, H, proj_dim) : 1; + float *qkvz = falloc((int64_t)B * proj_dim); + int gpu_block = 0; float *b = falloc(vh); float *a = falloc(vh); float *beta= falloc(vh); @@ -2245,11 +2580,15 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas const float *xs = x + (int64_t)s * H; extern double g_dn_sub[4]; double _d0 = tm_now(); - /* projections (single-token matmuls). One fused GEMV when this layer's - * dnproj is placed on a GPU, the two CPU matmuls otherwise. */ - if (!qt_dnproj_matmul(layer, qkvz, xs, H, conv_dim + value_dim)) { - matmul_d(qkv, xs, l->dn_qkv, 1, H, conv_dim); - matmul_d(z, xs, l->dn_z, 1, H, value_dim); + if (s % B == 0) { + int rows = S - s < B ? S - s : B; + gpu_block = qt_dnproj_matmul_batch(layer, qkvz, xs, rows, H, proj_dim); + } + float *qkv = qkvz + (int64_t)(s % B) * proj_dim; + float *z = qkv + conv_dim; + if (!gpu_block) { + matmul_d(qkv, xs, &l->dn_qkv, 1, H, conv_dim); + matmul_d(z, xs, &l->dn_z, 1, H, value_dim); } matmul(b, xs, l->dn_b, 1, H, vh); matmul(a, xs, l->dn_a, 1, H, vh); @@ -2347,7 +2686,8 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas outr[(int64_t)h * vdim + d] = val * zr[d] / (1.f + expf(-zr[d])); } } - matmul_d(out + (int64_t)s * H, outr, l->dn_out, 1, value_dim, H); + if (!qtd(l->qth_dnout, out + (int64_t)s * H, outr, value_dim, H)) + matmul_d(out + (int64_t)s * H, outr, &l->dn_out, 1, value_dim, H); if (tm_on() && S==1){ g_dn_sub[3]+=tm_now()-_d0; } if (layer == 0 && s == 0 && getenv("DN_DBG")) { FILE *dbg = fopen(getenv("DN_DBG"), "wb"); @@ -2371,6 +2711,117 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas free(conv_out); free(q); free(k); free(outv); free(outr); free(kv); free(delta); } +/* The rest of the dense trunk, offered to the placer by name and layer with + * the bytes of the dense-i8 copies (docs/qwen36-cuda-tier.md, "Placement"): + * dnout -- the DeltaNet out_proj, one matrix per DeltaNet layer + * attnproj -- q, k, v, o of every attention layer, offered as one item + * shexp -- gate, up, down of the shared expert, every layer + * Measured on ds (CPU, 8 threads): of the 37.5 ms a DeltaNet layer stack + * costs per decoded token, 23.4 are the input projections (already placeable + * as "dnproj"), 8.3 the out_proj and norm, 3.3 the convolution and 2.4 the + * recurrence -- the matmuls are the cost, not the recurrence, so this is + * where the trunk goes. Offer order after lmhead and dnproj: dnout, attnproj, + * shexp, each in layer order, so a partial placement is whole layers. A + * component is offered only when every matrix of it has a dense-i8 copy. */ +static void trunk_offer_dense(Model *m){ + Cfg *c = &m->c; + for (int i = 0; i < c->n_layers; i++) { + if (c->is_attn[i]) continue; + size_t b = qdw_bytes(&m->L[i].dn_out); + if (b) qt_trunk_offer("dnout", i, b); + } + for (int i = 0; i < c->n_layers; i++) { + if (!c->is_attn[i]) continue; + Layer *l = &m->L[i]; + size_t bq = qdw_bytes(&l->q), bk = qdw_bytes(&l->k), bv = qdw_bytes(&l->v), bo = qdw_bytes(&l->o); + if (bq && bk && bv && bo) qt_trunk_offer("attnproj", i, bq + bk + bv + bo); + } + for (int i = 0; i < c->n_layers; i++) { + Layer *l = &m->L[i]; + size_t bg = qdw_bytes(&l->sh_g), bu = qdw_bytes(&l->sh_u), bd = qdw_bytes(&l->sh_d); + if (bg && bu && bd) qt_trunk_offer("shexp", i, bg + bu + bd); + } +} +/* After qt_init decided: upload what was placed, keep the handles in the + * Layer. Every matrix falls back on its own, so a failed upload costs one + * GEMV on the CPU, never the component. Returns the number of matrices + * placed; `vram_bytes` gets their size. */ +static int trunk_place_dense(Model *m, double *vram_bytes){ + Cfg *c = &m->c; int placed = 0; double vram = 0; + for (int i = 0; i < c->n_layers; i++) { + Layer *l = &m->L[i]; + if (!c->is_attn[i]) { + l->qth_dnout = qdw_place(&l->dn_out, qt_place_of("dnout", i)); + if (l->qth_dnout) { placed++; vram += (double)qdw_bytes(&l->dn_out); } + } else { + int dev = qt_place_of("attnproj", i); + l->qth_q = qdw_place(&l->q, dev); l->qth_k = qdw_place(&l->k, dev); + l->qth_v = qdw_place(&l->v, dev); l->qth_o = qdw_place(&l->o, dev); + if (l->qth_q) { placed++; vram += (double)qdw_bytes(&l->q); } + if (l->qth_k) { placed++; vram += (double)qdw_bytes(&l->k); } + if (l->qth_v) { placed++; vram += (double)qdw_bytes(&l->v); } + if (l->qth_o) { placed++; vram += (double)qdw_bytes(&l->o); } + } + int dev = qt_place_of("shexp", i); + l->qth_shg = qdw_place(&l->sh_g, dev); l->qth_shu = qdw_place(&l->sh_u, dev); l->qth_shd = qdw_place(&l->sh_d, dev); + if (l->qth_shg) { placed++; vram += (double)qdw_bytes(&l->sh_g); } + if (l->qth_shu) { placed++; vram += (double)qdw_bytes(&l->sh_u); } + if (l->qth_shd) { placed++; vram += (double)qdw_bytes(&l->sh_d); } + } + if (vram_bytes) *vram_bytes = vram; + return placed; +} + +/* Measured, not assumed. The placer prices a trunk component by the bytes it + * saves on the CPU's memory bus, which presumes the GPU answers a GEMV faster + * than the CPU does. Four Tesla M10 (sm_50) said otherwise: every placed + * component ran slower there, lm_head 68.8 ms against 41.7 on the CPU, and + * decode fell from 3.56 to 2.68 tok/s (#1652). So before any trunk upload, + * one DeltaNet input projection (the most numerous placed matrix) is timed + * both ways on the device that would host it, ten GEMVs each, best of three + * rounds after a warm-up, and the trunk goes to VRAM only if the GPU wins. + * Only the automatic placement is questioned: a hand-written COLI_PLACE + * stands. COLI_TRUNK_PROBE=0 skips the probe and trusts the placer. The + * probe's copy stays resident (one projection, ~25 MB on the 35B). */ +static int trunk_probe_gpu_wins(Model *m){ + const char *e = getenv("COLI_TRUNK_PROBE"); + if (e && *e == '0') return 1; + if (!qt_place_is_auto()) return 1; + Cfg *c = &m->c; + QW *w = NULL; int dev = QT_PLACE_CPU; + for (int i = 0; i < c->n_layers && !w; i++) { + if (c->is_attn[i]) continue; + if (m->L[i].dn_qkv.q) { w = &m->L[i].dn_qkv; dev = qt_place_of("dnproj", i); } + } + if (!w) return 1; /* dense-i8 off: nothing will be placed */ + if (dev == QT_PLACE_CPU) dev = qt_place_of("lmhead", 0); + if (dev == QT_PLACE_CPU) return 1; /* nothing placed: nothing to measure */ + int I = w->I, O = w->O; + int h = qt_dense_init(w->q, w->sc, I, O, dev); + if (h < 0) return 1; /* cannot measure: the placer's word stands */ + float *x = malloc((size_t)I * sizeof(float)), *y = malloc((size_t)O * sizeof(float)); + if (!x || !y) { free(x); free(y); return 1; } + for (int i = 0; i < I; i++) x[i] = sinf(0.37f * (float)i); + double gpu = 1e30, cpu = 1e30; + for (int r = 0; r < 3; r++) { + for (int k = 0; k < 3; k++) if (!qt_dense_matmul(h, y, x, I, O)) { free(x); free(y); return 1; } + double t0 = now_s(); + for (int k = 0; k < 10; k++) if (!qt_dense_matmul(h, y, x, I, O)) { free(x); free(y); return 1; } + double tg = (now_s() - t0) / 10; + for (int k = 0; k < 3; k++) matmul_q(y, x, w->q, w->sc, I, O); + t0 = now_s(); + for (int k = 0; k < 10; k++) matmul_q(y, x, w->q, w->sc, I, O); + double tc = (now_s() - t0) / 10; + if (tg < gpu) gpu = tg; + if (tc < cpu) cpu = tc; + } + free(x); free(y); + int wins = gpu < cpu; + fprintf(stderr, "[place] probe: one [%d x %d] int8 GEMV takes %.3f ms on CUDA dev %d, %.3f ms on the CPU -> trunk %s\n", + O, I, gpu * 1e3, dev, cpu * 1e3, wins ? "to VRAM" : "stays on the CPU"); + return wins; +} + static void layers_forward_range(Model *m, float *x, int S, int pos_base, int layer_begin, int layer_end, int allow_prefetch, FILE *lf) { @@ -2438,10 +2889,13 @@ static int g_pin_use_logit = 0; /* 1 quando questa richiesta e ripartita da typedef struct { float **rec, **conv; int n_layers; } Q36PinState; /* Stato della lettura del prefill: dichiarato qui perche step() lo consulta e - * step() viene prima del codice di servizio che lo accende. */ + * step() viene prima del codice di servizio che lo accende. Solo il servizio + * lo accende e solo il servizio definisce serve_echo: senza main non esiste. */ +#ifndef QWEN36_NO_MAIN static int g_echo_k = 0; /* 0 = spento */ static const char *g_echo_id = NULL; static void serve_echo(const char *id, int pos, int token, const float *lo, int V, int k); +#endif static float *step(Model *m, const int *ids, int S, int pos_base) { Cfg *c = &m->c; int D = c->hidden; @@ -2482,6 +2936,7 @@ static float *step(Model *m, const int *ids, int S, int pos_base) { * token, il cui predittore sta nello stato precedente. Per questo il * chiamante arretra di uno il riuso del prefisso quando la lettura e * accesa: cosi il primo token dell'opzione ricade sempre qui dentro. */ +#ifndef QWEN36_NO_MAIN if (g_echo_k > 0 && g_echo_id && S > 0) { float *erow = falloc(D), *elog = falloc(c->vocab); /* Il primo token fresco e predetto dallo stato PRECEDENTE, che dopo un @@ -2492,17 +2947,18 @@ static float *step(Model *m, const int *ids, int S, int pos_base) { for (int p = 0; p + 1 < S; p++) { rmsnorm_row(erow, x + (int64_t)p*D, m->final_norm, D, c->eps); if (!qt_lmhead_matmul(elog, erow, D, c->vocab)) - matmul_d(elog, erow, m->lm_head, 1, D, c->vocab); + matmul_d(elog, erow, &m->lm_head, 1, D, c->vocab); serve_echo(g_echo_id, pos_base + p + 1, ids[p+1], elog, c->vocab, g_echo_k); } free(erow); free(elog); } +#endif float *last = falloc(D); rmsnorm_row(last, x + (int64_t)(S-1)*D, m->final_norm, D, c->eps); float *logit = falloc(c->vocab); double _th = tm_now(); if (!qt_lmhead_matmul(logit, last, D, c->vocab)) - matmul_d(logit, last, m->lm_head, 1, D, c->vocab); + matmul_d(logit, last, &m->lm_head, 1, D, c->vocab); if (tm_on()) { tm_add(S, 5, tm_now()-_th); if (S==1) g_tm_dec_tokens++; else g_tm_pre_tokens += S; } free(x); free(last); if (lf) fclose(lf); @@ -2563,7 +3019,7 @@ static void pilot_prefetch(Model *m, int lnext, const float *x, int S) { Layer *l = &m->L[lnext]; float *nrm_x = falloc((int64_t)S * D); for (int s = 0; s < S; s++) rmsnorm_row(nrm_x + (int64_t)s*D, x + (int64_t)s*D, l->post_ln, D, c->eps); - matmul_d(logits, nrm_x, l->gate, S, D, E); /* int8 copy (f32 may be freed) */ + matmul_d(logits, nrm_x, &l->gate, S, D, E); /* int8 copy (f32 may be freed) */ free(nrm_x); for (int s = 0; s < S; s++) { float *pr = logits + (int64_t)s*E; @@ -3092,14 +3548,36 @@ static void hits_emit(Model *m){ } static double tm_sum(int idx){ return (g_tm_dec[idx]+g_tm_pre[idx])/1e3; } /* ms -> s */ +/* The generation budget a request gets. max_tokens is a CEILING, not a + * target (#260/#382, the rule GLM and DeepSeek V4 already apply): the prompt + * must fit with room for one token (none for a read-only logprobs request, + * docs/brio.md), and the budget is then clamped to what the context can hold. + * Returns the budget, or -1 when the PROMPT does not fit. Refusing when + * prompt + budget exceeded the context (#1641) turned the gateway's default + * output budget -- 8192 here, the whole default context -- into a 400 on + * every message of `coli chat` and on every request without max_tokens. */ +static int qwen36_serve_budget(int np, int max_tok, int max_ctx, int read_only){ + if (np < 1) return -1; + int room = max_ctx - np; + if (read_only) return room < 0 ? -1 : (max_tok < room ? max_tok : room); + if (room < 1) return -1; + return max_tok > room ? room : max_tok; +} + static void serve_one(Model *m, ServeReq *q){ int *ids=NULL, np=0; encode_text(q->payload, &ids, &np); /* payload is raw prompt text; qwen36 adds no BOS */ int max_ctx = qwen36_max_ctx(); - if(np<1 || np+q->max_tok>max_ctx){ + int budget = qwen36_serve_budget(np, q->max_tok, max_ctx, q->logprobs > 0); + if(budget < 0){ printf("ERROR %s CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d\n",q->id,np,q->max_tok,max_ctx); fflush(stdout); free(ids); return; } + if(budget < q->max_tok){ + fprintf(stderr,"[serve] max_tokens %d clamped to %d (context %d - prompt %d); raise Q36_MAXT for longer answers\n", + q->max_tok, budget, max_ctx, np); + q->max_tok = budget; + } printf("ACCEPT %s %d\n",q->id,np); fflush(stdout); m->max_t = np + q->max_tok; /* Grow the cache BEFORE deciding, so the decision sees the state that will @@ -3317,6 +3795,15 @@ int main(int argc, char **argv) { g_pilot = getenv("PILOT") ? atoi(getenv("PILOT")) : 0; g_wide = getenv("WIDE") ? atoi(getenv("WIDE")) : 1; if (g_wide < 1) g_wide = 1; if (g_wide > 4) g_wide = 4; + g_cache_route = getenv("CACHE_ROUTE") ? atoi(getenv("CACHE_ROUTE")) : 0; /* prefer resident experts (VRAM tier, then RAM cache) inside the top-M window; changes which experts run: docs/CACHE_ROUTE.md */ + g_route_j = getenv("ROUTE_J") ? atoi(getenv("ROUTE_J")) : 2; /* true top-J always taken, resident or not, under CACHE_ROUTE=1 */ + g_route_m = getenv("ROUTE_M") ? atoi(getenv("ROUTE_M")) : 12; /* rank window inside which a resident expert may replace an unresident one */ + g_route_p = getenv("ROUTE_P") ? (float)atof(getenv("ROUTE_P")) : 0.f; /* cumulative-mass window for CACHE_ROUTE (0 = fixed M) */ + g_route_alpha = getenv("ROUTE_ALPHA") ? (float)atof(getenv("ROUTE_ALPHA")) : 1.f; /* scale substituted experts' gate mass before renorm (1 = off) */ + g_route_agree = getenv("ROUTE_AGREE") ? atoi(getenv("ROUTE_AGREE")) : g_cache_route; /* overlap% + KL vs the true top-K in the footer; auto-on under CACHE_ROUTE=1 */ + if (g_cache_route) + fprintf(stderr, "[qwen36] CACHE_ROUTE=1 J=%d M=%d P=%.2f alpha=%.2f: VRAM tier > RAM cache > disk inside the top-M window (lossy: A/B it)\n", + g_route_j, g_route_m, g_route_p, g_route_alpha); if (getenv("OPENAI")) g_openai = 1; /* OpenAI-compatible output */ const char *mv = getenv("MODEL"); if (mv && *mv) g_model = mv; int hot_n = getenv("HOT") ? atoi(getenv("HOT")) : 0; @@ -3394,36 +3881,10 @@ int main(int argc, char **argv) { g_expert_gs = m.c.expert_gs; if (g_expert_gs) fprintf(stderr, "[qwen36] group-scaled experts: gs=%d\n", g_expert_gs); fprintf(stderr, "resident weights loaded in %.1fs | RSS after load: %.2f GB\n", m.dense_load_s, rss_gb()); - /* quantize the large dense matrices to int8 (COLI_DENSE_I8=0 disables) */ - if (dense_i8_on()) { - double tq = now_s(); - Cfg *qc = &m.c; int D2 = qc->hidden; - int q_out = qc->q_heads * qc->q_head_dim, kv_out = qc->kv_heads * qc->k_head_dim; - for (int i = 0; i < qc->n_layers; i++) { - Layer *l = &m.L[i]; - qdw_register(l->q, D2, q_out); qdw_register(l->k, D2, kv_out); - qdw_register(l->v, D2, kv_out); qdw_register(l->o, qc->o_in, D2); - qdw_register(l->gate, D2, qc->n_experts); - qdw_register(l->sh_g, D2, qc->shared_inter); qdw_register(l->sh_u, D2, qc->shared_inter); - qdw_register(l->sh_d, qc->shared_inter, D2); - qdw_register(l->dn_qkv, D2, qc->dn_conv_dim); - qdw_register(l->dn_z, D2, qc->dn_vheads * qc->dn_vdim); - qdw_register(l->dn_out, qc->dn_vheads * qc->dn_vdim, D2); - } - qdw_register(m.lm_head, D2, qc->vocab); - /* Free the f32 originals -- the pointers only serve as lookup keys in - * matmul_d from here on (never dereferenced again). - * COLI_KEEP_F32=1 keeps them (debug). */ - double freed = 0; - if (!getenv("COLI_KEEP_F32")) { - for (int i = 0; i < g_qdw_n; i++) { - freed += (double)g_qdw[i].I * g_qdw[i].O * sizeof(float); - free((void*)g_qdw[i].w); - } - } - fprintf(stderr, "[dense-i8] %d matrices quantized in %.1f s, %.1f GB f32 freed\n", - g_qdw_n, now_s()-tq, freed/1073741824.0); - } + /* dense matrices are quantized to int8 (+ int4 planar per COLI_DENSE_BITS/ + * COLI_DENSE_INT4, see dense_int4_wanted) during model_init above + * (COLI_DENSE_I8=0 disables it) -- see load_tq/QW; model_init_range + * already logged the count and freed bytes. */ /* Optional CUDA VRAM expert tier (COLI_CUDA=1): hot experts live in * DEVICE_LOCAL memory across the configured GPUs, misses fall back to the @@ -3434,7 +3895,7 @@ int main(int argc, char **argv) { * riservare qualunque budget: e' int4 impacchettato che va in VRAM come * fmt=4, int8 come fmt=1. Sbagliare qui era #1331 -- budget riservato, * planned=1, e zero promozioni per tutta la vita del processo. */ - int expert_is_int4 = 1; + int expert_is_int4 = 1, expert_mixed = 0; { char probe[256]; snprintf(probe, sizeof(probe), @@ -3442,41 +3903,46 @@ int main(int argc, char **argv) { st_tensor *pt = st_find(&m.S, probe); int64_t want = 2*(int64_t)m.c.inter*m.c.hidden + (int64_t)m.c.hidden*m.c.inter; if (pt && pt->nbytes == want) expert_is_int4 = 0; /* int8: un byte per elemento */ + /* mixed (convert_qwen36.py --down-bits): int4 gate/up + int8 down = 2/3 of int8 */ + if (pt && pt->nbytes == want * 2 / 3) { expert_is_int4 = 0; expert_mixed = 1; } } /* Una riga, sempre: e' l'unico modo di verificare il probe dall'esterno * (CI sul container tiny int8, #1331) senza una scheda. */ fprintf(stderr, "[qwen36] expert format on disk: %s\n", - expert_is_int4 ? "int4 packed (tier fmt=4)" : "int8 (tier fmt=1)"); + expert_mixed ? "int4 gate/up + int8 down (mixed; CPU path, no VRAM tier yet)" + : expert_is_int4 ? "int4 packed (tier fmt=4)" : "int8 (tier fmt=1)"); g_expert_is_int4 = expert_is_int4; + g_expert_mixed = expert_mixed; g_expert_down_gs = expert_mixed ? m.c.expert_down_gs : 0; + if (expert_mixed && m.c.expert_down_bits == 0) + fprintf(stderr, "[qwen36] mixed layout on disk but qwen36_meta.json has no expert_down_bits -- down scales assumed per row\n"); /* Offer the dense trunk to the placer before the tier decides its budget: * sizes only, from the same dense-i8 entries the uploads below will use. * No entry (dense-i8 off) means nothing to offer, and the CPU path stands. */ { int O_qkv = m.c.dn_conv_dim, O_z = m.c.dn_vheads * m.c.dn_vdim; - for (int i = 0; i < g_qdw_n; i++) - if (g_qdw[i].w == m.lm_head) - qt_trunk_offer("lmhead", 0, (size_t)g_qdw[i].I * g_qdw[i].O + (size_t)g_qdw[i].O * sizeof(float)); + if (m.lm_head.q) + qt_trunk_offer("lmhead", 0, (size_t)m.lm_head.I * m.lm_head.O + (size_t)m.lm_head.O * sizeof(float)); for (int i = 0; i < m.c.n_layers; i++) { if (m.c.is_attn[i]) continue; - int have = 0; - for (int j = 0; j < g_qdw_n; j++) - if (g_qdw[j].w == m.L[i].dn_qkv || g_qdw[j].w == m.L[i].dn_z) have++; - if (have == 2) + if (m.L[i].dn_qkv.q && m.L[i].dn_z.q) qt_trunk_offer("dnproj", i, (size_t)(O_qkv + O_z) * m.c.hidden + (size_t)(O_qkv + O_z) * sizeof(float)); } + trunk_offer_dense(&m); /* dnout, attnproj, shexp: the rest of the per-token dense work */ } - if (qt_init(m.c.n_layers, m.c.n_experts, m.c.hidden, m.c.inter, cap, m.c.topk, + if (expert_mixed && getenv("COLI_CUDA") && getenv("COLI_CUDA")[0] == '1') + fprintf(stderr, "[qwen36] COLI_CUDA=1 ignored: the VRAM expert tier does not take the mixed layout yet (one format per expert)\n"); + if (!expert_mixed && + qt_init(m.c.n_layers, m.c.n_experts, m.c.hidden, m.c.inter, cap, m.c.topk, m.c.expert_gs, expert_is_int4)) { fprintf(stderr, "[gpu] MoE experts -> CUDA VRAM tier\n"); atexit(qt_shutdown); + /* The placer predicted; measure before uploading a byte of trunk. */ + if (!trunk_probe_gpu_wins(&m)) qt_trunk_withdraw("measured slower than the CPU"); /* R4 role split: park the dense-i8 lm_head on COLI_LMHEAD_GPU. The - * qdw entry keyed by m.lm_head holds the int8 rows + per-row scales - * the CPU path uses; the GPU applies the identical semantics. */ - for (int i = 0; i < g_qdw_n; i++) - if (g_qdw[i].w == m.lm_head) { - qt_lmhead_init(g_qdw[i].q, g_qdw[i].sc, g_qdw[i].I, g_qdw[i].O); - break; - } + * QW struct on m.lm_head holds the int8 rows + per-row scales the + * CPU path uses; the GPU applies the identical semantics. */ + if (m.lm_head.q) + qt_lmhead_init(m.lm_head.q, m.lm_head.sc, m.lm_head.I, m.lm_head.O); /* R4 step 2: DeltaNet input projections, per layer, wherever * COLI_PLACE puts them. qkv and z are both [O_x, hidden] int8 with * per-row scales, so fusing them is a concatenation along O -- two @@ -3490,11 +3956,8 @@ int main(int argc, char **argv) { if (m.c.is_attn[i]) continue; int dev = qt_place_of("dnproj", i); if (dev == QT_PLACE_CPU) continue; - const int8_t *q1 = NULL, *q2 = NULL; const float *s1 = NULL, *s2 = NULL; - for (int j = 0; j < g_qdw_n; j++) { - if (g_qdw[j].w == m.L[i].dn_qkv) { q1 = g_qdw[j].q; s1 = g_qdw[j].sc; } - if (g_qdw[j].w == m.L[i].dn_z) { q2 = g_qdw[j].q; s2 = g_qdw[j].sc; } - } + const int8_t *q1 = m.L[i].dn_qkv.q, *q2 = m.L[i].dn_z.q; + const float *s1 = m.L[i].dn_qkv.sc, *s2 = m.L[i].dn_z.sc; if (!q1 || !q2) continue; /* dense-i8 off: CPU path stands */ int8_t *qf = malloc((size_t)Of * Hd); float *sf = malloc((size_t)Of * sizeof(float)); @@ -3512,6 +3975,15 @@ int main(int argc, char **argv) { fprintf(stderr, "[dnp] %d DeltaNet-Projektionen auf GPU (%.2f GB VRAM)\n", placed, vram / 1073741824.0); } + /* The rest of the trunk, wherever the placer put it: out_proj, + * attention projections, shared expert. Handles live in the Layer. */ + { + double vram = 0; + int placed = trunk_place_dense(&m, &vram); + if (placed) + fprintf(stderr, "[dense] %d trunk matrices on GPU (dnout/attnproj/shexp, %.2f GB VRAM)\n", + placed, vram / 1073741824.0); + } /* Warmstart: fill the VRAM budget BEFORE the first token (heat order * when HEAT_FILE exists, natural order otherwise), loading all RAM * slots along the way. */ @@ -3536,6 +4008,7 @@ int main(int argc, char **argv) { printf("TF-NLL: %.4f nats/token over %d tokens | ppl = %.2f\n", nll, scored, exp(nll)); printf("Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0, (unsigned long long)m.hits, (unsigned long long)m.miss); + route_footer(stdout, &m); printf("Speed: %.2f tok/s (%.1fs for %d tokens) | PEAK RSS: %.2f GB\n", scored/dt, dt, scored, rss_gb()); free(buf); free(arena); return 0; } @@ -3601,6 +4074,7 @@ int main(int argc, char **argv) { fprintf(stderr, "\nPEAK RSS: %.2f GB\n", rss_gb()); fprintf(stderr, "Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0, (unsigned long long)m.hits, (unsigned long long)m.miss); + route_footer(stderr, &m); fprintf(stderr, "Speed: %.2f tok/s (%.1fs for %d tokens)\n", n_new/dt, dt, n_new); free(buf); free(arena); /* Oracle mode is a gate, not a report: a mismatch must fail the caller. @@ -3629,12 +4103,12 @@ typedef struct { static void qwen36_segment_layer_free(Layer *layer) { free(layer->in_ln); free(layer->post_ln); - free(layer->q); free(layer->k); free(layer->v); free(layer->o); - free(layer->qn); free(layer->kn); free(layer->gate); free(layer->gate_bias); - free(layer->sh_g); free(layer->sh_u); free(layer->sh_d); free(layer->sh_gate); - free(layer->dn_qkv); free(layer->dn_z); free(layer->dn_b); free(layer->dn_a); + qw_free(&layer->q); qw_free(&layer->k); qw_free(&layer->v); qw_free(&layer->o); + free(layer->qn); free(layer->kn); qw_free(&layer->gate); free(layer->gate_bias); + qw_free(&layer->sh_g); qw_free(&layer->sh_u); qw_free(&layer->sh_d); free(layer->sh_gate); + qw_free(&layer->dn_qkv); qw_free(&layer->dn_z); free(layer->dn_b); free(layer->dn_a); free(layer->dn_conv); free(layer->dn_dtbias); free(layer->dn_alog); - free(layer->dn_norm); free(layer->dn_out); + free(layer->dn_norm); qw_free(&layer->dn_out); } static void qwen36_segment_model_destroy(Qwen36SegmentEngine *engine) { @@ -4022,7 +4496,7 @@ static void qwen36_edge_engine_destroy(void *engine_impl) { Qwen36EdgeEngine *engine = (Qwen36EdgeEngine *)engine_impl; if (!engine) return; free(engine->model.embed); - free(engine->model.lm_head); + qw_free(&engine->model.lm_head); free(engine->model.final_norm); free(engine->model.c.is_attn); st_destroy(&engine->model.S); @@ -4061,9 +4535,10 @@ static int qwen36_edge_engine_open( engine->model.embed = load_t_n( &engine->model, "model.embed_tokens.weight", (int64_t)config->vocab * config->hidden); - engine->model.lm_head = load_t_n( - &engine->model, "lm_head.weight", - (int64_t)config->vocab * config->hidden); + /* quantize=0: this engine never ran the old post-hoc qdw_register pass + * either (only main()'s static Model did), so lm_head stays f32-only here, + * exactly as before. */ + load_tq(&engine->model, "lm_head.weight", config->hidden, config->vocab, 0, "lmhead", &engine->model.lm_head); engine->model.final_norm = load_t_n( &engine->model, "model.norm.weight", config->hidden); engine->model.quant_bits = container_layer_is_int4(&engine->model, 0) ? 4 : 8; @@ -4211,7 +4686,7 @@ static int qwen36_edge_select(void *engine_impl, } rmsnorm_row(normalized, input + (size_t)row * config->hidden, engine->model.final_norm, config->hidden, config->eps); - matmul_d(logits, normalized, engine->model.lm_head, + matmul_d(logits, normalized, &engine->model.lm_head, 1, config->hidden, config->vocab); if (coli_edge_argmax(logits, (uint32_t)config->vocab, &request->token_ids[row], @@ -4242,7 +4717,7 @@ static int qwen36_edge_logits(void *engine_impl, rmsnorm_row(normalized, input + (size_t)row * config->hidden, engine->model.final_norm, config->hidden, config->eps); matmul_d(request->logits + (size_t)row * config->vocab, - normalized, engine->model.lm_head, + normalized, &engine->model.lm_head, 1, config->hidden, config->vocab); } free(normalized); diff --git a/c/qwen36_tier.c b/c/qwen36_tier.c index e06c2de6a..742769d9a 100644 --- a/c/qwen36_tier.c +++ b/c/qwen36_tier.c @@ -45,7 +45,7 @@ static struct { QSlot *slot; /* [nl*ne] */ pthread_mutex_t mx; pthread_t th; - int th_stop; + int th_stop, waiters; /* upload ring with staging copies */ struct { int layer, eid; uint8_t *w; float *s; int v_layer, v_eid; } q[QT_QCAP]; int qh, qt_, qn; @@ -68,6 +68,13 @@ static struct { uint32_t *heat0; /* heat table loaded from HEAT_FILE */ } G; +/* Count parked callers so shutdown can reclaim their shared storage safely. */ +static void wait_take_locked(void){ + G.waiters++; + pthread_cond_wait(&G.cv_take,&G.mx); + if(--G.waiters==0 && G.th_stop) pthread_cond_broadcast(&G.cv_take); +} + static QSlot *qs(int layer, int eid){ return &G.slot[(size_t)layer*G.ne + eid]; } static int home(int eid){ return eid % G.ndev; } @@ -156,7 +163,7 @@ static void *uploader(void *arg){ pthread_cond_broadcast(&G.cv_take); /* queue space available */ if(ve>=0){ /* LFRU swap: free the victim only when no group is in flight */ - while(G.issue_open && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx); + while(G.issue_open && !G.th_stop) wait_take_locked(); QSlot *v=qs(vl,ve); if(G.th_stop && G.issue_open){ /* Shutting down with a group still open: qt_take() -- the only @@ -205,6 +212,11 @@ static void *uploader(void *arg){ && coli_cuda_tensor_upload(&td, w+2*mb, sc+2*G.Ih, 2, G.Ih, G.D, dv); } free(w); free(sc); + if(!ok){ + if(tg) coli_cuda_tensor_free(tg); + if(tu) coli_cuda_tensor_free(tu); + if(td) coli_cuda_tensor_free(td); + } pthread_mutex_lock(&G.mx); QSlot *s=qs(layer,eid); if(ok){ s->tg=tg; s->tu=tu; s->td=td; s->resident=1; G.uploads++; } @@ -408,6 +420,29 @@ static double auto_displaced_value(int di, size_t room, int k, size_t exp_bytes, return value; } +/* Whether the trunk placement is the automatic one (COLI_PLACE unset or + * "auto"): the only decision the engine's startup probe may overturn. A + * hand-written list, or "off", is the user's word and stands. */ +int qt_place_is_auto(void){ return G_auto_on && auto_mode(); } + +/* Undo the automatic trunk placement: every offer back to the CPU, lm_head and + * the DeltaNet projections included, and the bytes it had taken back into each + * device's expert budget. The engine calls this BEFORE any trunk upload, when + * its startup probe measured the GPU GEMV slower than the CPU's: on four Tesla + * M10 every placed component lost, lm_head 68.8 ms against 41.7 on the CPU, + * decode 2.68 against 3.56 tok/s (#1652). Nothing to undo when the placement + * was not automatic. */ +void qt_trunk_withdraw(const char *why){ + if(!qt_place_is_auto()) return; + size_t back = 0; + for(int o = 0; o < G_offer_n; o++) G_offer[o].dev = QT_PLACE_CPU; + G_auto_lmh = QT_PLACE_CPU; G_lmh.dev_ok = 0; + for(int l = 0; l < QT_DN_MAX_LAYERS; l++) G_auto_dnp[l] = QT_PLACE_CPU; + for(int i = 0; i < G.ndev; i++){ back += G_trunk_bytes[i]; G.budget[i] += G_trunk_bytes[i]; G_trunk_bytes[i] = 0; } + fprintf(stderr,"[place] trunk stays on the CPU (%s): %.2f GB of VRAM back to the experts\n", + why ? why : "withdrawn", back/1073741824.0); +} + static void auto_place(int nl, int ne, int topk, const size_t *capacity, const uint32_t *heat0){ size_t room[QT_MAX_DEV]; for(int i = 0; i < G.ndev; i++){ room[i] = capacity[i]; G_trunk_bytes[i] = 0; } @@ -468,6 +503,7 @@ static const float *G_fp8_lut; static int G_upload_sync; /* QT_UPLOAD_SYNC=1: qt_issue waits for in-flight uploads first (tests) */ int qt_init_fp8(int nl, int ne, int D, int Ih, int cap, int topk, const float *e4m3_lut){ + if(G.on) return 0; G_fp8_stream = 1; G_fp8_lut = e4m3_lut; int ok = qt_init(nl, ne, D, Ih, cap, topk, 0, 0); if(!ok) G_fp8_stream = 0; @@ -494,6 +530,7 @@ static size_t dev_alloc_footprint(size_t bytes){ int qt_init(int nl, int ne, int D, int Ih, int cap, int topk, int expert_gs, int expert_is_int4){ + if(G.on) return 0; const char *e=getenv("COLI_CUDA"); if(!(e && *e=='1')) return 0; if(cap != ne && !G_fp8_stream){ @@ -531,7 +568,7 @@ int qt_init(int nl, int ne, int D, int Ih, int cap, int topk, int expert_gs, * the caller repeat every device in COLI_GPUS as well -- forgetting that * would silently drop a component back to the CPU mid-A/B. */ { - static const char *comps[] = {"lmhead","dnproj","dnout","attnproj"}; + static const char *comps[] = {"lmhead","dnproj","dnout","attnproj","shexp"}; for(size_t ci=0; ci CPU\n",ld); } /* every other component's devices, deduplicated */ - static const char *comps[] = {"dnproj","dnout","attnproj"}; + static const char *comps[] = {"dnproj","dnout","attnproj","shexp"}; for(size_t ci=0; ci COLI_GPUS bleibt\n",ed); } else if(!G_auto_on && G_place_n && qt_place_named("experts")){ fprintf(stderr,"[place] experts=cpu -> VRAM-Tier aus\n"); - return 0; + goto fail_storage; } else if(nres && !G_auto_on){ int w=0; for(int i=0;i= G_dense_n || !G_dense[h].on) return 0; - if(coli_cuda_matmul(&G_dense[h].t, y, x, NULL, NULL, 1, 1, I, O, G_dense[h].dev, 0)) return 1; +int qt_dense_matmul_batch(int h, float *y, const float *x, int S, int I, int O){ + if(h < 0 || h >= G_dense_n || !G_dense[h].on || S <= 0) return 0; + if(coli_cuda_matmul(&G_dense[h].t, y, x, NULL, NULL, 1, S, I, O, G_dense[h].dev, 0)) return 1; fprintf(stderr,"[dense] handle %d GPU matmul failed; CPU from here on\n", h); G_dense[h].on = 0; return 0; } +int qt_dense_matmul(int h, float *y, const float *x, int I, int O){ + return qt_dense_matmul_batch(h, y, x, 1, I, O); +} int qt_dense_count(void){ return G_dense_n; } -int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O){ - if(layer < 0 || layer >= QT_DN_MAX_LAYERS || !G_dnp[layer].on) return 0; - if(coli_cuda_matmul(&G_dnp[layer].t,y,x,NULL,NULL,1,1,I,O,G_dnp[layer].dev,0)) +int qt_dnproj_ready(int layer){ + return layer >= 0 && layer < QT_DN_MAX_LAYERS && G_dnp[layer].on; +} +int qt_dnproj_matmul_batch(int layer, float *y, const float *x, int S, int I, int O){ + if(!qt_dnproj_ready(layer) || S <= 0) return 0; + if(coli_cuda_matmul(&G_dnp[layer].t,y,x,NULL,NULL,1,S,I,O,G_dnp[layer].dev,0)) return 1; fprintf(stderr,"[dnp] layer %d GPU matmul failed; CPU from here on\n", layer); G_dnp[layer].on = 0; return 0; } +int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O){ + return qt_dnproj_matmul_batch(layer, y, x, 1, I, O); +} + int qt_lmhead_matmul(float *y, const float *x, int I, int O){ if(!G_lmh.on) return 0; /* cached-tensor path: upload params are ignored once *t exists */ @@ -899,7 +963,7 @@ void qt_note_block(int layer,int eid, pthread_mutex_lock(&G.mx); if(G_fp8_stream) stream_point(s,g4,u4,d4,gs,us,ds); else if(!s->g4){ s->g4=g4; s->u4=u4; s->d4=d4; s->gs=gs; s->us=us; s->ds=ds; } - while(G.qn>=QT_QCAP && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx); + while(G.qn>=QT_QCAP && !G.th_stop) wait_take_locked(); enqueue_locked(layer,eid,-1,-1,0); if(G_fp8_stream) stream_forget(s); pthread_mutex_unlock(&G.mx); @@ -992,7 +1056,7 @@ void qt_note_planned(int layer,int eid, } if(G_fp8_stream) stream_point(s,g4,u4,d4,gs,us,ds); else if(!s->g4){ s->g4=g4; s->u4=u4; s->d4=d4; s->gs=gs; s->us=us; s->ds=ds; } - while(G.qn>=QT_QCAP && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx); + while(G.qn>=QT_QCAP && !G.th_stop) wait_take_locked(); if(!enqueue_locked(layer,eid,-1,-1,1)){ /* not enqueueable (e.g. already resident): return the reservation */ if(s->planned) G.used[home(eid)]-=G.exp_bytes; @@ -1011,7 +1075,7 @@ void qt_note_planned(int layer,int eid, void qt_fill_wait(void){ if(!G.on) return; pthread_mutex_lock(&G.mx); - while(G.inflight>0 && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx); + while(G.inflight>0 && !G.th_stop) wait_take_locked(); pthread_mutex_unlock(&G.mx); } @@ -1058,7 +1122,8 @@ uint32_t qt_issue(int layer,const int *eids,int K,const float *x){ * did not get scheduled once before the run was over (0 uploads, 0 hits, * six entries still queued). No group is open here, so the wait cannot * meet a swap parked on issue_open. */ - if(G_upload_sync) while(G.inflight>0 && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx); + if(G_upload_sync) while(G.inflight>0 && !G.th_stop) wait_take_locked(); + if(G.th_stop){ pthread_mutex_unlock(&G.mx); return 0; } if(layer==0) qt_lfru_tick_locked(); G.issue_open=1; for(int k=0;ktg) coli_cuda_tensor_free(s->tg); + if(s->tu) coli_cuda_tensor_free(s->tu); + if(s->td) coli_cuda_tensor_free(s->td); + } + free(G.slot); G.slot=NULL; + free(G.is_x); G.is_x=NULL; G.is_x_floats=0; + free(G.fill_order); G.fill_order=NULL; + free(G.heat0); G.heat0=NULL; + pthread_cond_destroy(&G.cv_take); + pthread_cond_destroy(&G.cv); + pthread_mutex_destroy(&G.mx); + G.issue_open=0; + memset(G.is_cnt,0,sizeof G.is_cnt); G.on=0; G_fp8_stream=0; coli_cuda_shutdown(); diff --git a/c/qwen36_tier.h b/c/qwen36_tier.h index 5caef9f6e..41a73bf73 100644 --- a/c/qwen36_tier.h +++ b/c/qwen36_tier.h @@ -65,6 +65,12 @@ int qt_place_of(const char *component, int layer); * bytes come out of that device's expert budget. Sizes only; the tensors * follow through qt_lmhead_init / qt_dnproj_init as before. */ void qt_trunk_offer(const char *component, int layer, size_t bytes); +/* The automatic placement is a prediction; the engine measures it at startup + * (one GEMV both ways, qwen36.c trunk_probe_gpu_wins) and withdraws the whole + * trunk when the GPU loses, giving the bytes back to the expert budget. Only + * the automatic placement can be withdrawn; a COLI_PLACE list stands. */ +int qt_place_is_auto(void); +void qt_trunk_withdraw(const char *why); /* DeltaNet input projections, qkv ++ z fused into one resident tensor per * layer: one GEMV instead of two, and the engine's qkv/z buffers are laid out @@ -72,6 +78,8 @@ void qt_trunk_offer(const char *component, int layer, size_t bytes); int qt_dnproj_init(int layer, const int8_t *q, const float *sc, int I, int O, int device); int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O); +int qt_dnproj_ready(int layer); +int qt_dnproj_matmul_batch(int layer, float *y, const float *x, int S, int I, int O); /* Generic resident dense matrix (int8 per-row, one GEMV per call), addressed * by a handle: the Qwen3.8 trunk uses this for every matrix it places. Offer * the size with qt_trunk_offer(name, layer, bytes) before qt_init, ask @@ -79,6 +87,8 @@ int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O); * Returns the handle (>= 0) or -1 (stays on the CPU). */ int qt_dense_init(const int8_t *q, const float *sc, int I, int O, int device); int qt_dense_matmul(int handle, float *y, const float *x, int I, int O); +/* Row-major x[S,I] -> y[S,O], using the same resident int8 tensor. */ +int qt_dense_matmul_batch(int handle, float *y, const float *x, int S, int I, int O); int qt_dense_count(void); /* fp8 streaming mode (Qwen3.8): experts arrive as e4m3 bytes with 128x128 @@ -86,6 +96,8 @@ int qt_dense_count(void); * the tier copies what it uploads inside the qt_note call and keeps no * pointer into the engine's slot. e4m3_lut is quant.h's E4M3_LUT, published * to the backend so fmt=8 uploads are accepted. */ +/* Init returns 0 without changing an active tier. Shut down before reinit; + * callers must serialize init/shutdown with new work. */ int qt_init_fp8(int n_layers, int n_experts, int hidden, int inter, int cap_experts_per_layer, int topk, const float *e4m3_lut); int qt_init(int n_layers, int n_experts, int hidden, int inter, @@ -108,8 +120,10 @@ void qt_note(int layer, int eid, * the GPU. Compute the misses on the CPU, then call qt_take(). */ uint32_t qt_issue(int layer, const int *eids, int K, const float *x); -/* Collect the GPU results and accumulate val[k]*y_k into out[hidden]. */ -void qt_take(uint32_t mask, const float *val, int K, float *out); +/* Collect all GPU results and accumulate val[k]*y_k into out[hidden]. + * Returns 0 on collection failure, leaving out unchanged. The caller must + * stop inference: experts selected by qt_issue were not computed on CPU. */ +int qt_take(uint32_t mask, const float *val, int K, float *out); /* Warmstart: plan the full fill set (heat order, budget reserved), then any * number of loader threads may call qt_note_planned per planned expert. */ @@ -135,17 +149,22 @@ static inline int qt_lmhead_matmul(float*a,const float*b,int c,int d){(void)a;( #define QT_PLACE_CPU (-1) static inline int qt_place_of(const char*a,int b){(void)a;(void)b;return QT_PLACE_CPU;} static inline void qt_trunk_offer(const char*a,int b,size_t c){(void)a;(void)b;(void)c;} +static inline int qt_place_is_auto(void){return 0;} +static inline void qt_trunk_withdraw(const char*a){(void)a;} static inline int qt_dnproj_init(int a,const int8_t*b,const float*c,int d,int e,int f){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;return 0;} static inline int qt_dnproj_matmul(int a,float*b,const float*c,int d,int e){(void)a;(void)b;(void)c;(void)d;(void)e;return 0;} +static inline int qt_dnproj_ready(int a){(void)a;return 0;} +static inline int qt_dnproj_matmul_batch(int a,float*b,const float*c,int d,int e,int f){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;return 0;} static inline int qt_dense_init(const int8_t*a,const float*b,int c,int d,int e){(void)a;(void)b;(void)c;(void)d;(void)e;return -1;} static inline int qt_dense_matmul(int a,float*b,const float*c,int d,int e){(void)a;(void)b;(void)c;(void)d;(void)e;return 0;} +static inline int qt_dense_matmul_batch(int a,float*b,const float*c,int d,int e,int f){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;return 0;} static inline int qt_dense_count(void){return 0;} static inline int qt_ready(void){return 0;} static inline int qt_is_resident(int a,int b){(void)a;(void)b;return 0;} static inline void qt_shutdown(void){} static inline void qt_note(int a,int b,const uint8_t*c,const uint8_t*d,const uint8_t*e,const float*f,const float*g,const float*h){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;(void)g;(void)h;} static inline uint32_t qt_issue(int a,const int*b,int c,const float*d){(void)a;(void)b;(void)c;(void)d;return 0;} -static inline void qt_take(uint32_t a,const float*b,int c,float*d){(void)a;(void)b;(void)c;(void)d;} +static inline int qt_take(uint32_t a,const float*b,int c,float*d){(void)b;(void)c;(void)d;return a==0;} static inline int qt_plan_fill(int*a,int*b,int c){(void)a;(void)b;(void)c;return 0;} static inline void qt_note_planned(int a,int b,const uint8_t*c,const uint8_t*d,const uint8_t*e,const float*f,const float*g,const float*h){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;(void)g;(void)h;} static inline int qt_fill_next(int*a,int*b){(void)a;(void)b;return 0;} diff --git a/c/qwen38.c b/c/qwen38.c index 9eea6cfb1..494f0b389 100644 --- a/c/qwen38.c +++ b/c/qwen38.c @@ -1514,8 +1514,10 @@ static int q38_prefix_cache_save(Model *m,const int *ids,int len,const float *lo static int q38_prefix_restore(Model *m,const int *ids,int len){ if(!m||!ids||len<1||!g_q38_prefix.valid||g_q38_prefix.owner!=m|| g_q38_prefix.len<1||g_q38_prefix.len>len|| - memcmp(g_q38_prefix.ids,ids,(size_t)g_q38_prefix.len*sizeof(int)))return 0; - q38_prefix_copy_state(m,0);m->kv_len=g_q38_prefix.len;return g_q38_prefix.len; + memcmp(g_q38_prefix.ids,ids,(size_t)g_q38_prefix.len*sizeof(int))|| + !kv_prefix_holds(&m->kvp,g_q38_prefix.ids,g_q38_prefix.len))return 0; + q38_prefix_copy_state(m,0);m->kv_len=g_q38_prefix.len; + m->kvp.len=g_q38_prefix.len;return g_q38_prefix.len; } static const float *q38_prefix_cached_logits(Model *m){ @@ -1595,6 +1597,22 @@ static void serve_hits(Model *m){ printf("HITS %d %d %s\n",rows,E,hex); fflush(stdout); free(hex); free(bm); } +/* The generation budget a request gets. max_tokens is a CEILING, not a + * target (#260/#382, the rule GLM and DeepSeek V4 already apply): the prompt + * must fit with room for one token (none for a read-only logprobs request, + * docs/brio.md), and the budget is then clamped to what the context can hold. + * Returns the budget, or -1 when the PROMPT does not fit. Refusing when + * prompt + budget exceeded the context (#1641) turned the gateway's default + * output budget -- 8192 here, the whole default context -- into a 400 on + * every message of `coli chat` and on every request without max_tokens. */ +static int q38_serve_budget(int np, int max_tok, int max_ctx, int read_only){ + if (np < 1) return -1; + int room = max_ctx - np; + if (read_only) return room < 0 ? -1 : (max_tok < room ? max_tok : room); + if (room < 1) return -1; + return max_tok > room ? room : max_tok; +} + static int serve_one(Model *m, ServeReq *q){ int *ids=NULL, np=0; encode_text_n(q->payload,(size_t)q->plen,&ids,&np); /* byte-counted prompt; qwen38 adds no BOS */ @@ -1619,11 +1637,16 @@ static int serve_one(Model *m, ServeReq *q){ } int max_ctx=m->kv_cap; /* max_tokens=0 in modalita jev: leggere il prompt e fermarsi (serve_codec.h) */ - int tok_min = (q->logprobs > 0) ? 0 : 1; - if(np<1 || np>max_ctx || q->max_tokmax_tok>max_ctx-np){ + int budget = q38_serve_budget(np, q->max_tok, max_ctx, q->logprobs > 0); + if(budget < 0){ printf("ERROR %s CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d\n",q->id,np,q->max_tok,max_ctx); fflush(stdout); free(ids); return 0; } + if(budget < q->max_tok){ + fprintf(stderr,"[serve] max_tokens %d clamped to %d (context %d - prompt %d); raise Q38_MAXT for longer answers\n", + q->max_tok, budget, max_ctx, np); + q->max_tok = budget; + } printf("ACCEPT %s %d\n",q->id,np); fflush(stdout); double request_started=now_s(); uint64_t hits_before=m->hits, misses_before=m->miss; @@ -1632,13 +1655,20 @@ static int serve_one(Model *m, ServeReq *q){ * il client ha dichiarato, e battono la cache automatica, che insegue solo * l'ultimo prompt. Se nessuno serve, si ricade su quella. */ int reuse=0; const float *pin_lo=NULL; + /* Image embeddings are not described by token identity. */ + if(m->vis_map && m->vis_rows_n>0){ + kv_prefix_taint(&m->kvp);q38_prefix_cache_invalidate(); + } { int ps=coli_pin_best(&g_q38_pins,ids,np); while(ps>=0){ ColiPin *k=&g_q38_pins.slot[ps]; Q38PinState *st=(Q38PinState*)k->state; - if(st && q38_pin_state_copy(m,&st,0)){ - m->kv_len=k->len; reuse=k->len; pin_lo=k->logit; + /* Pins omit K/V/indexer rows. Reject one whose rows were overwritten. */ + if(st && kv_prefix_holds(&m->kvp,k->ids,k->len) && + q38_pin_state_copy(m,&st,0)){ + m->kv_len=k->len; m->kvp.len=k->len; + reuse=k->len; pin_lo=k->logit; coli_pin_touch(&g_q38_pins,ps); break; } @@ -1918,7 +1948,7 @@ int main(int argc, char **argv) { Model m; model_init(&m, snap, cap, bits); q38_tier_start(&m, cap); /* COLI_CUDA=1: hot experts stream to VRAM (qwen36_tier.c) */ - q38_trunk_cpu_int8(&m); /* Q38_TRUNK_CPU_INT8=1: the trunk's int8 rows on the CPU (reference) */ + q38_trunk_cpu_int8(&m); /* the trunk's int8 rows on the CPU, BF16 released (Q38_TRUNK_CPU_INT8=0 keeps BF16) */ if(is_ref)ref_logits=read_reference_logits(ref_root,m.c.vocab); g_capture_last_logit=ref_logits!=NULL||getenv("DUMP")!=NULL; q38_telemetry_init(snap, &m); diff --git a/c/qwen38_core.h b/c/qwen38_core.h index 30dcfdb24..8c658e5ae 100644 --- a/c/qwen38_core.h +++ b/c/qwen38_core.h @@ -9,6 +9,7 @@ */ #ifndef COLI_QWEN38_CORE_H #define COLI_QWEN38_CORE_H +#include "kv_prefix.h" #include /* q38_ehit_mark publishes the lazy HITS table under a lock */ #define Q38_MAX_LAYERS 512 @@ -50,7 +51,7 @@ typedef struct { Q38WeightKind kind; unsigned owns_data:1, owns_scales:1; int gpu; /* 0 = CPU; else 1 + tier handle of an int8 copy resident in VRAM (decode, S == 1) */ - int8_t *q8; float *q8sc; /* Q38_TRUNK_CPU_INT8=1: the same int8 rows kept on the CPU (reference for the GPU path, no GPU needed) */ + int8_t *q8; float *q8sc; /* the trunk's int8 rows on the CPU (default; Q38_TRUNK_CPU_INT8=0 keeps BF16): the same rows the GPU holds, met by an int8 activation in idot.h */ } Q38Weight; typedef struct { float *norm; Q38Weight down, up, inject; } GatedResidual; @@ -110,6 +111,15 @@ typedef struct { int ready; /* 0 unknown, 1 resident, -1 incompatible */ } Q38ExpertScaleCache; +typedef enum { + Q38_EXPERT_BATCH_FALLBACK_NONE = 0, + Q38_EXPERT_BATCH_FALLBACK_DISABLED, + Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY, + Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK, + Q38_EXPERT_BATCH_FALLBACK_DUPLICATE, + Q38_EXPERT_BATCH_FALLBACK_LAYOUT, +} Q38ExpertBatchFallback; + typedef struct { Cfg c; shards S; @@ -127,6 +137,7 @@ typedef struct { float **DN_rec, **DN_conv; float **K, **V, **IK; int kv_len, kv_cap, max_t; + kv_prefix kvp; /* token identity of the live attention rows, not a snapshot */ st_tensor *ple_parts[Q38_MAX_PLE_PARTS]; char ple_part_names[Q38_MAX_PLE_PARTS][320]; int64_t ple_part_start[Q38_MAX_PLE_PARTS + 1]; @@ -138,8 +149,10 @@ typedef struct { int ple_history_len; int range_begin, range_end; int native_fp8, native_bf16, expert_prefetch, expert_parallel_reads; + Q38ExpertBatchFallback expert_batch_fallback; int prefill_batch; uint64_t resident_weight_bytes; + int trunk_table_built; /* q38_trunk_offer_all ran for this load (the table is process-wide, the model is not) */ double dense_load_s; /* vision. `vis_map` mappa la posizione ASSOLUTA nella sequenza alla riga di * `vis_rows`, oppure -1. Assoluta e non relativa al chunk: il prefill arriva @@ -264,6 +277,75 @@ static void q38_matmul_bf16(float *y,const float *x,const uint16_t *W, } } +/* The routed experts as the checkpoint ships them: e4m3 bytes with one f32 + * scale per 128x128 block. quant.h's matmul_fp8 decodes every byte through a + * 256-entry table, one gather per weight; this kernel decodes eight bytes at a + * time in registers (quant.h e4m3_decode8: a shift and one multiply, NaNs + * kept) and multiplies them with FMA. The block scale still applies once per + * block and the blocks still add in double, so the result differs from the + * scalar kernel only by the float summation order inside a block. For a + * batch of rows (prefill) the block is decoded once and held while every row + * runs through it: the matrix streams past once, not once per row. + * Q38_FP8_KERNEL=scalar restores the table kernel (bisecting a difference). */ +static int q38_fp8_vector_on(void) { + static int v=-1; + if(v<0){ const char *e=getenv("Q38_FP8_KERNEL"); v=!(e&&!strcmp(e,"scalar")); } + return v; +} +#ifdef __AVX2__ +#define Q38_FP8_ROWS 8 +static void q38_matmul_fp8_vec(float *y,const float *x,const uint8_t *q8, + const float *bscale,int S,int I,int O) { + int64_t nblkI=fp8_nblk(I); + #pragma omp parallel for schedule(static) + for(int o=0;ogpu&&weight->rows==O&&weight->cols==I&& qt_dense_matmul(weight->gpu-1,y,x,I,O))return; - if(S==1&&weight&&weight->q8&&weight->rows==O&&weight->cols==I){ - /* the int8 rows the GPU would hold, computed here: what the trunk - * quantization alone does to the output, GPU or not */ - const int8_t *q=weight->q8; const float *sc=weight->q8sc; - #pragma omp parallel for schedule(static) - for(int o=0;oq8&&weight->rows==O&&weight->cols==I){ + /* the trunk's int8 rows (the same the GPU holds) meet an int8 + * activation in the integer kernel: x quantized once per row with one + * scale, then maddubs / vpdpbusd dot products (idot.h). Decode and + * prefill take the same path, so GPU or not the trunk quantization is + * the only thing that separates the output from the BF16 run. */ + int8_t *xq=(int8_t*)malloc((size_t)S*I); float *sx=(float*)malloc((size_t)S*sizeof(float)); + if(!xq||!sx){fprintf(stderr,"OOM activation quantization\n");exit(1);} + for(int s=0;sq8,weight->q8sc,S,I,O); + free(xq);free(sx); return; } if(!weight||weight->rows!=O||weight->cols!=I||!weight->data){ @@ -293,7 +376,7 @@ static void q38_weight_matmul(float *y,const float *x,const Q38Weight *weight, else if(weight->kind==Q38_WEIGHT_BF16) q38_matmul_bf16(y,x,(const uint16_t*)weight->data,S,I,O); else if(weight->kind==Q38_WEIGHT_FP8&&weight->scales) - matmul_fp8(y,x,(const uint8_t*)weight->data,weight->scales,S,I,O); + q38_matmul_fp8(y,x,(const uint8_t*)weight->data,weight->scales,S,I,O); else {fprintf(stderr,"unsupported matmul weight kind %d\n",(int)weight->kind);exit(1);} } @@ -1310,26 +1393,72 @@ typedef struct { * residents are protected from victim selection, so no worker can overwrite a * slot another selected expert will consume. Smaller caches and heterogeneous * layouts retain the serial LRU path. */ +/* Says once per model why the parallel read path is not taken, so a slow + * prefill on a small cache or a converted container is not a mystery. */ +static int q38_expert_batch_fallback(Model *m,Q38ExpertBatchFallback reason, + int layer,int first,int second) { + if(m->expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_NONE)return 0; + m->expert_batch_fallback=reason; + fprintf(stderr,"[qwen38 expert I/O] parallel reads unavailable: "); + switch(reason){ + case Q38_EXPERT_BATCH_FALLBACK_DISABLED: + fprintf(stderr,"Q38_EXPERT_PARALLEL_READS=0");break; + case Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY: + fprintf(stderr,"cache holds %d experts/layer but the route needs %d; " + "lower --ctx/Q38_MAXT or raise --ram",first,second);break; + case Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK: + fprintf(stderr,"layer %d has no compatible resident FP8 scale bank",layer);break; + case Q38_EXPERT_BATCH_FALLBACK_DUPLICATE: + fprintf(stderr,"layer %d route repeats expert %d",layer,first);break; + case Q38_EXPERT_BATCH_FALLBACK_LAYOUT: + fprintf(stderr,"layer %d expert %d is not native block-FP8",layer,first);break; + default: + fprintf(stderr,"unknown reason");break; + } + fprintf(stderr,"; using serial expert reads\n"); + return 0; +} + static int q38_expert_get_batch(Model *m,int layer,const int *experts,int count, Slot **selected) { - if(!m->expert_parallel_reads||!experts||!selected||count<2|| - count>Q38_MAX_TOPK)return 0; + if(!experts||!selected||count<2)return 0; + if(!m->expert_parallel_reads) + return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_DISABLED,layer,0,0); LCache *cache=&m->cache[layer]; - if(cache->capcache->cap) + return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY, + layer,cache->cap,count); + if(!q38_prepare_expert_scale_bank(m,layer)) + return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK,layer,0,0); + /* The demand set is no longer bounded by the decode top-k: the MoE prefill + * hands over the whole chunk union (up to the cache cap) so its loads run + * one OMP wave instead of serial groups of Q38_MAX_TOPK. Load grouping + * never touches FP order: routed outputs are written per assignment and + * the per-position expert sum follows the router order, so a bigger wave + * only changes WHICH slots serve the reads, not the arithmetic. */ + Q38ExpertLoadJob *jobs=malloc((size_t)count*sizeof(*jobs)); + if(!jobs)return 0; for(int index=0;index=m->c.experts)return 0; + if(expert<0||expert>=m->c.experts){free(jobs);return 0;} q38_ehit_mark(m,layer,expert); for(int previous=0;previousby_expert[expert]; if(slot_index>=0){ - if(slot_index>=cache->n||cache->slots[slot_index].eid!=expert)return 0; + if(slot_index>=cache->n||cache->slots[slot_index].eid!=expert){free(jobs);return 0;} continue; } st_tensor *weight[3]; - if(!q38_native_fp8_expert_tensors(m,layer,expert,weight))return 0; + if(!q38_native_fp8_expert_tensors(m,layer,expert,weight)){ + free(jobs); + return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_LAYOUT, + layer,expert,0); + } } unsigned char *protected_slots=(unsigned char*)calloc((size_t)cache->cap,1); if(!protected_slots){fprintf(stderr,"OOM expert batch reservations\n");exit(1);} @@ -1390,6 +1519,7 @@ static int q38_expert_get_batch(Model *m,int layer,const int *experts,int count, cache->by_expert[jobs[job].expert]=(int)(slot-cache->slots); } } + free(jobs); return 1; } @@ -1519,17 +1649,59 @@ static void q38_ple(Model *m,const int *ids,int S,const float *hyper,float *out) q38_tm_add(m,Q38_TM_PLE,phase_started); } +/* The chunk ceiling and the workspace budget used to be compile-time only. An + * isolated prefill measurement (274 tokens, one forward) showed that the + * ceiling -- not the budget -- is what binds: 32 rows cut the prompt into nine + * chunks, each chunk touches ~91 distinct experts, so a loaded expert serves + * ~3.3 rows. That drags 4.69 MiB of FP8 weights in for three rows of + * activations, which is decode-grade arithmetic intensity inside a path that is + * supposed to be batched, and it shows: 1.29 TFLOP in 48.8 s is 26.5 GFLOP/s, + * a few percent of what the cores can do. Both values are therefore runtime + * knobs now. Widening the chunk cannot change any result -- boundaries alter + * neither routing nor accumulation order -- so this is a pure A/B. */ +static int q38_env_positive_int(const char *name,int default_value, + int max_value) { + const char *value=getenv(name); + if(!value||!*value)return default_value; + char *end=NULL;long parsed=strtol(value,&end,10); + if(end==value||*end||parsed<1||parsed>(long)max_value){ + fprintf(stderr,"%s must be an integer in 1..%d\n",name,max_value); + exit(1); + } + return (int)parsed; +} + +static int q38_prefill_batch_rows(void) { + static int cached=0; + if(!cached) + cached=q38_env_positive_int("Q38_PREFILL_BATCH_ROWS", + Q38_PREFILL_BATCH_ROWS,1<<20); + return cached; +} + +/* Expressed in MiB because the byte count is the thing a human gets wrong. */ +static uint64_t q38_prefill_workspace_bytes(void) { + static uint64_t cached=0; + if(!cached) + cached=(uint64_t)q38_env_positive_int( + "Q38_PREFILL_WORKSPACE_MIB", + (int)(Q38_PREFILL_WORKSPACE_BYTES>>20),4096)<<20; + return cached; +} + /* Choose a context-independent prefill chunk whose private workspace fits the * common target. Callers provide exact fixed and per-row byte counts; even a * hostile-but-valid geometry gets one row rather than an unbounded allocation. */ static int q38_bounded_prefill_rows(int requested,uint64_t fixed, uint64_t per_row) { - int rows=requested1;rows--) if(per_row<=UINT64_MAX/(uint64_t)rows&& fixed<=UINT64_MAX-per_row*(uint64_t)rows&& - fixed+per_row*(uint64_t)rows<=Q38_PREFILL_WORKSPACE_BYTES) + fixed+per_row*(uint64_t)rows<=budget) return rows; return 1; } @@ -1655,6 +1827,11 @@ static void q38_attention(Model *m,Layer *l,int layer,const float *x,int S,int p float *ip=falloc((int64_t)S*(IQ+c->idx_kheads)*ID); q38_dense_matmul(m,qp,x,&l->q,S,H,QH*2*D);q38_dense_matmul(m,kp,x,&l->k,S,H,KVH*D);q38_dense_matmul(m,vp,x,&l->v,S,H,KVH*D); q38_dense_matmul(m,ip,x,&l->idx_qk,S,H,(IQ+c->idx_kheads)*ID); + /* Cause before parallelism: the K/V/IK writes are disjoint per position + * (each s writes only its own row) and must be complete before the + * ranking, which reads the whole IK[0..pos] prefix. Guarded on S>1 so + * decode keeps the serial path it has today. */ + #pragma omp parallel for schedule(static) if(S>1) for(int s=0;sIK[layer]+(int64_t)pos*ID,ip+(int64_t)s*(IQ+1)*ID+(int64_t)IQ*ID,(size_t)ID*sizeof(float)); } - float *heads=falloc((int64_t)S*QH*D),*qidx=falloc((int64_t)IQ*ID),*pool=falloc(ID); - int *selected=(int*)malloc((size_t)maxsel*sizeof(int)); - if(!selected){fprintf(stderr,"OOM QSA selection\n");exit(1);} + float *heads=falloc((int64_t)S*QH*D); + /* Ranking and attention are independent per position: no shared writes + * (heads is row-disjoint, the scratch is per-thread) and no FP order + * changes inside a position, so the result is bit-identical to the + * serial path. The scheduling is dynamic because the ranking cost grows + * with the position (the IK prefix to read is O(pos)). */ + double index_dt=0,attn_dt=0; + /* Wall, not aggregate CPU: the reduction below sums per-thread seconds, so + * at prefill these two phases would report ~20x what the clock saw while + * every other phase reports wall -- on a 3006-token prompt the phase sum + * came to 510 s against a 268 s TTFT, the parts outweighing the whole. + * Take the clock across the whole team and split it by CPU share. At + * decode S==1 the loop is serial, cpu_total equals the wall, and the + * rescale below is an exact no-op. */ + double qsa_wall_started=now_s(); + #pragma omp parallel for schedule(dynamic,8) reduction(+:index_dt,attn_dt) if(S>1) for(int s=0;sidx_qn,ID,c->eps);q38_rope(qh,ID,c->rotary_dim,pos,c->theta);} int take=blocksidx_budget/R?blocks:c->idx_budget/R,nsel=0; Q38Block *rank=blocks?(Q38Block*)malloc((size_t)blocks*sizeof(Q38Block)):NULL; @@ -1682,7 +1875,7 @@ static void q38_attention(Model *m,Layer *l,int layer,const float *x,int S,int p if(blocks)qsort(rank,(size_t)blocks,sizeof(Q38Block),q38_block_desc); for(int z=0;zqn,D,c->eps);q38_rope(qh,D,c->rotary_dim,pos,c->theta); @@ -1693,10 +1886,18 @@ static void q38_attention(Model *m,Layer *l,int layer,const float *x,int S,int p for(int j=0;jV[layer]+((int64_t)khidx*m->kv_cap+selected[j])*D;for(int d=0;d0.0){ + index_dt=qsa_wall*(index_dt/cpu_total); + attn_dt =qsa_wall*(attn_dt /cpu_total); } + m->timers.seconds[Q38_TM_QSA_INDEX]+=index_dt; + m->timers.seconds[Q38_TM_QSA_ATTENTION]+=attn_dt; q38_dense_matmul(m,out,heads,&l->o,S,QH*D,H); - free(qp);free(kp);free(vp);free(ip);free(heads);free(qidx);free(pool);free(selected); + free(qp);free(kp);free(vp);free(ip);free(heads); } /* The single-row path is intentionally kept separate from prefill. Decode is @@ -1725,15 +1926,16 @@ static void q38_tier_note(int layer,int eid,const Slot *ex) { * stays for prefill and as fallback. Q38_TRUNK_GPU=0 keeps the trunk on the * CPU (parity runs against the BF16 reference). */ typedef struct { Q38Weight *w; char name[16]; int layer; } Q38TrunkItem; -static Q38TrunkItem *g_trunk; static int g_trunk_n, g_trunk_cap; +static Q38TrunkItem *g_trunk; static int g_trunk_n, g_trunk_cap, g_trunk_offer_gpu; +static long g_trunk_min_kb; static const char *g_trunk_skip; /* read once per load in q38_trunk_offer_all */ static void q38_trunk_add(Q38Weight *w,const char *name,int layer) { if(!w||!w->data||(w->kind!=Q38_WEIGHT_BF16&&w->kind!=Q38_WEIGHT_F32))return; size_t bytes=(size_t)w->rows*w->cols+(size_t)w->rows*sizeof(float); /* Q38_TRUNK_MIN_KB (default 1024): a round trip costs more than a tiny - * GEMV saves; Q38_TRUNK_SKIP=name,name: leave those components on the - * CPU (bisecting a numeric difference, or a component that does not pay) */ - static long min_kb=-1; static const char *skip; - if(min_kb<0){ const char *e=getenv("Q38_TRUNK_MIN_KB"); min_kb=e?atol(e):1024; skip=getenv("Q38_TRUNK_SKIP"); } + * GEMV saves, and a tiny matrix in BF16 costs nothing on the CPU either; + * Q38_TRUNK_SKIP=name,name: leave those components in BF16 on the CPU + * (bisecting a numeric difference, or a component that does not pay) */ + long min_kb=g_trunk_min_kb; const char *skip=g_trunk_skip; if(bytes<(size_t)min_kb*1024)return; if(skip&&*skip){ size_t n=strlen(name); const char *s=skip; @@ -1747,15 +1949,19 @@ static void q38_trunk_add(Q38Weight *w,const char *name,int layer) { } Q38TrunkItem *it=&g_trunk[g_trunk_n++]; it->w=w; it->layer=layer; snprintf(it->name,sizeof it->name,"%s",name); - qt_trunk_offer(it->name,layer,bytes); + if(g_trunk_offer_gpu)qt_trunk_offer(it->name,layer,bytes); } static int q38_trunk_enabled(void) { const char *e=getenv("Q38_TRUNK_GPU"); return !(e&&e[0]=='0'&&!e[1]); } -/* offers, before qt_init: lm_head first (the placer takes it first), then the - * layers in order so a partial placement is a prefix of the layers */ +/* the trunk table, built once: lm_head first (the placer takes it first), + * then the layers in order so a partial placement is a prefix of the layers. + * The same table feeds the CPU's int8 rows; the placer is told about the + * matrices only when the GPU trunk is enabled (Q38_TRUNK_GPU). */ static void q38_trunk_offer_all(Model *m) { - if(!q38_trunk_enabled())return; + g_trunk_n=0; m->trunk_table_built=1; /* rebuilt per load: a test opens several models in one process */ + g_trunk_offer_gpu=q38_trunk_enabled(); + { const char *e=getenv("Q38_TRUNK_MIN_KB"); g_trunk_min_kb=e?atol(e):1024; g_trunk_skip=getenv("Q38_TRUNK_SKIP"); } Cfg *c=&m->c; q38_trunk_add(&m->lm_head,"lmhead",0); for(int l=0;llayers;l++){ @@ -1794,19 +2000,33 @@ static void q38_trunk_quantize(const Q38Weight *w,int8_t **qp,float **scp) { } *qp=q; *scp=sc; } -/* Q38_TRUNK_CPU_INT8=1: keep the int8 rows on the CPU instead (or as well), - * so the quantization can be judged without a GPU (PPL, token parity) */ +/* The trunk on the CPU: int8 rows with one scale per row, and the BF16 copy + * released. The trunk is read whole on every token (3.6 G weights on the + * released checkpoint, more than the ten routed experts), so its bytes are + * the decode's floor: int8 halves them and the integer kernel keeps up with + * the memory. Default on; Q38_TRUNK_CPU_INT8=0 keeps the BF16 rows and the + * f32 kernel, the numeric reference. A matrix the tier already quantized for + * the GPU keeps those same rows here (prefill rows run on the CPU). */ +static int q38_trunk_cpu_int8_wanted(void) { + const char *e=getenv("Q38_TRUNK_CPU_INT8"); return !(e&&e[0]=='0'&&!e[1]); +} static void q38_trunk_cpu_int8(Model *m) { - const char *e=getenv("Q38_TRUNK_CPU_INT8"); - if(!e||e[0]!='1'||e[1])return; - if(!g_trunk_n) q38_trunk_offer_all(m); - double t0=now_s(); size_t bytes=0; + if(!q38_trunk_cpu_int8_wanted())return; + if(!m->trunk_table_built)q38_trunk_offer_all(m); /* the tier may have built it already */ + double t0=now_s(); size_t bytes=0,released=0; int n=0; for(int i=0;iq8)continue; - q38_trunk_quantize(w,&w->q8,&w->q8sc); bytes+=(size_t)w->rows*w->cols; + Q38Weight *w=g_trunk[i].w; + if(!w->q8)q38_trunk_quantize(w,&w->q8,&w->q8sc); + bytes+=(size_t)w->rows*w->cols+(size_t)w->rows*sizeof(float); n++; + if(w->owns_data&&w->data){ + /* every path that reads this matrix now goes through q8 */ + uint64_t was=q38_weight_bytes(w); + free(w->data); w->data=NULL; w->owns_data=0; released+=was; + m->resident_weight_bytes-=was; m->resident_weight_bytes+=(size_t)w->rows*w->cols+(size_t)w->rows*sizeof(float); + } } - fprintf(stderr,"[qwen38] trunk: %d matrices int8 on the CPU (%.2f GiB) in %.1fs (Q38_TRUNK_CPU_INT8)\n", - g_trunk_n,bytes/1073741824.0,now_s()-t0); + fprintf(stderr,"[qwen38] trunk: %d matrices int8 on the CPU (%.2f GiB, %.2f GiB of BF16 released) in %.1fs; Q38_TRUNK_CPU_INT8=0 keeps BF16\n", + n,bytes/1073741824.0,released/1073741824.0,now_s()-t0); } /* after qt_init: quantize and upload what the placer accepted */ static void q38_trunk_place_all(Model *m) { @@ -1817,7 +2037,8 @@ static void q38_trunk_place_all(Model *m) { int dev=qt_place_of(it->name,it->layer); if(dev==QT_PLACE_CPU)continue; int O=w->rows,I=w->cols; - int8_t *q; float *sc; q38_trunk_quantize(w,&q,&sc); + if(!w->q8)q38_trunk_quantize(w,&w->q8,&w->q8sc); /* kept: the CPU answers prefill rows from the same bytes */ + const int8_t *q=w->q8; const float *sc=w->q8sc; int h=qt_dense_init(q,sc,I,O,dev); if(h>=0&&getenv("Q38_TRUNK_SELFTEST")){ /* DIAG: GPU int8 GEMV against the same int8 matrix on the CPU */ @@ -1829,11 +2050,10 @@ static void q38_trunk_place_all(Model *m) { fprintf(stderr,"[selftest] %-7s L%-2d [O=%d I=%d] ok=%d rel.err %.2e worst row %d gpu %.5g cpu %.5g\n",it->name,it->layer,O,I,ok,den>0?sqrt(num/den):-1.0,worst,yg[worst],yc[worst]); free(x);free(yg);free(yc); } - free(q); free(sc); if(h>=0){ w->gpu=h+1; placed++; placed_bytes+=(size_t)O*I; } } if(g_trunk_n) - fprintf(stderr,"[qtier] qwen38 trunk: %d of %d offered matrices resident as int8 (%.2f GiB) in %.1fs; the rest stays BF16 on the CPU\n", + fprintf(stderr,"[qtier] qwen38 trunk: %d of %d offered matrices resident as int8 (%.2f GiB) in %.1fs; the rest answers from the CPU\n", placed,g_trunk_n,placed_bytes/1073741824.0,now_s()-t0); } @@ -1900,7 +2120,10 @@ static void q38_moe_decode(Model *m,Layer *l,int layer,const float *x,int S,floa } /* GPU experts land after the CPU ones: same values, one more group in * the float sum (that is the only ordering difference to a CPU run). */ - qt_take(qmask,route_gates,K,ys); + if(!qt_take(qmask,route_gates,K,ys)){ + fprintf(stderr,"qwen38: CUDA expert collection failed at layer %d; stopping inference\n",layer); + exit(1); + } for(int d=0;dcache[layer].cap; - if(load_limit>Q38_MAX_TOPK)load_limit=Q38_MAX_TOPK; if(load_limit<1)load_limit=1; for(int unique_base=0;unique_basekvp); Cfg *c=&m->c; for(int i=0;ilayers;i++)if(!c->is_attn[i]){ memset(m->DN_rec[i],0,(size_t)c->dn_vheads*c->dn_kdim*c->dn_vdim*sizeof(float)); @@ -2146,6 +2369,7 @@ static void ensure_kv(Model *m) { m->K[i]=falloc((int64_t)c->kv_heads*m->max_t*c->head_dim);m->V[i]=falloc((int64_t)c->kv_heads*m->max_t*c->head_dim);m->IK[i]=falloc((int64_t)m->max_t*c->idx_dim); } m->kv_cap=m->max_t; + kv_prefix_alloc(&m->kvp,m->kv_cap); /* ensure_kv discards the old rows */ } /* Run only the requested native layer interval over hyper-residual activations. @@ -2226,6 +2450,10 @@ static float *step(Model *m,const int *ids,int S,int pos_base) { q38_gr_read(m,&l->mlp_gr,hyper,S,mixed,inject);q38_moe(m,l,i,mixed,S,block);q38_gr_apply(c,hyper,block,inject,S); } q38_gr_read(m,&m->final_gr,hyper,S,mixed,NULL);m->kv_len=pos_base+S; + /* Rewinding and writing a shorter branch invalidates its old tail. */ + if(m->kvp.len>pos_base)m->kvp.len=pos_base; + kv_prefix_record(&m->kvp,ids,pos_base,S); + if(m->vis_map && m->vis_rows_n>0)kv_prefix_taint(&m->kvp); float *logit=falloc(c->vocab);double phase_started=now_s(); /* Lettura del prefill: la posizione p predice il token p+1. Il primo token * fresco lo predice la fotografia del prefisso, quando c'e. Pagata solo da @@ -2310,6 +2538,7 @@ static void q38_layer_free(Layer *l) { static void q38_model_free(Model *m) { if(!m) return; + kv_prefix_free(&m->kvp); for(int i=0;ic.layers;i++) { if(m->L)q38_layer_free(&m->L[i]); if(m->cache) { diff --git a/c/resource_plan.py b/c/resource_plan.py index 18087a585..955f29a77 100644 --- a/c/resource_plan.py +++ b/c/resource_plan.py @@ -434,6 +434,8 @@ def read_ssd_probe(model_dir): def discover_gpus(): + if sys.platform == "darwin": + return _discover_metal_gpus() # NVIDIA first; if there are none (or no nvidia-smi), fall back to ROCm/HIP so # a working AMD engine isn't planned CPU-only and --gpu N stops failing (#662). devices = _discover_nvidia_gpus() @@ -442,6 +444,33 @@ def discover_gpus(): return _discover_amd_gpus() +def _discover_metal_gpus(): + """Return Apple Metal devices without pretending unified RAM is VRAM.""" + try: + result = subprocess.run( + ["system_profiler", "SPDisplaysDataType", "-json"], + text=True, capture_output=True, check=True, timeout=10) + displays = json.loads(result.stdout).get("SPDisplaysDataType", []) + except (OSError, subprocess.SubprocessError, ValueError, TypeError): + return [] + devices = [] + for index, display in enumerate(displays): + if not isinstance(display, dict): + continue + metal = display.get("spdisplays_mtlgpufamilysupport") + if not metal: + continue + name = display.get("sppci_model") or display.get("_name") + if not isinstance(name, str) or not name: + continue + # Apple Silicon has one unified pool. Its free capacity cannot be used + # as an independent VRAM budget, so retain device identity only. + devices.append({"index": index, "name": name, + "total_bytes": 0, "free_bytes": None, + "unified_memory": True, "backend": "metal"}) + return devices + + def _discover_nvidia_gpus(): command = ["nvidia-smi", "--query-gpu=index,name,memory.total,memory.free", "--format=csv,noheader,nounits"] @@ -934,6 +963,26 @@ def _next_actions(bottleneck_class, projected_hit, probe_state, probe_gbs, } +def _family_expert_cache_knob(family_id, cache_bytes): + """Env var that actually sizes the expert LRU, when it is not RAM_GB. + + Kimi K3 reads K3_EXPERT_GB (default 8) and treats RAM_GB as a ceiling; + main() never takes argv as a cap. GLM-5.3 reads GLM53_EXPERT_GB and never + RAM_GB. Exporting only RAM_GB therefore left both engines on their own + default cache under --auto-tier. + """ + if (not isinstance(cache_bytes, int) or isinstance(cache_bytes, bool) + or cache_bytes <= 0): + return None + if family_id == "kimi": + return ("K3_EXPERT_GB", + "Kimi K3 sizes the expert LRU from K3_EXPERT_GB; RAM_GB is only a ceiling") + if family_id == "glm53": + return ("GLM53_EXPERT_GB", + "GLM-5.3 sizes the expert LRU from GLM53_EXPERT_GB, not RAM_GB") + return None + + def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0, available_memory=None, available_disk=None, gpus=None, policy="quality", physical_cpus=None, cpu_sockets=None, @@ -1100,6 +1149,13 @@ def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0, tune = _auto_tune(bottleneck_class, projected_hit, planning_gpus, cpu_sockets, plan_has_metal=False, engine_group=resolved.descriptor.engine_group) + # DRAFT/PIPE/PIN/NUMA stay colibri.c-only (#1585). These two are the + # opposite leftover: the engines that size the expert LRU from their own + # *EXPERT_GB variable, which RAM_GB does not set. + knob = _family_expert_cache_knob(resolved.descriptor.id, cache_bytes) + if knob: + name, reason = knob + tune[name] = {"value": f"{cache_bytes / GB:.3f}", "reason": reason} probe_state, probe_gbs = ssd_probe_state(info["path"]) actions = _next_actions(bottleneck_class, projected_hit, probe_state, probe_gbs, planning_gpus) @@ -1184,6 +1240,10 @@ def environment_for_plan(plan, env=None, cuda_enabled=True): result.setdefault("REPIN", "64") ram = plan["tiers"]["ram"] result.setdefault("RAM_GB", f"{ram['budget_bytes'] / GB:.3f}") + knob = _family_expert_cache_knob(plan.get("model", {}).get("family_id"), + ram.get("expert_cache_bytes")) + if knob: + result.setdefault(knob[0], f"{ram['expert_cache_bytes'] / GB:.3f}") planned_cap = ram.get("cache_slots_per_layer") if (plan.get("model", {}).get("family_id") == "qwen38" and (not isinstance(planned_cap, int) or @@ -1241,14 +1301,22 @@ def format_plan(plan): f"cap {tiers['ram']['cache_slots_per_layer']}/layer"] vram = tiers["vram"] if vram["devices"]: - names = ", ".join( - f"{gpu['index']}:{gpu['name']}" - + ("" if plans_placement(gpu) else " (identity only)") - for gpu in vram["devices"]) - trunk = vram.get("trunk_bytes", 0) - lines.append("VRAM " + (f"{format_bytes(trunk)} int8 trunk + " if trunk else "") + - f"{format_bytes(vram['budget_bytes'])} hot tier · " - f"~{vram['expert_capacity']} experts · {names}") + metal_identity = [gpu for gpu in vram["devices"] + if gpu.get("backend") == "metal" + and gpu.get("free_bytes") is None] + if len(metal_identity) == len(vram["devices"]): + names = ", ".join(f"{gpu['index']}:{gpu['name']}" + for gpu in metal_identity) + lines.append(f"Metal {names} · unified memory · no independent VRAM budget") + else: + names = ", ".join( + f"{gpu['index']}:{gpu['name']}" + + ("" if plans_placement(gpu) else " (identity only)") + for gpu in vram["devices"]) + trunk = vram.get("trunk_bytes", 0) + lines.append("VRAM " + (f"{format_bytes(trunk)} int8 trunk + " if trunk else "") + + f"{format_bytes(vram['budget_bytes'])} hot tier · " + f"~{vram['expert_capacity']} experts · {names}") else: # Backend-neutral, matching the accelerator wording #903 settled on: # an AMD or Intel host that finds nothing is not "no NVIDIA device". diff --git a/c/serve_budget.h b/c/serve_budget.h new file mode 100644 index 000000000..36c8ae507 --- /dev/null +++ b/c/serve_budget.h @@ -0,0 +1,21 @@ +#ifndef COLI_SERVE_BUDGET_H +#define COLI_SERVE_BUDGET_H + +/* max_tokens is a ceiling, not a target (#260/#382/#1641). + * Returns the tokens the request may generate, or -1 when the PROMPT does + * not fit. Generation needs one free position; a read-only logprobs request + * may fill the context exactly. Refusing when prompt + budget exceeded the + * context turned coli chat's interactive default (16384) into a 400 on every + * Kimi/Inkling message against the default 8192-token window. */ + +static inline int coli_serve_budget(int prompt, int requested, int context, + int read_only) +{ + if (prompt < 1) return -1; + int room = context - prompt; + if (read_only) return room < 0 ? -1 : (requested < room ? requested : room); + if (room < 1) return -1; + return requested > room ? room : requested; +} + +#endif diff --git a/c/sse41_kernels.h b/c/sse41_kernels.h new file mode 100644 index 000000000..4f93eee40 --- /dev/null +++ b/c/sse41_kernels.h @@ -0,0 +1,162 @@ +#ifndef COLIBRI_SSE41_KERNELS_H +#define COLIBRI_SSE41_KERNELS_H +/* + * sse41_kernels.h — shared SSE 4.1 primitives for Colibri engines. + * + * The engines (c/deepseek_v4.c, c/colibri.c, c/kimi_k3.c, c/olmoe.c, c/inkling.c) + * have an `#if defined(__AVX2__)` dispatch for fast paths and fall through to scalar + * on pre-Haswell hardware (Sandy Bridge: AVX 1.0, no FMA, no AVX-2). This header + * provides the missing middle tier: 128-bit SIMD primitives, FMA-free. + * + * Why a separate header (not just inline in each .c): + * - 109 AVX2 sites total across 5 engines need patching. Copy-pasting the + * 128-bit intrinsics 109 times is a typo factory. A single macro definition + * is the difference between correct and wrong-on-100-sites. + * - The single most critical shared piece is the FMA-emulation macro: on + * Sandy Bridge, _mm_mul_ps + _mm_add_ps has double rounding vs. hardware + * FMA's single rounding, so the output is NOT bit-identical to AVX2 (1-2 ULP + * difference). A typo in a single copy is a silent correctness bug. + * + * This is the minimum needed for the SSE 4.1 fallback. More primitives can be + * added as additional engines are patched. + */ +#if defined(__SSE2__) + +#include +#include + +/* + * COLIBRI_FMA: emulate FMA on non-FMA hardware. + * + * On FMA hardware: maps to _mm_fmadd_ps (single rounding, 1 instruction). + * On Sandy Bridge (no FMA): separate mul+add, double rounding, 2 instructions. + * + * On Sandy Bridge this is NOT bit-identical to AVX2/FMA output — typically + * within 1-2 ULP. Test tolerance must accommodate this. + */ +#if defined(__FMA__) +# define COLIBRI_FMA(a, b, c) _mm_fmadd_ps((a), (b), (c)) +#else +# define COLIBRI_FMA(a, b, c) _mm_add_ps(_mm_mul_ps((a), (b)), (c)) +#endif + +/* + * SSE 4.1 (or lower) load/store helpers. Sandy Bridge has these natively. + * _mm_load_ps is aligned; _mm_loadu_ps is unaligned. For 128-bit (16-byte) + * data, aligned loads are faster but UB on misaligned pointers. Default to + * unaligned: buffers from malloc / numa_slab_bind have no 16-byte guarantee. + * Aligned loads can be added as a profiled follow-up. + */ +static inline __m128 colibri_sse41_loadu_ps(const float *p) { return _mm_loadu_ps(p); } +static inline void colibri_sse41_storeu_ps(float *p, __m128 v) { _mm_storeu_ps(p, v); } + +/* + * Min/max (SSE 4.1 native). Identical to AVX2, just narrower width. + */ +static inline __m128 colibri_sse41_min_ps(__m128 a, __m128 b) { return _mm_min_ps(a, b); } +static inline __m128 colibri_sse41_max_ps(__m128 a, __m128 b) { return _mm_max_ps(a, b); } + +/* + * Prefetch (SSE 1+, always available). Identical to AVX2/FMA path. + */ +static inline void colibri_sse41_prefetch(const void *p) { _mm_prefetch(p, _MM_HINT_T0); } + +/* Grouped-int4 kernels for the SSE4.1 tier. Callers select even group sizes + * and pass a multiple of four output rows; scalar dispatch handles the rest. + * Keep the scalar pair-sum, scale-multiply and accumulator-add order intact. */ +#if defined(__SSE4_1__) && !defined(__AVX2__) +/* Load one packed byte from each of four output rows, then unpack their low and + * high offset nibbles into four f32 lanes. Keeping independent output rows in + * the lanes preserves the scalar operation order within every row. */ +static inline void colibri_sse41_i4_rows4(const uint8_t *q4,int rb,int o,int byte, + __m128 *lo,__m128 *hi){ + const __m128i m4=_mm_set1_epi8(0x0F), b8=_mm_set1_epi8(8); + /* Read exactly one byte per row, including when a row is one byte wide. */ + __m128i by=_mm_cvtsi32_si128(q4[(int64_t)(o+0)*rb+byte]); + by=_mm_insert_epi8(by,q4[(int64_t)(o+1)*rb+byte],1); + by=_mm_insert_epi8(by,q4[(int64_t)(o+2)*rb+byte],2); + by=_mm_insert_epi8(by,q4[(int64_t)(o+3)*rb+byte],3); + __m128i qlo=_mm_sub_epi8(_mm_and_si128(by,m4),b8); + __m128i qhi=_mm_sub_epi8(_mm_and_si128(_mm_srli_epi16(by,4),m4),b8); + *lo=_mm_cvtepi32_ps(_mm_cvtepi8_epi32(qlo)); + *hi=_mm_cvtepi32_ps(_mm_cvtepi8_epi32(qhi)); +} + +static inline __m128 colibri_sse41_f32_rows4(const float *p,int stride,int o,int i){ + return _mm_set_ps(p[(int64_t)(o+3)*stride+i],p[(int64_t)(o+2)*stride+i], + p[(int64_t)(o+1)*stride+i],p[(int64_t)(o+0)*stride+i]); +} + +/* Process four output rows at once without a horizontal reduction. Each lane + * uses the scalar kernel's pair sum, scale multiply, and accumulator add in + * the same order, so the result can remain byte-identical on pre-FMA CPUs. */ +static void matmul_i4_grouped_sse41_rows4(float *y,const float *x, + const uint8_t *q4,const float *scale, + int S,int I,int O,int gs,int rb,int ng, + int o4){ + #pragma omp parallel for schedule(static) + for(int o=0;oI) end=I; + __m128 sc=colibri_sse41_f32_rows4(scale,ng,o,g); int i=base; + for(;i+1>1,&lo,&hi); + __m128 pair=_mm_add_ps(_mm_mul_ps(_mm_set1_ps(xs[i]),lo), + _mm_mul_ps(_mm_set1_ps(xs[i+1]),hi)); + a=_mm_add_ps(a,_mm_mul_ps(pair,sc)); + } + if(i>1,&lo,&hi); (void)hi; + a=_mm_add_ps(a,_mm_mul_ps(_mm_mul_ps(_mm_set1_ps(xs[i]),lo),sc)); + } + } + colibri_sse41_storeu_ps(y+(int64_t)s*O+o,a); + } + } +} + +/* Fused gate/up shares activation loads while keeping independent accumulators. */ +static void matmul_i4_grouped_pair_sse41_rows4(float *yg,float *yu,const float *x, + const uint8_t *qg,const float *sg, + const uint8_t *qu,const float *su, + int S,int I,int O,int gs,int rb, + int ng,int o4){ + #pragma omp parallel for schedule(static) + for(int o=0;oI) end=I; + __m128 scg=colibri_sse41_f32_rows4(sg,ng,o,g); + __m128 scu=colibri_sse41_f32_rows4(su,ng,o,g); int i=base; + for(;i+1>1,&gl,&gh); + colibri_sse41_i4_rows4(qu,rb,o,i>>1,&ul,&uh); + __m128 x0=_mm_set1_ps(xs[i]),x1=_mm_set1_ps(xs[i+1]); + __m128 gp=_mm_add_ps(_mm_mul_ps(x0,gl),_mm_mul_ps(x1,gh)); + __m128 up=_mm_add_ps(_mm_mul_ps(x0,ul),_mm_mul_ps(x1,uh)); + ag=_mm_add_ps(ag,_mm_mul_ps(gp,scg)); + au=_mm_add_ps(au,_mm_mul_ps(up,scu)); + } + if(i>1,&gl,&gh); + colibri_sse41_i4_rows4(qu,rb,o,i>>1,&ul,&uh); (void)gh; (void)uh; + __m128 xi=_mm_set1_ps(xs[i]); + ag=_mm_add_ps(ag,_mm_mul_ps(_mm_mul_ps(xi,gl),scg)); + au=_mm_add_ps(au,_mm_mul_ps(_mm_mul_ps(xi,ul),scu)); + } + } + colibri_sse41_storeu_ps(yg+(int64_t)s*O+o,ag); + colibri_sse41_storeu_ps(yu+(int64_t)s*O+o,au); + } + } +} +#endif /* __SSE4_1__ && !__AVX2__ */ + +#endif /* __SSE2__ */ +#endif /* COLIBRI_SSE41_KERNELS_H */ diff --git a/c/st.h b/c/st.h index 8ad5ffd3b..286cfd754 100644 --- a/c/st.h +++ b/c/st.h @@ -138,6 +138,121 @@ static inline float f16_to_f32(uint16_t h) { float f; memcpy(&f, &u, 4); return f; } +/* ---- bulk BF16/F16 -> F32, AVX2/SSE4.1/scalar tiers -------------------- + * st_read_f32/st_read_slice_f32 convert whole tensors (up to the embed/ + * lm_head matrix, vocab*hidden elements) through bf16_to_f32/f16_to_f32 one + * halfword at a time; that loop is pure overhead once the read syscall is + * off the critical path. The tiers below batch it, same idea and layering + * as gsgemv.h's AVX2/SSE4.1 kernels (immintrin.h only under the ISA guard, + * sse41_kernels.h not needed here -- no FMA, no shared load helper to reuse). + * + * BF16 -> F32 is an exact zero-pad widening for every bit pattern (BF16 + * shares F32's 8-bit exponent field, so there is no special-casing -- + * zero, normal, subnormal, inf, NaN all take the same `<<16`), so the + * vectorized tiers cannot disagree with the scalar reference. + * + * F16 -> F32's zero/normal/inf-NaN classes are each the same closed-form + * bit algebra as the scalar reference above, just run on several lanes at + * once -- no reassociation, nothing to round, so still bit-exact. True + * subnormals (exp==0, man!=0) need the scalar reference's shift-to-normalize + * loop, which does not vectorize; those (rare in real model weights) fall + * back to f16_to_f32 per element. Exhaustive verification (all 65536 + * patterns per format) lives in tests/test_st_f16_bf16_simd.c. */ +#if defined(__AVX2__) || defined(__SSE4_1__) +#include +#endif + +#if defined(__AVX2__) +static void bf16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) { + int64_t i = 0; + for (; i + 8 <= n; i += 8) { + __m128i h = _mm_loadu_si128((const __m128i*)(src + i)); + __m256i w = _mm256_slli_epi32(_mm256_cvtepu16_epi32(h), 16); + _mm256_storeu_ps(dst + i, _mm256_castsi256_ps(w)); + } + for (; i < n; i++) dst[i] = bf16_to_f32(src[i]); +} +static void f16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) { + int64_t i = 0; + const __m256i vsign_mask = _mm256_set1_epi32(0x8000); + const __m256i vexp_mask = _mm256_set1_epi32(0x1F); + const __m256i vman_mask = _mm256_set1_epi32(0x3FF); + const __m256i v112 = _mm256_set1_epi32(112); + const __m256i vinfnan_e = _mm256_set1_epi32(0x7F800000); + const __m256i vzero = _mm256_setzero_si256(); + const __m256i v31 = _mm256_set1_epi32(31); + for (; i + 8 <= n; i += 8) { + __m128i h16 = _mm_loadu_si128((const __m128i*)(src + i)); + __m256i h = _mm256_cvtepu16_epi32(h16); + __m256i sign = _mm256_slli_epi32(_mm256_and_si256(h, vsign_mask), 16); + __m256i exp = _mm256_and_si256(_mm256_srli_epi32(h, 10), vexp_mask); + __m256i man = _mm256_and_si256(h, vman_mask); + __m256i normal_u = _mm256_or_si256(sign, _mm256_or_si256( + _mm256_slli_epi32(_mm256_add_epi32(exp, v112), 23), _mm256_slli_epi32(man, 13))); + __m256i infnan_u = _mm256_or_si256(sign, _mm256_or_si256(vinfnan_e, _mm256_slli_epi32(man, 13))); + __m256i exp_is_zero = _mm256_cmpeq_epi32(exp, vzero); + __m256i man_is_zero = _mm256_cmpeq_epi32(man, vzero); + __m256i is_zero = _mm256_and_si256(exp_is_zero, man_is_zero); + __m256i is_subnorm = _mm256_andnot_si256(man_is_zero, exp_is_zero); + __m256i is_infnan = _mm256_cmpeq_epi32(exp, v31); + __m256i result = _mm256_blendv_epi8(normal_u, sign, is_zero); + result = _mm256_blendv_epi8(result, infnan_u, is_infnan); + _mm256_storeu_ps(dst + i, _mm256_castsi256_ps(result)); + int m = _mm256_movemask_ps(_mm256_castsi256_ps(is_subnorm)); + if (m) for (int k = 0; k < 8; k++) if ((m >> k) & 1) dst[i+k] = f16_to_f32(src[i+k]); + } + for (; i < n; i++) dst[i] = f16_to_f32(src[i]); +} +#elif defined(__SSE4_1__) +static void bf16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) { + int64_t i = 0; + for (; i + 4 <= n; i += 4) { + __m128i h = _mm_loadl_epi64((const __m128i*)(src + i)); + __m128i w = _mm_slli_epi32(_mm_cvtepu16_epi32(h), 16); + _mm_storeu_ps(dst + i, _mm_castsi128_ps(w)); + } + for (; i < n; i++) dst[i] = bf16_to_f32(src[i]); +} +static void f16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) { + int64_t i = 0; + const __m128i vsign_mask = _mm_set1_epi32(0x8000); + const __m128i vexp_mask = _mm_set1_epi32(0x1F); + const __m128i vman_mask = _mm_set1_epi32(0x3FF); + const __m128i v112 = _mm_set1_epi32(112); + const __m128i vinfnan_e = _mm_set1_epi32(0x7F800000); + const __m128i vzero = _mm_setzero_si128(); + const __m128i v31 = _mm_set1_epi32(31); + for (; i + 4 <= n; i += 4) { + __m128i h16 = _mm_loadl_epi64((const __m128i*)(src + i)); + __m128i h = _mm_cvtepu16_epi32(h16); + __m128i sign = _mm_slli_epi32(_mm_and_si128(h, vsign_mask), 16); + __m128i exp = _mm_and_si128(_mm_srli_epi32(h, 10), vexp_mask); + __m128i man = _mm_and_si128(h, vman_mask); + __m128i normal_u = _mm_or_si128(sign, _mm_or_si128( + _mm_slli_epi32(_mm_add_epi32(exp, v112), 23), _mm_slli_epi32(man, 13))); + __m128i infnan_u = _mm_or_si128(sign, _mm_or_si128(vinfnan_e, _mm_slli_epi32(man, 13))); + __m128i exp_is_zero = _mm_cmpeq_epi32(exp, vzero); + __m128i man_is_zero = _mm_cmpeq_epi32(man, vzero); + __m128i is_zero = _mm_and_si128(exp_is_zero, man_is_zero); + __m128i is_subnorm = _mm_andnot_si128(man_is_zero, exp_is_zero); + __m128i is_infnan = _mm_cmpeq_epi32(exp, v31); + __m128i result = _mm_blendv_epi8(normal_u, sign, is_zero); + result = _mm_blendv_epi8(result, infnan_u, is_infnan); + _mm_storeu_ps(dst + i, _mm_castsi128_ps(result)); + int m = _mm_movemask_ps(_mm_castsi128_ps(is_subnorm)); + if (m) for (int k = 0; k < 4; k++) if ((m >> k) & 1) dst[i+k] = f16_to_f32(src[i+k]); + } + for (; i < n; i++) dst[i] = f16_to_f32(src[i]); +} +#else +static void bf16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) { + for (int64_t i = 0; i < n; i++) dst[i] = bf16_to_f32(src[i]); +} +static void f16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) { + for (int64_t i = 0; i < n; i++) dst[i] = f16_to_f32(src[i]); +} +#endif + static int st_open_fd(shards *S, const char *path) { for (int i = 0; i < S->nfd; i++) if (!strcmp(S->paths[i], path)) return S->fds[i]; int fd = open(path, COMPAT_O_RDONLY); @@ -449,7 +564,9 @@ static const char *st_basename(const char *p) { static void st_index_load(st_index *ix, const char *dir) { if (ix->tried) return; ix->tried = 1; - char path[1200]; snprintf(path, sizeof(path), "%s/model.safetensors.index.json", dir); + char path[1200]; + int written = snprintf(path, sizeof(path), "%s/model.safetensors.index.json", dir); + if (written < 0 || (size_t)written >= sizeof(path)) return; FILE *f = fopen(path, "rb"); if (!f) return; fseek(f, 0, SEEK_END); long size = ftell(f); fseek(f, 0, SEEK_SET); @@ -954,9 +1071,9 @@ static int64_t st_read_f32(shards *S, const char *name, float *out, int drop) { if (t->dtype == 2) { memcpy(out, raw, t->nbytes); } else if (t->dtype == 0) { - uint16_t *p = (uint16_t *)raw; for (int64_t i = 0; i < t->numel; i++) out[i] = bf16_to_f32(p[i]); + bf16_to_f32_bulk((uint16_t *)raw, out, t->numel); } else { - uint16_t *p = (uint16_t *)raw; for (int64_t i = 0; i < t->numel; i++) out[i] = f16_to_f32(p[i]); + f16_to_f32_bulk((uint16_t *)raw, out, t->numel); } free(raw); if (drop) posix_fadvise(t->fd, t->off, t->nbytes, POSIX_FADV_DONTNEED); @@ -1342,8 +1459,8 @@ static void st_read_slice_f32(shards *S, const char *name, int64_t elem_off, int if (nb) st_pread_full(t->fd, raw, nb, boff, "pread slice"); /* dev #331: chunked + EINTR + honest short-read */ if (nb) { if (t->dtype == 2) memcpy(out, raw, (size_t)nb); - else if (t->dtype == 0) { uint16_t *p = raw; for (int64_t i = 0; i < n_elems; i++) out[i] = bf16_to_f32(p[i]); } - else { uint16_t *p = raw; for (int64_t i = 0; i < n_elems; i++) out[i] = f16_to_f32(p[i]); } + else if (t->dtype == 0) bf16_to_f32_bulk((uint16_t *)raw, out, n_elems); + else f16_to_f32_bulk((uint16_t *)raw, out, n_elems); } free(raw); if (drop && nb) posix_fadvise(t->fd, boff, nb, POSIX_FADV_DONTNEED); diff --git a/c/tests/bench_cuda_resident_batch.cu b/c/tests/bench_cuda_resident_batch.cu new file mode 100644 index 000000000..2f911fbca --- /dev/null +++ b/c/tests/bench_cuda_resident_batch.cu @@ -0,0 +1,80 @@ +/* Bounded resident-int8 projection microbenchmark; no model files required. + * Times synchronous host API calls (copies + compute), excluding upload. + * Compare S one-row calls with one S-row call on identical resident weights. + * This is not full-model prefill throughput or a CPU/GPU comparison. */ +#include "../backend_cuda.h" +#include +#include +#include +#include +#include + +static double median(std::vector values) { + std::sort(values.begin(),values.end()); + return values[values.size()/2]; +} + +static bool measure(int I,int O,int S,int device) { + std::vector q((size_t)I*O); + std::vector scale(O), x((size_t)S*I), serial((size_t)S*O), batch(serial.size()); + for(size_t i=0;i times[2], ratios; + for(int rep=0;rep<9 && ok;rep++) { + double pair[2]={}; + for(int arm=0;arm<2 && ok;arm++) { + int mode=(rep+arm)%2; + auto start=std::chrono::steady_clock::now(); + ok=run(mode!=0); + pair[mode]=std::chrono::duration(std::chrono::steady_clock::now()-start).count(); + times[mode].push_back(pair[mode]); + } + if(ok) ratios.push_back(pair[0]/pair[1]); + for(size_t i=0;i1e-5*(1+std::fabs(reference))) ok=false; + } + coli_cuda_tensor_free(tensor); + if(!ok){std::fprintf(stderr,"FAIL: resident batch I=%d O=%d S=%d\n",I,O,S);return false;} + std::printf("{\"input\":%d,\"output\":%d,\"rows\":%d,\"pairs\":9," + "\"serial_median_ms\":%.6f,\"batch_median_ms\":%.6f," + "\"paired_speedup_median\":%.6f,\"cpu_sample_max_abs_error\":%.9g," + "\"serial_ms\":[",I,O,S,median(times[0]),median(times[1]),median(ratios),max_error); + for(size_t i=0;i #include -#include +#if !defined(__HIPCC__) +#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */ +#endif #include #include diff --git a/c/tests/bench_dsv4_mxfp8.cu b/c/tests/bench_dsv4_mxfp8.cu index 209433ee4..c072719a2 100644 --- a/c/tests/bench_dsv4_mxfp8.cu +++ b/c/tests/bench_dsv4_mxfp8.cu @@ -1,4 +1,6 @@ -#include +#if !defined(__HIPCC__) +#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */ +#endif #include #include #include diff --git a/c/tests/bench_i4p_gidot.c b/c/tests/bench_i4p_gidot.c new file mode 100644 index 000000000..4843a0703 --- /dev/null +++ b/c/tests/bench_i4p_gidot.c @@ -0,0 +1,135 @@ +/* Microbenchmark: per-row vs multi-row K1b grouped planar IDOT (quant.h). + * NOT a unit test -- test_int_kernel_exact.c proves correctness/bit-equality. + * + * This measures the multi-row claim behind the fmt=4 tile work: the per-row + * kernel re-loads and re-masks every weight block once per activation row, so + * batched calls (prefill batch-union rows, the serve mux's decode batch) pay + * S times the weight traffic. The 1x4 tile (and AMX on Sapphire Rapids) pays + * it once per tile. The OLD kernel below is a verbatim copy of the pre-tile + * body; the NEW one is the real dispatcher, so on an AMX host this also + * benches the tile-unit path (AMX_S_MIN gates it; force with AMX_S_MIN=2). + * + * Run: make tests/bench_i4p_gidot ARCH=native && ./tests/bench_i4p_gidot + * (not in TEST_BINS -- not a gate) */ +#define main coli_glm_main_unused +#include "../colibri.c" +#undef main +#include +#include + +static uint32_t rs=0x2545F491u; +static uint32_t xr(void){ rs^=rs<<13; rs^=rs>>17; rs^=rs<<5; return rs; } + +/* ---- OLD kernel: verbatim copy of the pre-tile per-row body ---- */ +static void gidot_old(float *y, const int8_t *xq, const float *sx, + const int32_t *xsg, const uint8_t *q4, + const float *scale, int S, int I, int O, int gs){ + int rb=(I+1)/2, ng=(I+gs-1)/gs, bpg=gs/64; + #pragma omp parallel for schedule(static) + for(int o=0;o>1); + const int8_t *xb=xr2+base; +#if defined(coli_dpbusd256) + const __m256i m4=_mm256_set1_epi8(0x0F); + __m256i bb=_mm256_loadu_si256((const __m256i*)blk); + __m256i acc=_mm256_setzero_si256(); + acc=coli_dpbusd256(acc,_mm256_and_si256(bb,m4), + _mm256_loadu_si256((const __m256i*)xb)); + acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(bb,4),m4), + _mm256_loadu_si256((const __m256i*)(xb+32))); + d+=hsum256_i32(acc); +#elif defined(__AVX2__) + const __m256i m4=_mm256_set1_epi8(0x0F); + const __m256i ones=_mm256_set1_epi16(1); + __m256i bb=_mm256_loadu_si256((const __m256i*)blk); + __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(bb,m4), + _mm256_loadu_si256((const __m256i*)xb)); + __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(bb,4),m4), + _mm256_loadu_si256((const __m256i*)(xb+32))); + __m256i acc=_mm256_add_epi32(_mm256_madd_epi16(p0,ones), + _mm256_madd_epi16(p1,ones)); + d+=hsum256_i32(acc); +#else + for(int k=0;k<32;k++){ + d+=(int32_t)(blk[k]&0xF)*xb[k]; + d+=(int32_t)(blk[k]>>4)*xb[k+32]; + } +#endif + } + a=fmaf((float)(d-8*xg[g]),scl[g],a); + } + if(g*gs>1]; + d+=(int32_t)((i&1)?(byte>>4):(byte&0xF))*xr2[i]; + } + a=fmaf((float)(d-8*xg[g]),scl[g],a); + } + y[(int64_t)s*O+o]=a*sx[s]; + } + } +} + +#define O_DIM 2048 +#define I_DIM 2048 +#define GS 64 +#define REPS 400 +static int cmp_d(const void*a,const void*b){ double x=*(const double*)a,y=*(const double*)b; return xy?1:0; } + +int main(void){ + int rb=(I_DIM+1)/2, ng=(I_DIM+GS-1)/GS; + int Ss[]={1,4,8,16}; + uint8_t *q4=malloc((size_t)O_DIM*rb); + float *sc=malloc((size_t)O_DIM*ng*sizeof(float)); + for(size_t i=0;i<(size_t)O_DIM*rb;i++) q4[i]=(uint8_t)(xr()&0xFF); + planarize_i4(q4,O_DIM,I_DIM); + for(size_t i=0;i<(size_t)O_DIM*ng;i++) sc[i]=0.0005f+(xr()%911)*1e-6f; + + int Smax=16; + int8_t *xq=malloc((size_t)Smax*I_DIM); + float *sx=malloc(Smax*sizeof(float)); + int32_t *xsg=malloc((size_t)Smax*ng*sizeof(int32_t)); + for(size_t i=0;i<(size_t)Smax*I_DIM;i++){ int v=(int)(xr()%255)-127; xq[i]=(int8_t)v; } + for(int s=0;s +#include +#include +#include +#include + +enum { HIDDEN = 4096, OUT = 4096, N_MATRICES = 48 }; + +static long vmhwm_kb(void) { + FILE *f = fopen("/proc/self/status", "r"); + if (!f) return -1; + char line[256]; long kb = -1; + while (fgets(line, sizeof line, f)) { + if (!strncmp(line, "VmHWM:", 6)) { sscanf(line + 6, "%ld", &kb); break; } + } + fclose(f); + return kb; +} + +static void fill(float *w, int64_t n, int salt) { + for (int64_t i = 0; i < n; i++) w[i] = (float)(((i * 2654435761u + salt) % 2003) - 1000) * 0.001f; +} + +static void quantize_row_major(const float *w, int I, int O, int8_t *q, float *sc) { + for (int o = 0; o < O; o++) { + const float *r = w + (int64_t)o * I; float am = 0.f; + for (int i = 0; i < I; i++) { float a = fabsf(r[i]); if (a > am) am = a; } + float s = am > 1e-12f ? am / 127.f : 1.f; sc[o] = s; float inv = 1.f / s; + int8_t *d = q + (int64_t)o * I; + for (int i = 0; i < I; i++) { int v = (int)lrintf(r[i] * inv); if (v > 127) v = 127; if (v < -127) v = -127; d[i] = (int8_t)v; } + } +} + +static void run_old(void) { + int64_t n = (int64_t)HIDDEN * OUT; + float **f32s = malloc(N_MATRICES * sizeof(float*)); + /* pass 1: "load" every matrix's f32 copy (as model_init_range used to, + * for every dense matrix in the whole model, before any quantization) */ + for (int m = 0; m < N_MATRICES; m++) { + f32s[m] = malloc((size_t)n * sizeof(float)); + fill(f32s[m], n, m); + } + printf("after loading all %d f32 matrices: VmHWM=%ld MiB\n", N_MATRICES, vmhwm_kb() / 1024); + /* pass 2: quantize every matrix (as the old post-hoc qdw_register loop + * did), freeing each f32 copy only after ALL quantization is done */ + int8_t **qs = malloc(N_MATRICES * sizeof(int8_t*)); + float **scs = malloc(N_MATRICES * sizeof(float*)); + for (int m = 0; m < N_MATRICES; m++) { + qs[m] = malloc((size_t)n); + scs[m] = malloc((size_t)OUT * sizeof(float)); + quantize_row_major(f32s[m], HIDDEN, OUT, qs[m], scs[m]); + } + printf("after quantizing all (pre-free): VmHWM=%ld MiB\n", vmhwm_kb() / 1024); + for (int m = 0; m < N_MATRICES; m++) free(f32s[m]); + printf("[old pattern] peak VmHWM=%ld MiB\n", vmhwm_kb() / 1024); +} + +static void run_new(void) { + int64_t n = (int64_t)HIDDEN * OUT; + int8_t **qs = malloc(N_MATRICES * sizeof(int8_t*)); + float **scs = malloc(N_MATRICES * sizeof(float*)); + /* load_tq: read one matrix's f32 copy, quantize it, free it, next matrix */ + for (int m = 0; m < N_MATRICES; m++) { + float *w = malloc((size_t)n * sizeof(float)); + fill(w, n, m); + qs[m] = malloc((size_t)n); + scs[m] = malloc((size_t)OUT * sizeof(float)); + quantize_row_major(w, HIDDEN, OUT, qs[m], scs[m]); + free(w); + } + printf("[new pattern] peak VmHWM=%ld MiB\n", vmhwm_kb() / 1024); +} + +int main(int argc, char **argv) { + if (argc != 2 || (strcmp(argv[1], "old") && strcmp(argv[1], "new"))) { + fprintf(stderr, "usage: %s old|new\n", argv[0]); return 2; + } + printf("N_MATRICES=%d each %dx%d f32 (%.1f MiB) -- %s pattern\n", + N_MATRICES, HIDDEN, OUT, (double)HIDDEN * OUT * 4 / 1048576.0, argv[1]); + if (!strcmp(argv[1], "old")) run_old(); else run_new(); + return 0; +} diff --git a/c/tests/bench_qwen36_dense_batch.c b/c/tests/bench_qwen36_dense_batch.c index 63f0fc224..36e19e55a 100644 --- a/c/tests/bench_qwen36_dense_batch.c +++ b/c/tests/bench_qwen36_dense_batch.c @@ -39,13 +39,13 @@ static double shared_run(Model *m,Layer *l,const float *x,const float *seed, static int shared_benchmark(void) { Model m;memset(&m,0,sizeof(m));m.c.hidden=I;m.c.shared_inter=O; Layer l;memset(&l,0,sizeof(l)); - l.sh_g=falloc((int64_t)O*I);l.sh_u=falloc((int64_t)O*I); - l.sh_d=falloc((int64_t)I*O);l.sh_gate=falloc(I); - for(int64_t i=0;i<(int64_t)O*I;i++){l.sh_g[i]=value(i,2);l.sh_u[i]=value(i,3);} - for(int64_t i=0;i<(int64_t)I*O;i++)l.sh_d[i]=value(i,4); + l.sh_g.w=falloc((int64_t)O*I);l.sh_u.w=falloc((int64_t)O*I); + l.sh_d.w=falloc((int64_t)I*O);l.sh_gate=falloc(I); + for(int64_t i=0;i<(int64_t)O*I;i++){((float*)l.sh_g.w)[i]=value(i,2);((float*)l.sh_u.w)[i]=value(i,3);} + for(int64_t i=0;i<(int64_t)I*O;i++)((float*)l.sh_d.w)[i]=value(i,4); for(int i=0;i 3\n",ts,tb,ts/tb,S*3); - for(int i=0;i +#include +#include + +enum { N = 4 * 1024 * 1024, REPS = 5 }; + +static double now_s(void) { + struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); + return (double)ts.tv_sec + (double)ts.tv_nsec * 1e-9; +} + +static double scalar_bf16(const uint16_t *src, float *dst, int64_t n) { + double t0 = now_s(); + for (int64_t i = 0; i < n; i++) dst[i] = bf16_to_f32(src[i]); + return now_s() - t0; +} +static double scalar_f16(const uint16_t *src, float *dst, int64_t n) { + double t0 = now_s(); + for (int64_t i = 0; i < n; i++) dst[i] = f16_to_f32(src[i]); + return now_s() - t0; +} +static double bulk_bf16(const uint16_t *src, float *dst, int64_t n) { + double t0 = now_s(); bf16_to_f32_bulk(src, dst, n); return now_s() - t0; +} +static double bulk_f16(const uint16_t *src, float *dst, int64_t n) { + double t0 = now_s(); f16_to_f32_bulk(src, dst, n); return now_s() - t0; +} + +int main(void) { + uint16_t *src = malloc((size_t)N * sizeof(uint16_t)); + float *a = malloc((size_t)N * sizeof(float)); + float *b = malloc((size_t)N * sizeof(float)); + if (!src || !a || !b) { fprintf(stderr, "OOM\n"); return 2; } + /* representative mix: mostly normal-range values (real weight magnitudes), + * a scattering of exact zeros, occasional subnormal/inf/nan bit patterns -- + * not the uniform all-65536-once sweep the exactness test uses. */ + for (int64_t i = 0; i < N; i++) { + int64_t m = i % 97; + if (m == 0) src[i] = 0; + else if (m == 1) src[i] = 0x7C00; /* +inf */ + else if (m == 2) src[i] = 0x03FF; /* max subnormal */ + else src[i] = (uint16_t)((i * 2654435761u) & 0x7BFF); + } + + printf("st f16/bf16 simd bench: N=%d elements (%.1f MiB src)\n", N, N * sizeof(uint16_t) / 1048576.0); + + double ts = 0, tb = 0; + for (int r = 0; r < REPS; r++) { + if (r & 1) { tb += bulk_bf16(src, b, N); ts += scalar_bf16(src, a, N); } + else { ts += scalar_bf16(src, a, N); tb += bulk_bf16(src, b, N); } + } + printf("bf16->f32: scalar %.6f s bulk %.6f s speedup %.2fx\n", ts / REPS, tb / REPS, ts / tb); + + ts = 0; tb = 0; + for (int r = 0; r < REPS; r++) { + if (r & 1) { tb += bulk_f16(src, b, N); ts += scalar_f16(src, a, N); } + else { ts += scalar_f16(src, a, N); tb += bulk_f16(src, b, N); } + } + printf("f16->f32: scalar %.6f s bulk %.6f s speedup %.2fx\n", ts / REPS, tb / REPS, ts / tb); + + free(src); free(a); free(b); + return 0; +} diff --git a/c/tests/check_data_logprob_gaps.py b/c/tests/check_data_logprob_gaps.py index 0abb41655..767647f7e 100644 --- a/c/tests/check_data_logprob_gaps.py +++ b/c/tests/check_data_logprob_gaps.py @@ -4,18 +4,24 @@ Asserts, over a captured raw engine-stdout transcript, that EVERY generated DATA frame of an opted-in request carries the per-token numeric channel ("DATA [tid tlp]*k") -- accepted draft tokens included. -Speculatively-accepted draft tokens bypass the pick_tok call sites in the mux -loop (packet Fork 5's implementation risk), so a gap would show up as a -legacy 3-field DATA frame in the middle of an opted-in generation. +On today's engine, an accepted speculative-draft token is emitted through +the SAME mux_data() call as any other generated token (mux_spec_emit -> +mux_data, passing the same logits row and requested top-k the ordinary +emit sites use), so the gap this check hunts -- a legacy 3-field DATA +frame appearing mid-generation because a draft-accept path bypassed the +numeric-channel emit -- is NOT producible by the engine as shipped. This +check is retained as REGRESSION COVERAGE for that invariant (a future mux +change that adds a new emit call site without the tail would reintroduce +exactly this defect class), not as a currently-live defect hunt. -Run recipe (orchestrator; needs a real model -- CPU or CUDA build): +Run recipe (orchestrator; needs a real model -- CPU or explicit-CUDA build): # single KV slot + model drafts live = the speculative serve regime cd c && make glm SERVE=1 SERVE_BATCH=1 KV_SLOTS=1 DRAFT=2 CTX=4096 \ - ./colibri < submit.txt | tee engine_stdout.raw + ./colibri < submit.raw | tee engine_stdout.raw - where submit.txt contains one opted-in generation request, e.g. (prompt + where submit.raw contains one opted-in generation request, e.g. (prompt "Hello" = 5 payload bytes, 64 new tokens, greedy, logprobs top-5): SUBMIT 7 0 5 64 0 1 0 logprobs=5 @@ -27,109 +33,654 @@ stderr log ("[MTP] ..." / spec acceptance lines) so the run genuinely exercised the draft-accept emit path rather than trivially passing. -Then: +A raw capture taken this way is a plain pipeline: it is not itself hash-bound +to the binary, container, input, and environment that produced it, and +NOTHING in this module binds it either -- the preamble gate below checks +only that the BANNER/LOADED text is a well-formed, self-consistent record of +what the engine printed about itself; it does not verify that text against +an independently computed binary or container digest. Whichever capture step +feeds this checker, prefer one that additionally binds and retains that +provenance (distinct raw stdout/stderr, the direct engine exit status, and a +hash over the binary/container/input/environment/protocol payloads) over a +bare pipeline or a lone DATA/HIT record, which is not decisive evidence on +its own. - python3 tests/check_data_logprob_gaps.py engine_stdout.raw --id 7 --topk 5 +Then, against the resulting transcript, the numeric-channel check runs: + + python3 tests/check_data_logprob_gaps.py engine_stdout.raw \ + --id 7 --topk 5 --vocab 154880 Checks performed: - - every DATA frame for --id has exactly 5 + 2k header fields, a parseable - float lp, and k == --topk (k may be < topk only if vocab < topk). - A non-finite lp ("nan"/"inf": degenerate logits, e.g. an all -inf row - after grammar masking) counts as PRESENT -- the channel carried a value, - which is exactly what this audit is for -- and is FLAGGED in the output - without failing the check (whether the engine should serialize such rows - differently is a U7b server-side register question, not a gap); - - ECHO frames (if the request also echoed) cover contiguous positions - 0..P-1, position 0 carrying "nan 0"; + - the transcript's capture shape (full-process / ready-suffix / request- + only) is named by capture_mode() and enforced explicitly: the leading + line must be a recognized global preamble record or a legitimate + request-frame kind, and a transcript with any global record present + but not correctly led (a "invalid" capture per capture_mode()) fails + loud -- no transcript reaches the rest of the checks unclassified; + - the startup BANNER/LOADED preamble, where present, parses exactly + (the engine_evidence grammar); an unparsed preamble is a named + failure quoting the offending line, never a silent pass; + - exactly one ACCEPT and one DONE exist for --id, DONE's emitted count + equals the positive DATA count, and no targeted ERROR or post-DONE + frame exists; + - every DATA/ECHO numeric tail has exactly the advertised fields and + min(--topk,--vocab) unique token ids in range. Each numeric token + (target logprob and every top-k logprob) must be spelled as the + fixed six-decimal form dev's engine prints today ("fixed6", e.g. + "-0.300000"), the %.17g form an earlier engine build printed + ("c17g", e.g. "-2.7000000000000002"), or an exact nan/inf/-inf + spelling -- whichever form parses -- and the VALUE that spelling + carries must then be finite and non-positive; a nan/inf/-inf + spelling always fails that value check, by construction, since none + of those values are finite. That spelling is recognized at all only + so a non-finite token gets one specific, named rejection ("target + logprob is not finite/non-positive", quoting the exact token and + byte offset) instead of a generic "matches neither form" failure -- + it is not an accepted value, just a diagnosed one. On today's engine + (colibri.c) this is also not a normal, tolerable outcome to shrug + off: grammar-constrained decoding (GRAMMAR=/SCHEMA=, method F) + proposes speculative draft tokens verified against the model's own + unmasked distribution and never zeroes or -inf's any vocabulary + logit, so a non-finite emitted logprob can only reach the wire + through the total-vocabulary numerical-collapse fallback documented + at sample.h's dist_build()/argmax_v() (every logit in the row NaN or + -inf) -- which those functions' own comments and stderr warning + label "a numerical blow-up upstream", not a degenerate-but-valid + row. Failing loud on it here names a real engine bug instead of + quietly passing it through. + A token that happens to satisfy BOTH finite spellings exactly + (e.g. '%.17g' % -1.234567 == '-1.234567', itself also an exact + six-decimal spelling) is + "ambiguous": there is no way to tell, from the token alone, which + engine format produced it, so no per-frame or per-transcript + "consistent format" rule is enforced -- one was tried and had to be + dropped (see the module history) because it produced false rejections + on real %.17g transcripts purely from this ambiguity. The observed + form counts (fixed6-only / c17g-only / ambiguous / special) are + reported in the summary line as INFORMATION ONLY; they never affect + the verdict; + - ACCEPT's canonical prompt length P has exactly P ECHO frames at + positions 0..P-1, position 0 carrying "nan 0"; - payload framing (n bytes + newline) stays byte-exact throughout, so a - single malformed frame cannot hide by desynchronizing the parse. + single malformed frame cannot hide by desynchronizing the parse; every + reported problem cites the byte offset of the offending frame's own + header line (not the frame that follows it). Exit 0 = no gaps; non-zero = at least one gap/malformed frame (listed). """ import argparse +import collections import math +import os +import re import sys +_TOOLS_DIR = os.path.join(os.path.dirname(os.path.dirname( + os.path.abspath(__file__))), "tools") +if _TOOLS_DIR not in sys.path: + sys.path.insert(0, _TOOLS_DIR) +from engine_evidence import PreambleError, parse_engine_preamble + + +_INT32_MAX = 2**31 - 1 +_UINT64_MAX = 2**64 - 1 +_UINT_RE = re.compile(rb"(?:0|[1-9][0-9]*)") +_C17G_RE = re.compile( + rb"-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?" + rb"(?:e[+-](?:0[0-9]|[1-9][0-9]{1,2}))?") +_FIXED6_RE = re.compile(rb"-?(?:0|[1-9][0-9]*)\.[0-9]{6}") +_SPECIAL_TOKENS = {b"nan": math.nan, b"inf": math.inf, b"-inf": -math.inf} +_READY = b"\x01\x01READY\x01\x01" +_GLOBAL_KINDS = frozenset(( + b"BANNER", b"LOADED", b"READY", b"STAT", b"HWINFO", b"TIERS", + b"EMAP", b"HITS", b"PROF", +)) +_TARGETED_KINDS = frozenset(( + b"ACCEPT", b"DATA", b"ECHO", b"DONE", b"ERROR", +)) +_LOWER_HEX_RE = re.compile(rb"[0-9a-f]*") + + +def _uint(token, label, maximum=_INT32_MAX): + if not _UINT_RE.fullmatch(token): + raise ValueError(f"noncanonical {label}: {token!r}") + value = int(token.decode("ascii")) + if value > maximum: + raise ValueError(f"{label} exceeds {maximum}: {token!r}") + return value + + +def _c17g(token, label): + if not _C17G_RE.fullmatch(token): + raise ValueError(f"noncanonical {label} %.17g token: {token!r}") + value = float(token.decode("ascii")) + if not math.isfinite(value) or format(value, ".17g").encode("ascii") != token: + raise ValueError(f"not an exact finite {label} %.17g spelling: {token!r}") + return value + + +def _fixed6(token, label): + if not _FIXED6_RE.fullmatch(token): + raise ValueError(f"noncanonical {label} %.6f token: {token!r}") + value = float(token.decode("ascii")) + if not math.isfinite(value) or format(value, ".6f").encode("ascii") != token: + raise ValueError(f"not an exact finite {label} %.6f spelling: {token!r}") + return value + + +def _numeric_value(token, label): + """Classify and parse one engine numeric-tail token. + + Returns (value, form). "special" is an exact nan/inf/-inf spelling -- + libc's own %f/%g rendering of a non-finite double, which either wire + format's snprintf call can emit identically. Otherwise the token is + checked against BOTH finite grammars independently (never short- + circuited): the fixed six-decimal form dev's engine prints today + ("fixed6") and the %.17g form an earlier engine build printed + ("c17g"). A token can satisfy both -- '%.17g' % -1.234567 == + '-1.234567', which is ALSO the exact six-decimal spelling of that + same double -- and when it does, the form is "ambiguous": there is no + way to tell, from the token alone, which engine format produced it. + A token matching neither raises ValueError. + + An earlier version of this function tried fixed6 first and reported + it whenever fixed6 matched, silently hiding the ambiguous case; that + misclassified a large fraction of genuine %.17g tokens as fixed6 and + fed a since-removed "consistent form per frame" check false mixed- + form rejections on real transcripts. Classification is now purely + informational (see check()/main()) precisely because it cannot be + made unambiguous from the token alone. + """ + if token in _SPECIAL_TOKENS: + return _SPECIAL_TOKENS[token], "special" + value = None + is_fixed6 = is_c17g = False + try: + value = _fixed6(token, label) + is_fixed6 = True + except ValueError: + pass + try: + value = _c17g(token, label) + is_c17g = True + except ValueError: + pass + if is_fixed6 and is_c17g: + return value, "ambiguous" + if is_fixed6: + return value, "fixed6" + if is_c17g: + return value, "c17g" + raise ValueError( + f"{label} matches neither the fixed6 nor the c17g nor the " + f"nan/inf spelling: {token!r}") + + +def _fixed_metric(token, places, label, lower=0.0, upper=None): + pattern=rb"-?(?:0|[1-9][0-9]*)\."+rb"[0-9]{"+str(places).encode()+rb"}" + if not re.fullmatch(pattern,token): + raise ValueError(f"noncanonical {label}: {token!r}") + value=float(token.decode("ascii")) + if not math.isfinite(value) or valueupper): + raise ValueError(f"{label} outside [{lower},{upper}]: {token!r}") + return value + + +def _header_fields(line, byte_offset): + problems=[] + if not line: + return [],[f"blank protocol header at byte {byte_offset}"] + if (any(byte<0x20 or byte>0x7e for byte in line) or + line.startswith(b" ") or line.endswith(b" ") or b" " in line): + problems.append(f"noncanonical ASCII/space header at byte {byte_offset}: {line!r}") + return line.split(),problems + + +def _global_header(line, byte_offset): + """Validate one exact production mux-global line. + + Return None when the line is not a recognized global. Recognized globals + return a synthetic one-field marker so check() owns their process-level + lifecycle before any numeric token can collide with a request id. + """ + if (line.startswith(b"== GLM C engine") or + line.startswith(b"loaded in")): + kind = b"BANNER" if line.startswith(b"==") else b"LOADED" + problems = [] + try: + text = line.decode("ascii") + parsed = parse_engine_preamble(text) + if parsed is None or parsed["kind"].encode("ascii") != kind: + raise PreambleError("owned preamble kind mismatch") + except (UnicodeDecodeError, PreambleError) as exc: + problems.append( + f"malformed {kind.decode()} preamble at byte {byte_offset}: " + f"{exc}; {line!r}") + return [kind], problems + + if line == _READY: + return [b"READY"], [] + + kind = line.split(b" ", 1)[0] + if kind == b"READY": + return [b"READY"], [ + f"malformed READY global at byte {byte_offset}: expected {_READY!r}; {line!r}" + ] + if kind not in _GLOBAL_KINDS - {b"READY"}: + return None + problems = [] + try: + if kind == b"STAT": + fields = line.split(b" ") + if (len(fields) != 5 or fields[1:4] != [b"0", b"0.00", b"0.0"]): + raise ValueError("expected exact startup STAT fields") + _fixed_metric(fields[4], 2, "STAT RSS") + elif kind == b"HWINFO": + # The final CPU|GPU field is produced by two %s conversions. It is + # printable free text and may contain repeated spaces; the six + # numeric/structural prefixes remain exact single-space fields. + fields = line.split(b" ", 6) + if len(fields) != 7: + raise ValueError("expected six HWINFO prefixes and CPU|GPU tail") + _uint(fields[1], "HWINFO core count") + _fixed_metric(fields[2], 1, "HWINFO total RAM") + _fixed_metric(fields[3], 1, "HWINFO available RAM") + _uint(fields[4], "HWINFO GPU count") + _fixed_metric(fields[5], 1, "HWINFO total VRAM") + tail = fields[6] + if (tail.count(b"|") != 1 or + any(byte < 0x20 or byte > 0x7e for byte in tail)): + raise ValueError("HWINFO tail is not printable CPU|GPU text") + elif kind == b"TIERS": + fields = line.split(b" ") + if len(fields) != 6: + raise ValueError("expected five TIERS fields") + for index, label in enumerate(("VRAM", "RAM", "disk"), 1): + _uint(fields[index], f"TIERS {label} count") + _fixed_metric(fields[4], 2, "TIERS VRAM GB") + _fixed_metric(fields[5], 2, "TIERS RAM GB") + elif kind in (b"EMAP", b"HITS"): + fields = line.split(b" ") + if len(fields) != 4: + raise ValueError(f"expected three {kind.decode()} fields") + rows = _uint(fields[1], f"{kind.decode()} row count") + cols = _uint(fields[2], f"{kind.decode()} column count") + if cols < 1 or (kind == b"HITS" and rows < 1): + raise ValueError( + f"{kind.decode()} rows/columns outside producer domain") + payload = fields[3] + if not _LOWER_HEX_RE.fullmatch(payload): + raise ValueError(f"{kind.decode()} payload is not lowercase hex") + cells = rows * cols + expected = cells * 2 if kind == b"EMAP" else ((cells + 7) // 8) * 2 + if len(payload) != expected: + raise ValueError( + f"{kind.decode()} payload length {len(payload)} != {expected}") + if kind == b"EMAP": + for index in range(cells): + cell = int(payload[2 * index:2 * index + 2], 16) + tier, heat = cell >> 6, cell & 0x3f + if tier > 2: + raise ValueError( + f"EMAP cell {index} tier {tier} outside [0,2]") + if heat > 32: + raise ValueError( + f"EMAP cell {index} heat {heat} outside [0,32]") + elif cells & 7: + final = int(payload[-2:], 16) + used_mask = (1 << (cells & 7)) - 1 + if final & ~used_mask: + raise ValueError("HITS final byte has nonzero padding bits") + else: # PROF + fields = line.split(b" ") + if len(fields) != 10: + raise ValueError("expected nine PROF fields") + _fixed_metric(fields[1], 3, "PROF wall seconds") + _uint(fields[2], "PROF prompt count") + _uint(fields[3], "PROF completion count") + for index, label in enumerate( + ("disk", "wait", "matmul", "attention", "head"), 4): + _fixed_metric(fields[index], 3, f"PROF {label} seconds") + _uint(fields[9], "PROF forward count", _UINT64_MAX) + except (ValueError, IndexError) as exc: + problems.append( + f"malformed {kind.decode()} global at byte {byte_offset}: {exc}; {line!r}") + return [kind], problems + + def parse_frames(blob): - """Parse the serve-mux stdout byte stream into (header_fields, payload).""" + """Parse frames fail-closed, returning (frames, framing_problems). + + Each frame is (fields, payload, byte_offset), where byte_offset is the + offset of THIS frame's own header line -- captured before the cursor + advances past it, so every problem this module reports can cite the + byte offset of the frame it is actually complaining about, not the + frame that happens to follow it in the transcript. + """ frames = [] + problems = [] i = 0 n = len(blob) while i < n: + line_start = i j = blob.find(b"\n", i) if j < 0: + problems.append(f"truncated header at byte {line_start}") break line = blob[i:j] i = j + 1 - fields = line.split() + global_header = _global_header(line, line_start) + if global_header is None: + fields, header_problems = _header_fields(line, line_start) + else: + fields, header_problems = global_header + problems.extend(header_problems) if not fields: continue kind = fields[0] - if kind in (b"DATA", b"ECHO") and len(fields) >= 3: + if kind in (b"DATA", b"ECHO"): + if len(fields) < 3: + problems.append( + f"short {kind.decode(errors='replace')} header at " + f"byte {line_start}: {fields!r}") + break try: - size = int(fields[2]) + size = _uint(fields[2],"payload size") except ValueError: - frames.append((fields, None)) - continue + problems.append( + f"invalid payload size at byte {line_start}: {fields!r}") + break + if size < 0: + problems.append( + f"negative payload size at byte {line_start}: {fields!r}") + break + if size > n - i: + problems.append( + f"truncated payload at byte {line_start}: need {size} bytes") + break payload = blob[i:i + size] i += size if i < n and blob[i:i + 1] == b"\n": i += 1 else: - frames.append((fields, b"")) - continue - frames.append((fields, payload)) + problems.append( + f"missing payload terminator at byte {line_start} " + f"after {fields!r}") + break + frames.append((fields, payload, line_start)) else: - frames.append((fields, None)) - return frames + frames.append((fields, None, line_start)) + return frames, problems + +def capture_mode(frames): + """Name the explicit transcript shape without grading its lifecycle.""" + if not frames: + return "request-only" + first = frames[0][0][0] + if first == b"BANNER": + return "full-process" + if first == b"READY": + return "ready-suffix" + if any(fields[0] in _GLOBAL_KINDS for fields, _, _ in frames): + return "invalid" + return "request-only" -def check(frames, request_id, topk): - rid = str(request_id).encode() + +def _numeric_tail(fields, field_offset, expected_k, vocab, label, byte_offset): + """Validate one DATA/ECHO numeric tail starting at fields[field_offset]. + + Returns (problems, forms) where forms is the LIST of numeric-token + "form" tags (see _numeric_value) observed in this tail, in order -- + a list, not a set, so the caller can tally counts. No rule requires a + tail's forms to agree with one another: per-token classification is + inherently ambiguous (a token can be an exact spelling under both + finite grammars at once), so there is no reliable way to tell a + genuinely mixed-format tail from an entirely single-format one that + merely contains some ambiguous tokens -- see _numeric_value. + """ problems = [] - flags = [] + forms = [] + try: + lp, lp_form = _numeric_value(fields[field_offset], f"{label} target logprob") + k = _uint(fields[field_offset + 1], f"{label} top-k count", 32) + except (ValueError, IndexError): + return ([f"malformed {label} numeric fields at byte {byte_offset}: " + f"{fields!r}"], forms) + forms.append(lp_form) + if not math.isfinite(lp) or lp > 0.0: + problems.append( + f"{label} target logprob is not finite/non-positive at byte " + f"{byte_offset}: {fields!r}") + if k != expected_k: + problems.append( + f"{label} top-k {k} != expected {expected_k} at byte " + f"{byte_offset}: {fields!r}") + want_fields = field_offset + 2 + 2 * k + if len(fields) != want_fields: + return (problems + [f"{label} field count {len(fields)} != " + f"{want_fields} at byte {byte_offset}: {fields!r}"], + forms) + ids = [] + for idx in range(k): + try: + token_id = _uint(fields[field_offset + 2 + 2 * idx], + f"{label} token id", _INT32_MAX) + token_lp, token_form = _numeric_value( + fields[field_offset + 3 + 2 * idx], f"{label} token logprob") + except ValueError: + problems.append( + f"malformed {label} top-k pair {idx} at byte {byte_offset}: " + f"{fields!r}") + continue + forms.append(token_form) + if not 0 <= token_id < vocab: + problems.append( + f"{label} token id {token_id} outside [0,{vocab}) at byte " + f"{byte_offset}: {fields!r}") + if token_id in ids: + problems.append( + f"{label} duplicate token id {token_id} at byte " + f"{byte_offset}: {fields!r}") + ids.append(token_id) + if not math.isfinite(token_lp) or token_lp > 0.0: + problems.append( + f"{label} token {token_id} logprob is not finite/non-positive " + f"at byte {byte_offset}: {fields!r}") + return problems, forms + + +def check(frames, framing_problems, request_id, topk, vocab): + problems = list(framing_problems) + mode = capture_mode(frames) + if frames: + first_kind = frames[0][0][0] + if first_kind not in _GLOBAL_KINDS and first_kind not in _TARGETED_KINDS: + problems.append( + f"unrecognized frame opens the transcript at byte " + f"{frames[0][2]} (not a global preamble or a request " + f"frame): {frames[0][0]!r}") + if mode == "invalid": + problems.append( + "invalid capture mode: global records present without a " + "recognized BANNER/READY lead frame") + try: + rid = str(request_id).encode("ascii",errors="strict") + rid_value=_uint(rid,"requested id",_UINT64_MAX) + if rid_value < 1: + raise ValueError("requested id must be positive") + except (UnicodeEncodeError,ValueError) as exc: + return 0, 0, problems+[str(exc)], mode, frozenset() data_frames = 0 echo_positions = [] - for fields, payload in frames: - if len(fields) < 2 or fields[1] != rid: - continue + accepts = [] + dones = [] + accept_prompts = [] + done_values = [] + done_seen = False + data_seen = False + global_indices = {kind: [] for kind in _GLOBAL_KINDS} + expected_k = min(topk, vocab) + forms_used = [] + for frame_index, (fields, payload, offset) in enumerate(frames): kind = fields[0] + if kind in _GLOBAL_KINDS: + global_indices[kind].append(frame_index) + continue + if len(fields)<2: + continue + try: + frame_id=_uint(fields[1],"frame request id",_UINT64_MAX) + except ValueError as exc: + if kind in _TARGETED_KINDS: + problems.append(f"{exc} at byte {offset}") + continue + if frame_id!=rid_value: + continue + if done_seen: + problems.append( + f"frame for id {request_id} after DONE at byte {offset}: " + f"{fields!r}") + if kind == b"ACCEPT": + accepts.append(frame_index) + if len(fields) != 3: + problems.append( + f"malformed ACCEPT frame at byte {offset}: {fields!r}") + else: + try: + prompt=_uint(fields[2],"ACCEPT prompt length") + if prompt<1: + raise ValueError + except ValueError: + problems.append( + f"invalid ACCEPT prompt length at byte {offset}: " + f"{fields!r}") + else: + accept_prompts.append(prompt) + continue + if kind == b"ERROR": + problems.append( + f"target request returned ERROR at byte {offset}: {fields!r}") + continue + if kind == b"DONE": + dones.append(frame_index) + done_seen = True + if len(fields) != 9 or fields[2] != b"STAT": + problems.append( + f"malformed DONE frame at byte {offset}: {fields!r}") + done_values.append(None) + continue + try: + emitted=_uint(fields[3],"DONE emitted count") + tps=_fixed_metric(fields[4],2,"DONE tokens/second") + hit=_fixed_metric(fields[5],1,"DONE hit percentage",upper=100.0) + rss=_fixed_metric(fields[6],2,"DONE RSS") + prompt=_uint(fields[7],"DONE prompt-token count") + flag=_uint(fields[8],"DONE length_limited",1) + if prompt<1: + raise ValueError("DONE prompt-token count is not positive") + except ValueError: + problems.append( + f"malformed DONE stats at byte {offset}: {fields!r}") + done_values.append(None) + continue + done_values.append((emitted,tps,hit,rss,prompt,flag)) + continue if kind == b"DATA": + data_seen = True data_frames += 1 + if not accepts: + problems.append( + f"DATA before ACCEPT at byte {offset}: {fields!r}") + if payload is None: + problems.append( + f"DATA missing validated payload at byte {offset}: " + f"{fields!r}") if len(fields) == 3: - problems.append(f"GAP: legacy 3-field DATA frame #{data_frames}" - f" (payload {payload!r}) has no logprob") - continue - try: - lp = float(fields[3]) - k = int(fields[4]) - except (ValueError, IndexError): - problems.append(f"malformed DATA numeric fields: {fields!r}") + problems.append( + f"GAP: legacy 3-field DATA frame #{data_frames} at byte " + f"{offset} (payload {payload!r}) has no logprob") continue - if not math.isfinite(lp): - # PRESENT, not a gap: the channel carried a value; degenerate - # logits (an all -inf row, say) legitimately produce nan/inf. - # Flagged so a reviewer sees it; never fails the audit. - flags.append(f"non-finite lp in DATA frame #{data_frames}" - f" (present, flagged): {fields!r}") - if k != topk: - problems.append(f"DATA top-k {k} != requested {topk}: {fields!r}") - if len(fields) != 5 + 2 * k: - problems.append(f"DATA field count {len(fields)} != 5+2k: {fields!r}") + tail_problems, tail_forms = _numeric_tail( + fields, 3, expected_k, vocab, "DATA", offset) + problems.extend(tail_problems) + forms_used.extend(tail_forms) elif kind == b"ECHO": + if data_seen: + problems.append( + f"ECHO after DATA at byte {offset}: {fields!r}") + if not accepts: + problems.append( + f"ECHO before ACCEPT at byte {offset}: {fields!r}") + if payload is None: + problems.append( + f"ECHO missing validated payload at byte {offset}: " + f"{fields!r}") try: - pos = int(fields[3]) + pos = _uint(fields[3],"ECHO position") except (ValueError, IndexError): - problems.append(f"malformed ECHO frame: {fields!r}") + problems.append( + f"malformed ECHO frame at byte {offset}: {fields!r}") continue echo_positions.append(pos) - if pos == 0 and (fields[4] != b"nan" or fields[5] != b"0"): - problems.append(f"ECHO position 0 should carry 'nan 0': {fields!r}") - if pos > 0 and len(fields) < 6: - problems.append(f"short ECHO frame: {fields!r}") - if echo_positions and echo_positions != list(range(len(echo_positions))): - problems.append(f"ECHO positions not contiguous from 0: {echo_positions}") - return data_frames, len(echo_positions), problems, flags + if pos == 0: + if len(fields) != 6 or fields[4] != b"nan" or fields[5] != b"0": + problems.append( + f"ECHO position 0 should carry exactly 'nan 0' at " + f"byte {offset}: {fields!r}") + else: + tail_problems, tail_forms = _numeric_tail( + fields, 4, expected_k, vocab, "ECHO", offset) + problems.extend(tail_problems) + forms_used.extend(tail_forms) + else: + problems.append( + f"unknown targeted frame kind at byte {offset}: {fields!r}") + any_globals = any(global_indices[kind] for kind in _GLOBAL_KINDS) + if global_indices[b"BANNER"] or global_indices[b"LOADED"]: + required = ( + (b"BANNER", 0), (b"LOADED", 1), (b"READY", 2), (b"STAT", 3), + (b"HWINFO", 4), (b"TIERS", 5), (b"EMAP", 6), + ) + for owned_kind, expected_index in required: + found = global_indices[owned_kind] + if (owned_kind in (b"BANNER", b"LOADED", b"READY", b"STAT") and + found != [expected_index]): + problems.append( + f"expected one {owned_kind.decode()} as frame " + f"{expected_index}, found {found}") + elif (owned_kind not in (b"BANNER", b"LOADED", b"READY", b"STAT") and + (not found or found[0] != expected_index)): + problems.append( + f"expected startup {owned_kind.decode()} as frame " + f"{expected_index}, found {found}") + if len(frames) > 7 and frames[7][0][0] not in (b"ACCEPT", b"ERROR"): + problems.append( + f"unexpected full-process pre-request frame 7: {frames[7][0]!r}") + elif any_globals: + ready = global_indices[b"READY"] + stat = global_indices[b"STAT"] + if ready != [0]: + problems.append(f"expected one READY as frame 0, found {ready}") + if stat != [1]: + problems.append(f"expected one startup STAT as frame 1, found {stat}") + if len(accepts) != 1: + problems.append(f"expected exactly one ACCEPT, found {len(accepts)}") + elif len(accept_prompts)==1: + prompt=accept_prompts[0] + if echo_positions!=list(range(prompt)): + problems.append(f"ECHO positions {echo_positions} != required 0..{prompt-1}") + if len(dones) != 1: + problems.append(f"expected exactly one DONE, found {len(dones)}") + elif len(done_values)==1 and done_values[0] is not None: + emitted,_,_,_,done_prompt,_=done_values[0] + if emitted!=data_frames: + problems.append(f"DONE emitted {emitted} != observed DATA {data_frames}") + if len(accept_prompts)==1 and done_prompt!=accept_prompts[0]: + problems.append(f"DONE prompt count {done_prompt} != ACCEPT {accept_prompts[0]}") + if data_frames<=0: + problems.append("no DATA frames found for this id -- wrong id, or the run produced nothing") + return data_frames, len(echo_positions), problems, mode, collections.Counter(forms_used) def main(): @@ -138,21 +689,35 @@ def main(): parser.add_argument("--id", required=True, help="request id to audit") parser.add_argument("--topk", type=int, required=True, help="the SUBMIT logprobs=k value the request used") + parser.add_argument("--vocab", type=int, required=True, + help="checkpoint vocabulary size used to bound token ids") args = parser.parse_args() + try: + if _uint(args.id.encode("ascii"), "requested id", _UINT64_MAX) < 1: + raise ValueError("requested id must be positive") + except (UnicodeEncodeError, ValueError) as exc: + parser.error(str(exc)) + if not 1 <= args.topk <= 32: + parser.error("--topk must be in 1..32 for an opted-in B1 evidence run") + if args.vocab < 1: + parser.error("--vocab must be positive") blob = open(args.transcript, "rb").read() - frames = parse_frames(blob) - data_frames, echo_frames, problems, flags = check(frames, args.id, args.topk) - print(f"[gapcheck] id={args.id}: {data_frames} DATA frames, " - f"{echo_frames} ECHO frames, {len(flags)} flagged non-finite") - if data_frames == 0: - problems.append("no DATA frames found for this id -- wrong id, or the " - "run produced nothing (check ERROR frames)") - for flag in flags: - print(f"[gapcheck] FLAG: {flag}") + frames, framing_problems = parse_frames(blob) + data_frames, echo_frames, problems, mode, forms_used = check( + frames, framing_problems, args.id, args.topk, args.vocab) + # Informational only -- see _numeric_value: per-token form + # classification is inherently ambiguous, so this tally never affects + # the verdict, only what a reviewer sees. + forms_summary = (", ".join(f"{form}={forms_used[form]}" + for form in sorted(forms_used)) + if forms_used else "none") + print(f"[gapcheck] id={args.id}: capture_mode={mode}, {data_frames} DATA " + f"frames, {echo_frames} ECHO frames, numeric form tally: " + f"{forms_summary}") for problem in problems: print(f"[gapcheck] {problem}") - verdict = ("FAIL" if problems - else "PASS: every generated token carries a logprob value") + verdict = ("FAIL" if problems else + "PASS: complete request; every generated token carries a valid logprob table") print(f"[gapcheck] {verdict}") return 1 if problems else 0 diff --git a/c/tests/fixtures/gguf_dequant_golden.npz b/c/tests/fixtures/gguf_dequant_golden.npz new file mode 100644 index 000000000..00f27e4f4 Binary files /dev/null and b/c/tests/fixtures/gguf_dequant_golden.npz differ diff --git a/c/tests/fixtures/olmoe_quantize_row_golden.npz b/c/tests/fixtures/olmoe_quantize_row_golden.npz new file mode 100644 index 000000000..5059eb190 Binary files /dev/null and b/c/tests/fixtures/olmoe_quantize_row_golden.npz differ diff --git a/c/tests/gguf_fixture.py b/c/tests/gguf_fixture.py new file mode 100644 index 000000000..4babd9ab5 --- /dev/null +++ b/c/tests/gguf_fixture.py @@ -0,0 +1,216 @@ +#!/usr/bin/env python3 +"""Deterministic GGUF v3 fixture builder for tests/test_gguf_reader.py. + +Builds a small, valid GGUF v3 file exercising every metadata value type +(scalar ints/floats/bool/string, flat arrays and a nested array) plus tensors +of the raw types (F32, F16, BF16) and one block-quantized type (Q8_0). + +This module is deliberately self-contained: it defines its own copy of the +GGUF spec constants instead of importing gguf_reader.py. The reader tests are +therefore a genuine cross-check of two independent implementations of the same +public spec, not a round-trip of one implementation against itself. + +The file layout mirrors ggml-org/ggml/docs/gguf.md: header, metadata KV, +tensor infos, padding to general.alignment, then aligned tensor data whose +offsets are relative to the data section start. Shapes are stored reversed +relative to the PyTorch convention, like real GGUF writers do. +""" + +import os +import random +import struct + +GGUF_MAGIC = b"GGUF" +DEFAULT_ALIGNMENT = 32 +GGUF_VERSION = 3 + +METADATA_UINT8 = 0 +METADATA_INT8 = 1 +METADATA_UINT16 = 2 +METADATA_INT16 = 3 +METADATA_UINT32 = 4 +METADATA_INT32 = 5 +METADATA_FLOAT32 = 6 +METADATA_BOOL = 7 +METADATA_STRING = 8 +METADATA_ARRAY = 9 +METADATA_UINT64 = 10 +METADATA_INT64 = 11 +METADATA_FLOAT64 = 12 + + +def _align(value, alignment): + return (value + alignment - 1) // alignment * alignment + + +def _encode_string(value): + data = value.encode("utf-8") + return struct.pack("> 16) & 0xFFFF) + + +def fixture_tensors(seed): + rng = random.Random(seed + 1) + specs = [ + ("tensor.f32.2d", 0, [8, 16]), # F32 (logical [16, 8]) + ("tensor.f16.2d", 1, [8, 16]), # F16 + ("tensor.bf16.1d", 30, [16]), # BF16 + ("tensor.q8_0.2d", 8, [32, 16]), # Q8_0, dims[0] % 32 == 0 + ] + tensors = [] + for name, ggml_type, dims in specs: + if ggml_type in (0, 1, 30): + values = [rng.uniform(-1.0, 1.0) for _ in range(_numel(dims))] + if ggml_type == 0: + payload = b"".join(struct.pack(" 1: + payload *= dims[1] + else: + raise ValueError("unsupported fixture type %d" % ggml_type) + tensors.append({"name": name, "ggml_type": ggml_type, + "dims": dims, "payload": payload}) + return tensors + + +def _numel(dims): + total = 1 + for dim in dims: + total *= dim + return total + + +def write_gguf_fixture(path, seed=20260825, alignment=DEFAULT_ALIGNMENT): + """Write the deterministic GGUF fixture and return (metadata, tensors).""" + metadata = fixture_metadata(seed, alignment) + tensors = fixture_tensors(seed) + + out = GGUF_MAGIC + out += struct.pack(" 1 else "gguf_fixture.gguf" + if os.path.dirname(output): + os.makedirs(os.path.dirname(output), exist_ok=True) + metadata, tensors = write_gguf_fixture(output) + print("wrote %s: %d metadata KV, %d tensors" % (output, len(metadata), len(tensors))) diff --git a/c/tests/test_glm53_chat_template.py b/c/tests/glm53_chat_template_harness.py similarity index 63% rename from c/tests/test_glm53_chat_template.py rename to c/tests/glm53_chat_template_harness.py index 066bff6e9..a379e9c0f 100644 --- a/c/tests/test_glm53_chat_template.py +++ b/c/tests/glm53_chat_template_harness.py @@ -13,10 +13,17 @@ Serve chat_template.jinja del checkpoint. Se non c'e' il test si dichiara saltato invece di passare: un test che non ha trovato il suo riferimento non ha -verificato niente, e dirlo verde sarebbe peggio che non averlo. +verificato niente, e dirlo verde sarebbe peggio che non averlo -- percio' un +salto esce con codice 2, distinto dallo 0 di un confronto riuscito. + +RIFERIMENTO (scaricato 2026-09-10): + repo zai-org/GLM-5.3-Flash + file chat_template.jinja + sha256 34d5ee66b12fa6446cdae131c352b8f68cd85369e0e6fda115583805fada3891 + hf download zai-org/GLM-5.3-Flash chat_template.jinja USO: - python3 tests/test_glm53_chat_template.py --template PATH/chat_template.jinja + python3 tests/glm53_chat_template_harness.py --template PATH/chat_template.jinja """ import argparse import json @@ -71,7 +78,8 @@ EFFORTS = {"low": "low", "high": "high", "xhigh": None, None: None} -def reference(template_text, *, messages, tools=None, reasoning_effort=None): +def reference(template_text, *, messages, tools=None, reasoning_effort=None, + add_generation_prompt=True): import jinja2 environment = jinja2.Environment(trim_blocks=False, lstrip_blocks=False, extensions=["jinja2.ext.loopcontrols"]) @@ -81,7 +89,7 @@ def reference(template_text, *, messages, tools=None, reasoning_effort=None): environment.filters["tojson"] = ( lambda value, ensure_ascii=False, **kw: json.dumps(value, ensure_ascii=ensure_ascii)) rendered = environment.from_string(template_text) - arguments = {"messages": messages, "add_generation_prompt": True} + arguments = {"messages": messages, "add_generation_prompt": add_generation_prompt} if tools: arguments["tools"] = tools if reasoning_effort: @@ -97,12 +105,12 @@ def main() -> int: if not arguments.template.exists(): print(f"SKIP: manca {arguments.template}; il riferimento non c'e' e " f"questo test non ha verificato nulla") - return 0 + return 2 # salto != successo (vedi docstring) try: import jinja2 # noqa: F401 except ImportError: print("SKIP: jinja2 non installato; senza non c'e' riferimento") - return 0 + return 2 # salto != successo (vedi docstring) sys.path.insert(0, str(Path(__file__).resolve().parents[1])) import openai_server @@ -166,9 +174,66 @@ def main() -> int: return 1 checked += 1 + # Prosecuzione: l'ultimo turno assistant e' da CONTINUARE, non uno gia' finito. Nel + # template e' add_generation_prompt=False -- l'altro ramo dell'`if` che il gateway + # ha sempre fissato a True -- e il riferimento qui e' quel ramo reso con jinja2, + # non una forma scritta a mano. Il prompt finisce dentro il turno, sull'apertura + # del client, e non su un nuovo <|assistant|>. + # + # Nota su #1327: il blocco chiuso in fondo al prompt e' fuori + # distribuzione, e infatti resta vietato -- ma la posizione qui e' un'altra, + # seguito da contenuto vero, che e' la forma che il template + # scrive davanti a ogni turno passato. Per questo una prosecuzione vuota e' + # rifiutata a monte (tests/test_openai_server.py, TrailingAssistantTurnTest). + aperto = [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + # L'effort esce ESPLICITO su entrambi i lati, e su ognuno dei tre livelli che il + # template sa esprimere: 'low' e 'high' passano invariati, 'xhigh' rende Max come il + # default del template. 'minimal'/'medium' non sono qui apposta -- il gateway li + # riduce a Low/High (la sua scala e' piu' ricca del template), quindi contro il + # riferimento non c'e' niente da confrontare. Passare l'effort esplicito toglie di + # mezzo la differenza sul default (gateway High, template Max), che non e' una + # questione di prosecuzione: e' la scala dei livelli, e vale identica sul ramo True. + for effort in ("low", "high", "xhigh"): + produced = openai_server.render_chat_for_arch( + aperto, enable_thinking=True, reasoning_effort=effort, add_generation_prompt=False) + expected = reference(template_text, messages=aperto, reasoning_effort=effort, + add_generation_prompt=False) + if produced != expected: + print(f"FAIL prosecuzione (reasoning_effort={effort!r}): il turno aperto non " + f"rende come il template con add_generation_prompt=False") + for position, (left, right) in enumerate(zip(produced, expected)): + if left != right: + start = max(0, position - 40) + print(f" primo scostamento a {position}") + print(f" gateway: ...{produced[start:position + 40]!r}") + print(f" template: ...{expected[start:position + 40]!r}") + break + else: + print(f" lunghezze diverse: {len(produced)} contro {len(expected)}") + return 1 + if produced.endswith("<|assistant|>"): + print(f"FAIL prosecuzione (reasoning_effort={effort!r}): il prompt finisce su un " + f"nuovo turno invece di proseguire quello del client") + return 1 + if not produced.endswith("La capitale e'"): + print(f"FAIL prosecuzione (reasoning_effort={effort!r}): il prompt non finisce " + f"sull'apertura del client: {produced[-60:]!r}") + return 1 + # Controllo negativo: col ramo True lo stesso scambio DEVE finire sulla cue, o il + # confronto qui sopra non starebbe distinguendo niente. + if not openai_server.render_chat_for_arch( + aperto, enable_thinking=True, + reasoning_effort=effort).endswith("<|assistant|>"): + print(f"FAIL prosecuzione (reasoning_effort={effort!r}): il ramo normale non " + f"emette piu' il prompt di generazione") + return 1 + checked += 1 + print(f"PASS GLM-5.3 chat template: {checked} rese identiche a " f"chat_template.jinja, strumenti e livelli di ragionamento compresi, " - f"piu' il livello minimo, che rende come il template con effort low") + f"piu' il livello minimo, che rende come il template con effort low, " + f"piu' il turno aperto (add_generation_prompt=False)") return 0 diff --git a/c/tests/test_glm53_multimodal_tiny.py b/c/tests/glm53_multimodal_tiny_harness.py similarity index 97% rename from c/tests/test_glm53_multimodal_tiny.py rename to c/tests/glm53_multimodal_tiny_harness.py index bf2ae6db9..86ead4f40 100644 --- a/c/tests/test_glm53_multimodal_tiny.py +++ b/c/tests/glm53_multimodal_tiny_harness.py @@ -49,9 +49,9 @@ def main() -> int: # le due cose. Il generatore vuole transformers 5.16.1. print(f"SKIP: manca {arguments.fixture}; generalo con\n" f" python3 tools/make_glm53_multimodal_tiny.py --output ") - return 0 + return 2 # un salto non e' un successo - # Vedi test_glm53_tiny.py: in f32 si prova che il motore implementa il + # Vedi glm53_tiny_harness.py: in f32 si prova che il motore implementa il # modello, a bit ridotti che la quantizzazione conserva i token. bits = arguments.bits if arguments.logit_tolerance is None: diff --git a/c/tests/test_glm53_serve.py b/c/tests/glm53_serve_harness.py similarity index 99% rename from c/tests/test_glm53_serve.py rename to c/tests/glm53_serve_harness.py index ef78022c9..586e7b64a 100644 --- a/c/tests/test_glm53_serve.py +++ b/c/tests/glm53_serve_harness.py @@ -118,7 +118,7 @@ def main() -> int: # le due cose. Il generatore vuole transformers 5.16.1. print(f"SKIP: manca {arguments.fixture}; generalo con\n" f" python3 tools/make_glm53_multimodal_tiny.py --output ") - return 0 + return 2 # un salto non e' un successo binary = os.path.abspath(arguments.binary) prompt, tokens = "gu", 4 diff --git a/c/tests/test_glm53_streaming.py b/c/tests/glm53_streaming_harness.py similarity index 98% rename from c/tests/test_glm53_streaming.py rename to c/tests/glm53_streaming_harness.py index ec91856c2..d77c073af 100644 --- a/c/tests/test_glm53_streaming.py +++ b/c/tests/glm53_streaming_harness.py @@ -61,7 +61,7 @@ def main() -> int: # le due cose. Il generatore vuole transformers 5.16.1. print(f"SKIP: manca {arguments.quantized}; generalo con\n" f" python3 tools/make_glm53_streaming_pair.py --fixture --output ") - return 0 + return 2 # un salto non e' un successo reference = json.loads((arguments.quantized / "ref.json").read_text()) grid_h, grid_w = reference.get("grid", (0, 0)) diff --git a/c/tests/test_glm53_tiny.py b/c/tests/glm53_tiny_harness.py similarity index 98% rename from c/tests/test_glm53_tiny.py rename to c/tests/glm53_tiny_harness.py index 0a9a1d7e5..80a4b735b 100644 --- a/c/tests/test_glm53_tiny.py +++ b/c/tests/glm53_tiny_harness.py @@ -60,7 +60,7 @@ def main() -> int: print(f"SKIP: manca {arguments.fixture}/ref.json; generalo con\n" f" pip install -r tools/requirements-glm53-tiny.txt\n" f" python3 tools/make_glm53_tiny.py --output {arguments.fixture}") - return 0 + return 2 # un salto non e' un successo reference = json.loads((arguments.fixture / "ref.json").read_text()) prompt = ",".join(str(token) for token in reference["prompt_ids"]) expected_forcing = reference["teacher_forcing_ids"] diff --git a/c/tests/test_glm53_vision_serve.py b/c/tests/glm53_vision_serve_harness.py similarity index 95% rename from c/tests/test_glm53_vision_serve.py rename to c/tests/glm53_vision_serve_harness.py index 7e3ca7a7c..381a219d3 100644 --- a/c/tests/test_glm53_vision_serve.py +++ b/c/tests/glm53_vision_serve_harness.py @@ -18,7 +18,7 @@ vision finta. USO: - python3 tests/test_glm53_vision_serve.py --binary ./glm53 --fixture ~/glm53_mm_tiny + python3 tests/glm53_vision_serve_harness.py --binary ./glm53 --fixture ~/glm53_mm_tiny """ import argparse import json @@ -77,14 +77,14 @@ def main() -> int: # le due cose. Il generatore vuole transformers 5.16.1. print(f"SKIP: manca {arguments.fixture}; generalo con\n" f" python3 tools/make_glm53_multimodal_tiny.py --output ") - return 0 + return 2 # un salto non e' un successo try: import numpy from PIL import Image except ImportError as problem: print(f"SKIP: servono numpy e Pillow ({problem}); niente e' stato verificato") - return 0 + return 2 # un salto non e' un successo sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "tools")) from glm53_image import preprocess, load_config @@ -99,7 +99,7 @@ def main() -> int: tokens = (images[0][1] // merge) * (images[0][2] // merge) if images[1][1:] != images[0][1:]: print("SKIP: le due immagini di prova hanno griglie diverse") - return 0 + return 2 # un salto non e' un successo prompt = "gu" + IMAGE_OPEN + IMAGE_TOKEN * tokens + IMAGE_CLOSE + "xy" environment = {**os.environ, "SERVE": "1", "SERVE_BATCH": "1", diff --git a/c/tests/test_glm53_vulkan.py b/c/tests/glm53_vulkan_harness.py similarity index 95% rename from c/tests/test_glm53_vulkan.py rename to c/tests/glm53_vulkan_harness.py index 5005ffc52..5219a7a13 100644 --- a/c/tests/test_glm53_vulkan.py +++ b/c/tests/glm53_vulkan_harness.py @@ -20,7 +20,7 @@ USO: make VK=1 glm53 - python3 tests/test_glm53_vulkan.py --binary ./glm53 --fixture ~/glm53_mm_tiny + python3 tests/glm53_vulkan_harness.py --binary ./glm53 --fixture ~/glm53_mm_tiny """ import argparse import os @@ -64,7 +64,7 @@ def main() -> int: # le due cose. Il generatore vuole transformers 5.16.1. print(f"SKIP: missing {arguments.fixture}; generate it with\n" f" python3 tools/make_glm53_multimodal_tiny.py --output ") - return 0 + return 2 # a skip is not a pass binary = os.path.abspath(arguments.binary) reference = json.loads((arguments.fixture / "ref.json").read_text()) @@ -75,7 +75,7 @@ def main() -> int: reason = ("binary was not built with VK=1" if "Vulkan:" not in notes else "no usable Vulkan device") print(f"SKIP: {reason}; Vulkan path was not verified") - return 0 + return 2 # a skip is not a pass device = next((line for line in notes.splitlines() if "[VK] ready:" in line), "") for field in ("teacher_forcing", "greedy"): diff --git a/c/tests/golden_fixture_capture.py b/c/tests/golden_fixture_capture.py index 9ae6424e1..3633c7385 100644 --- a/c/tests/golden_fixture_capture.py +++ b/c/tests/golden_fixture_capture.py @@ -4,9 +4,11 @@ Captures full HTTP responses for a battery of NON-logprobs requests against a RUNNING colibri server, normalizes the volatile fields (ids, timestamps), and byte-diffs two capture directories. The engine-channel change (U7a) is opt-in -and no server request path opts in, so a pre-U7a capture and a post-U7a capture -of the same battery on the same box/backend/model must be byte-identical -- -this script is the mechanism proving that, not reviewer inspection. +and no server request path opts in: for the battery's generation-bearing +cases, which all run at temperature 0, a capture taken before the change +under test and one taken after it, on the same box/backend/model, must be +byte-identical after that normalization -- this script is the mechanism +proving that, not reviewer inspection. Usage: # 1. start the server on the CURRENT (pre-change) build, then: @@ -22,8 +24,15 @@ python3 tests/golden_fixture_capture.py diff fixtures_pre fixtures_post Battery: chat, chat+tools, chat streaming, completions without logprobs, and -the error cases whose behavior must not move (seed 400, array-prompt 400, -logprobs 400, out-of-range temperature 400) plus /v1/models. +the error cases whose behavior must not move (array-prompt 400, logprobs 400, +out-of-range temperature 400), plus one case that is a no-op from this change +on (seed accepted-and-ignored), plus /v1/models. + +A pre/post diff spanning this change prints `err_seed.json: MISSING on one +side` and `seed_accepted.json: MISSING on one side`, then exits 1: the +rename means the two captures carry different keys for this one case, so +`diff()` cannot pair them under either name. That pair is expected across +this change and is the only expected difference. Normalization: every "id"/"created" field (recursively, and per SSE event) is replaced with a constant; nothing else is touched. Generation-bearing requests @@ -76,8 +85,8 @@ def battery(model): ("completions_stop", "POST", "/v1/completions", {"model": model, "prompt": "Count: one, two,", "max_tokens": 16, "temperature": 0, "stop": ["five"]}), - ("err_seed", "POST", "/v1/completions", - {"model": model, "prompt": "hello", "max_tokens": 1, "seed": 1234}), + ("seed_accepted", "POST", "/v1/completions", + {"model": model, "prompt": "hello", "max_tokens": 1, "temperature": 0, "seed": 1234}), ("err_array_prompt", "POST", "/v1/completions", {"model": model, "prompt": [1, 2, 3], "max_tokens": 1}), ("err_logprobs", "POST", "/v1/completions", diff --git a/c/tests/qwen36_fake_cuda.h b/c/tests/qwen36_fake_cuda.h index 46d8f7a33..129802141 100644 --- a/c/tests/qwen36_fake_cuda.h +++ b/c/tests/qwen36_fake_cuda.h @@ -19,6 +19,7 @@ * sleeps here; NULL (the default) uploads instantly. */ #ifndef QWEN36_FAKE_CUDA_H #define QWEN36_FAKE_CUDA_H +#include #include #include @@ -110,9 +111,14 @@ void coli_cuda_stats(int device, size_t *count, size_t *bytes) { * device). Counted, never computed: the placement tests check WHERE work * went; the arithmetic has its own oracle in the CUDA build. Parameters are * unused on purpose (CFLAGS carry -Wno-unused-parameter). */ -static int fake_matmuls; +static int fake_matmuls, fake_matmul_fail, fake_matmul_rows, fake_matmul_fail_at; int coli_cuda_matmul(ColiCudaTensor **tensor, float *y, const float *x, const void *weights, const float *scales, int fmt, int S, int I, int O, int device, int gs) { fake_matmuls++; + fake_matmul_rows = S; + if (fake_matmul_fail || fake_matmuls == fake_matmul_fail_at) { + for(int i=0;ifmt == 1 && t->w && t->sc && t->I == I && t->O == O) { const int8_t *q = (const int8_t *)t->w; diff --git a/c/tests/test_alloc_footprint_cuda.cu b/c/tests/test_alloc_footprint_cuda.cu index bb973ca91..79c6d9211 100644 --- a/c/tests/test_alloc_footprint_cuda.cu +++ b/c/tests/test_alloc_footprint_cuda.cu @@ -16,6 +16,9 @@ #include #include #include "../backend_cuda.h" +#if defined(__HIPCC__) +#include "../backend_gpu_compat.h" /* separate TU: needs the CUDA->HIP mapping itself */ +#endif static int failures = 0; diff --git a/c/tests/test_anthropic_messages.py b/c/tests/test_anthropic_messages.py index b8006b441..7d9bd6a54 100644 --- a/c/tests/test_anthropic_messages.py +++ b/c/tests/test_anthropic_messages.py @@ -11,6 +11,7 @@ - the Anthropic error envelope, which is not the OpenAI one. """ import json +import os import re import threading import unittest @@ -19,7 +20,8 @@ from urllib.request import Request, urlopen from openai_server import (APIServer, anthropic_to_openai, anthropic_tools, APIError, - render_chat, render_chat_inkling, render_chat_kimi, render_chat_v4) + render_chat, render_chat_glm53, render_chat_inkling, + render_chat_kimi, render_chat_v4) class FakeEngine: @@ -155,6 +157,24 @@ def test_message_response_shape(self): self.assertEqual(payload["usage"], {"input_tokens": 11, "output_tokens": 3}) self.assertTrue(payload["id"].startswith("msg_")) + def test_trailing_assistant_turn_follows_shared_continuation_switch(self): + """Off preserves the existing cue; on continues the turn on both endpoints.""" + messages = [{"role": "user", "content": "The capital of France is?"}, + {"role": "assistant", "content": "The capital is"}] + with patch("openai_server.ARCH", "glm53"): + with patch.dict(os.environ, {"COLI_CONTINUE_ASSISTANT": "0"}): + with self.post(self.base_body(messages=messages)) as response: + self.assertEqual(response.status, 200) + self.assertEqual(self.engine.prompts[-1], render_chat_glm53(messages)) + self.assertTrue(self.engine.prompts[-1].endswith("<|assistant|>")) + + with patch.dict(os.environ, {"COLI_CONTINUE_ASSISTANT": "1"}): + with self.post(self.base_body(messages=messages)) as response: + self.assertEqual(response.status, 200) + self.assertEqual(self.engine.prompts[-1], + render_chat_glm53(messages, add_generation_prompt=False)) + self.assertTrue(self.engine.prompts[-1].endswith("The capital is")) + def test_each_architecture_receives_its_native_chat_prompt(self): messages = [{"role": "user", "content": "Hi"}] renderers = { diff --git a/c/tests/test_backend_cuda.cu b/c/tests/test_backend_cuda.cu index 049ccf0a0..1603048c7 100644 --- a/c/tests/test_backend_cuda.cu +++ b/c/tests/test_backend_cuda.cu @@ -223,6 +223,106 @@ static int test_fmt6(int dev) { return 1; } +/* ---- fmt=8 (fp8-e4m3) absorb decode ----------------------------------- + * Exercises weight_at's new fmt=8 branch and absorb_scale's new per-128x128- + * block branch (this PR pair) through the REAL attention_absorb kernel and + * absorb_fmt_ok gate, against a CPU reference built the same way the fmt=0 + * absorb block in main() (below) is: independent score/softmax/context + * accumulation, only the weight lookup itself changes to an e4m3 block-scale + * dequant. Dims are chosen so BOTH the row-block and column-block axes carry + * a partial tail block (O=H*(Q+V)=160 -> nblkO=2, rows 128..159 partial; + * K=140 -> nblkI=2, cols 128..139 partial) -- the exact geometry the new + * branches must index correctly (blkO=row>>7, blkI=k>>7, scale index + * blkO*nblkI+blkI). The e4m3 reference decoder is arithmetic (sign/exp/mant, + * OCP E4M3-FN policy), not the engine's c_e4m3 LUT, so it cross-checks + * coli_cuda_fp8_set_lut's uploaded table rather than assuming it -- same + * independence discipline as t8_e4m3_ref's sibling in tests/test_fp8_cuda.cu. */ +static float t8_e4m3_ref(uint8_t b) { + int s = b >> 7, e = (b >> 3) & 15, m = b & 7; + if (e == 15 && m == 7) return NAN; /* E4M3-FN: only NaN, no inf */ + float v = e ? ldexpf(1.f + m/8.f, e-7) : ldexpf(m/8.f, -6); + return s ? -v : v; +} +static uint32_t t8_rng_state = 0xC001D00Du; +static uint8_t t8_rnd_byte(void) { + t8_rng_state ^= t8_rng_state<<13; t8_rng_state ^= t8_rng_state>>17; t8_rng_state ^= t8_rng_state<<5; + uint8_t b = (uint8_t)(t8_rng_state & 0xFF); + if ((b & 0x7F) == 0x7F) b &= (uint8_t)~1; /* avoid the two NaN byte patterns */ + return b; +} +static float t8_dequant(const uint8_t *q, const float *scale, int I, int row, int col, int nblkI) { + int blkO = row >> 7, blkI = col >> 7; + return t8_e4m3_ref(q[(size_t)row*I + col]) * scale[(size_t)blkO*nblkI + blkI]; +} + +static int test_fmt8_absorb(int dev) { + float lut[256]; for (int i = 0; i < 256; i++) lut[i] = t8_e4m3_ref((uint8_t)i); + if (!coli_cuda_fp8_set_lut(lut)) { std::fprintf(stderr,"fmt=8 absorb: set_lut failed\n"); return 0; } + + const int H = 2, Q = 40, V = 40, R = 2, K = 140, T = 3, O = H*(Q+V); + const int nblkO = (O+127)/128, nblkI = (K+127)/128, nblk = nblkO*nblkI; + uint8_t *w = (uint8_t*)std::malloc((size_t)O*K); + for (size_t i = 0; i < (size_t)O*K; i++) w[i] = t8_rnd_byte(); + float *wscale = (float*)std::malloc((size_t)nblk*sizeof(float)); + for (int i = 0; i < nblk; i++) wscale[i] = 0.01f + 0.002f*(float)i; + + /* Refusal must have held BEFORE this test ever ran (fmt=8 was invisible to + * absorb_fmt_ok on unpatched main) -- upload + launch below is the positive + * side of the same predicate this PR widened. */ + ColiCudaTensor *wt = nullptr; + if (!coli_cuda_tensor_upload(&wt, w, wscale, 8, K, O, dev)) { + std::fprintf(stderr,"fmt=8 absorb: weight upload rejected\n"); return 0; + } + + float *q = (float*)std::malloc((size_t)H*(Q+R)*sizeof(float)); + float *latent = (float*)std::malloc((size_t)T*K*sizeof(float)); + float *rope = (float*)std::malloc((size_t)T*R*sizeof(float)); + for (int i = 0; i < H*(Q+R); i++) q[i] = std::sin((float)(i+1)*0.037f); + for (int i = 0; i < T*K; i++) latent[i] = std::sin((float)(i+1)*0.019f)*0.5f; + for (int i = 0; i < T*R; i++) rope[i] = std::cos((float)(i+1)*0.041f)*0.3f; + float *ctx = (float*)std::malloc((size_t)H*V*sizeof(float)); + float scale = 1.f/std::sqrt((float)K); + + if (!coli_cuda_attention_absorb(wt, ctx, q, latent, rope, H, Q, R, V, K, T, scale)) { + std::fprintf(stderr,"fmt=8 absorb: kernel launch rejected (absorb_fmt_ok gate?)\n"); return 0; + } + + int bad = 0; + for (int h = 0; h < H; h++) { + int rbase = h*(Q+V); + float qa[512]; /* K<=512, this attention_absorb's own documented bound */ + for (int k = 0; k < K; k++) { + double a = 0; + for (int d = 0; d < Q; d++) a += (double)q[h*(Q+R)+d]*t8_dequant(w,wscale,K,rbase+d,k,nblkI); + qa[k] = (float)a; + } + float scores[T]; + for (int t = 0; t < T; t++) { + double a = 0; for (int k = 0; k < K; k++) a += (double)qa[k]*latent[t*K+k]; + for (int d = 0; d < R; d++) a += (double)q[h*(Q+R)+Q+d]*rope[t*R+d]; + scores[t] = (float)a*scale; + } + float mx = scores[0]; for (int t = 1; t < T; t++) mx = scores[t]>mx?scores[t]:mx; + float z = 0; for (int t = 0; t < T; t++) { scores[t] = std::exp(scores[t]-mx); z += scores[t]; } + for (int t = 0; t < T; t++) scores[t] /= z; + float cl[512]; + for (int k = 0; k < K; k++) { double a=0; for (int t=0;t1e-4f ? std::fabs(got-want)/std::fabs(want) : std::fabs(got-want); + if (rel > 1e-3f) { + std::fprintf(stderr,"fmt=8 absorb mismatch h=%d v=%d got=%.6f want=%.6f rel=%.4g\n",h,v,got,want,rel); + bad++; + } + } + } + coli_cuda_tensor_free(wt); + std::free(w); std::free(wscale); std::free(q); std::free(latent); std::free(rope); std::free(ctx); + return bad == 0; +} + int main(int argc, char **argv) { int devices[COLI_CUDA_MAX_DEVICES], ndev = argc > 1 ? argc - 1 : 1; if (ndev > COLI_CUDA_MAX_DEVICES) return 2; @@ -455,6 +555,20 @@ int main(int argc, char **argv) { coli_cuda_stats(-1, &count, &bytes); if (count || bytes) { std::fprintf(stderr,"fmt=6 leaked tensors\n"); return 1; } + /* fmt=8 absorb: same self-contained lifecycle discipline as fmt=6 above, + * BOTH halves. The byte half is load-bearing history: an earlier vintage of + * this feature found coli_cuda_tensor_free subtracting per-row scale bytes + * for a per-block-scaled fmt=8 tensor (upload charged scale_count = + * ceil(O/128)*ng, free subtracted O*ng), so the `tensor_bytes >= bytes` + * guard silently declined the subtraction and the diagnostic VRAM counter + * stuck non-zero forever after freeing ANY fmt=8 tensor. free's accounting + * now mirrors upload's charge expression exactly (see the comment in + * coli_cuda_tensor_free), and this is the assertion that keeps the two + * from drifting apart again for a tracked fmt=8 tensor. */ + if (!test_fmt8_absorb(d0)) return 1; + coli_cuda_stats(-1, &count, &bytes); + if (count || bytes) { std::fprintf(stderr,"fmt=8 absorb leaked tensors\n"); return 1; } + coli_cuda_shutdown(); std::printf("cuda backend: q8/q4/q2/f32/e8 correctness ok on %d device(s)\n", ndev); return 0; diff --git a/c/tests/test_backend_loader.py b/c/tests/test_backend_loader.py index af222d62f..adce6f749 100644 --- a/c/tests/test_backend_loader.py +++ b/c/tests/test_backend_loader.py @@ -918,7 +918,7 @@ def tearDownClass(cls): cls.fixture = None def test_abi_is_derived_from_the_loader_source(self): - """47 mandatory + 8 optional, parsed from backend_loader.c. + """47 mandatory + 9 optional, parsed from backend_loader.c. The counts are a deliberate tripwire: adding a RESOLVE to the loader widens the ABI every Windows DLL must satisfy, and that should be a @@ -929,11 +929,12 @@ def test_abi_is_derived_from_the_loader_source(self): """ f = self.fixture self.assertEqual(len(f.mandatory), 47) - self.assertEqual(len(f.optional), 8) # +matmul_mxfp4 (kimi_k3 via the DLL, #1405), +available_device_count (qwen36 tier, #1533) - self.assertEqual(len(f.exports), 55) + self.assertEqual(len(f.optional), 9) # +expert_mxfp4: optional Kimi SiTU pipeline + self.assertEqual(len(f.exports), 56) self.assertEqual(len(f.exports), len(f.mandatory) + len(f.optional)) self.assertIn("coli_cuda_init", f.mandatory) self.assertIn("coli_cuda_e8_set_grid", f.optional) + self.assertIn("coli_cuda_expert_mxfp4", f.optional) # attention_project_ragged: paged ragged KV runtime (#795). self.assertIn("coli_cuda_attention_project_ragged", f.mandatory) # fp8_set_lut: fmt=8 e4m3 dense/expert kernels (#817). diff --git a/c/tests/test_backend_loader_header_parity.py b/c/tests/test_backend_loader_header_parity.py new file mode 100644 index 000000000..55274256d --- /dev/null +++ b/c/tests/test_backend_loader_header_parity.py @@ -0,0 +1,73 @@ +"""Every coli_cuda_* the header exports must have a Windows loader forwarder. + +On Linux a host links backend_cuda.o directly, so a new COLI_CUDA_DLLEXPORT +prototype in backend_cuda.h is callable the moment backend_cuda.cu defines it. +On Windows the host only sees what backend_loader.c resolves and forwards. +A symbol added to the header but not to the loader therefore builds and runs +everywhere except a CUDA_DLL=1 host, where it surfaces as an undefined +reference -- which is how coli_cuda_available_device_count broke qwen36.exe. + +test_backend_loader.py derives its ABI FROM the loader, so it cannot see this +gap. This test compares the loader against the header instead. Pure text, no +compiler: it runs on every CI host, not only on Windows. +""" +import re +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent + +# Header exports deliberately absent from the Windows loader, with the reason. +# Adding a name here should be as conscious as adding a RESOLVE. +NOT_FORWARDED = { + # COLI_ANS (DietGPU) is wired only for the Linux CUDA=1 path. + "coli_cuda_tensor_upload_compressed", +} + + +def header_exports(): + src = (HERE / "backend_cuda.h").read_text(encoding="utf-8") + return set(re.findall( + r"COLI_CUDA_DLLEXPORT[^;(]*?\b(coli_cuda_\w+)\s*\(", src)) + + +def loader_resolved(): + src = (HERE / "backend_loader.c").read_text(encoding="utf-8") + names = re.findall(r"^\s+RESOLVE(?:_OPT)?\((\w+),", src, re.M) + return {"coli_cuda_" + n for n in names} + + +def loader_defined(): + src = (HERE / "backend_loader.c").read_text(encoding="utf-8") + return set(re.findall( + r"^[A-Za-z_][\w \t\*]*?\b(coli_cuda_\w+)\s*\([^;]*?\)\s*\{", src, re.M)) + + +class LoaderHeaderParityTest(unittest.TestCase): + def test_parsers_found_the_abi(self): + # Guard against a regex that silently matches nothing. + self.assertGreater(len(header_exports()), 40) + self.assertGreater(len(loader_resolved()), 40) + self.assertIn("coli_cuda_init", header_exports()) + + def test_every_header_export_is_resolved_by_the_loader(self): + missing = sorted(header_exports() - loader_resolved() - NOT_FORWARDED) + self.assertEqual(missing, [], + "declared COLI_CUDA_DLLEXPORT in backend_cuda.h but never " + "RESOLVE/RESOLVE_OPT'd in backend_loader.c; a CUDA_DLL=1 " + "host calling them fails to link: %s" % missing) + + def test_every_resolved_symbol_has_a_forwarder(self): + undefined = sorted(loader_resolved() - loader_defined()) + self.assertEqual(undefined, [], + "resolved from the DLL but no public wrapper is defined " + "in backend_loader.c: %s" % undefined) + + def test_exemptions_are_still_real(self): + stale = sorted(NOT_FORWARDED - header_exports()) + self.assertEqual(stale, [], "NOT_FORWARDED names no longer in header: %s" + % stale) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_benchmark_baseline.py b/c/tests/test_benchmark_baseline.py new file mode 100644 index 000000000..13b1a9a7e --- /dev/null +++ b/c/tests/test_benchmark_baseline.py @@ -0,0 +1,186 @@ +"""Incomplete or mismatched campaigns must not become performance rankings.""" +import contextlib +import copy +import io +import json +from pathlib import Path +import tempfile +import subprocess +import sys +from tests import test_benchmark_http_serving as http_fixture +import unittest +from unittest.mock import patch +from tools import benchmark_baseline as baseline +from tools import benchmark_http_serving as http + + +class BaselineTest(unittest.TestCase): + def setUp(self): + temp = tempfile.TemporaryDirectory() + self.addCleanup(temp.cleanup) + self.root = Path(temp.name) + example = Path(__file__).resolve().parents[2] / 'docs/baselines/three-engine.example.json' + self.spec = json.loads(example.read_text()) + self.spec.update(experiment='fixture', cache_policy='warm', reasoning_policy='off', + speculation_policy='off', quality_protocol='external fixture check') + self.spec['hardware'] = dict.fromkeys(self.spec['hardware'], 'fixture host') + sha = baseline.digest(example) + self.spec['model'] = dict(source_revision='model-revision', tokenizer_sha256=sha, + chat_template_sha256=sha) + for engine in self.spec['engines'].values(): + engine.update(served_model='fixture', revision='engine-revision', launch_command='serve fixture', + artifact_sha256=sha, weight_format='fixture', quantization='fp32') + self.spec['matrix'].update(concurrency=[1, 2], rounds=2, repeats=2, warmup_requests=1) + self.manifest = self.root / 'manifest.json' + (self.root / 'workload.jsonl').write_text('{"messages":[{"role":"user","content":"fixture"}]}\n') + self.save_manifest() + self.workload, self.sha = http.load_workload(self.root / 'workload.jsonl') + self.results = self.root / 'results' + + def save_manifest(self): + self.manifest.write_text(json.dumps(self.spec)) + + @staticmethod + def fake_run(**kwargs): + rows = [dict(index=i, success=True, start_seconds=0, duration_seconds=2, + first_output_seconds=.5, completion_tokens=4, finish_reason='stop', + error=None, http_status=200) + for i in range(len(kwargs['workload']) * kwargs['repeats'])] + return rows, http.summarize(rows, 3, kwargs.get('slo_first_output'), kwargs.get('slo_duration')) + + def collect_all(self): + with patch.object(http, 'run', side_effect=self.fake_run), contextlib.redirect_stdout(io.StringIO()): + for item in baseline.plan(self.spec): + baseline.collect(self.spec, self.workload, self.sha, item['engine'], item['round'], self.results) + + def change_report(self, change): + path = self.results / 'colibri-r1-c1.json' + data = json.loads(path.read_text()) + change(data) + path.write_text(json.dumps(data)) + + def compare(self): + return baseline.compare(self.spec, self.workload, self.sha, self.results) + + def test_relative_workload_and_rotation(self): + spec, workload, sha = baseline.load_manifest(self.manifest) + self.assertEqual((sha, workload), (self.sha, self.workload)) + self.assertEqual([x['engine'] for x in baseline.plan(spec)], + ['colibri', 'sglang', 'vllm', 'sglang', 'vllm', 'colibri']) + + def test_artifact_mismatch_requires_deployment_mode(self): + self.spec['engines']['vllm']['quantization'] = 'int4' + self.spec['comparison'] = 'matched_artifact' + self.save_manifest() + with self.assertRaisesRegex(ValueError, 'identical'): + baseline.load_manifest(self.manifest) + self.spec['comparison'] = 'deployment' + self.save_manifest() + baseline.load_manifest(self.manifest) + + def test_bad_matrix_and_placeholders(self): + for key, value in [('concurrency', [True]), ('concurrency', [1, 1]), + ('concurrency', [16]), ('temperature', float('nan')), ('rounds', 0)]: + spec = copy.deepcopy(self.spec) + spec['matrix'][key] = value + self.manifest.write_text(json.dumps(spec)) + with self.subTest(key=key, value=value), self.assertRaises(ValueError): + baseline.load_manifest(self.manifest) + self.spec['hardware']['gpu'] = 'REPLACE-gpu' + self.save_manifest() + with self.assertRaisesRegex(ValueError, 'gpu'): + baseline.load_manifest(self.manifest) + + def test_complete_report_boundaries(self): + self.collect_all() + report = self.compare() + self.assertEqual(report['status'], 'protocol_complete') + self.assertEqual(report['quality'], 'not_assessed') + self.assertEqual(report['token_latency'], 'not_measured') + cell = report['cells'][0] + self.assertEqual(cell['aggregate_completion_tokens_per_second']['median'], 8 / 3) + self.assertEqual(cell['reported_output_tokens']['count'], 4) + self.assertEqual(cell['slo_goodput_requests_per_second']['median'], 2 / 3) + + def test_missing_cell_is_not_dropped(self): + self.collect_all() + (self.results / 'colibri-r1-c1.json').unlink() + report = self.compare() + self.assertIn('missing colibri-r1-c1', report['issues']) + self.assertIsNone(report['cells'][0]['aggregate_completion_tokens_per_second']['median']) + self.assertEqual(report['cells'][0]['rounds_measured'], 1) + + def test_mismatched_and_modified_reports(self): + mutations = [lambda d: d.update(workload_sha256='different'), + lambda d: d['manifest'].update(cache_policy='different'), + lambda d: d.update(harness_sha256='different'), + lambda d: d['summary'].update(successful_completion_tokens_per_second=999), + lambda d: d['requests'].pop()] + self.collect_all() + path = self.results / 'colibri-r1-c1.json' + original = path.read_text() + for mutate in mutations: + path.write_text(original) + self.change_report(mutate) + with self.subTest(mutation=mutate), self.assertRaises(ValueError): + self.compare() + + def test_failed_empty_and_missing_usage(self): + self.collect_all() + def mutate(data): + data['requests'][0].update(success=False, error='http_error') + data['requests'][1].update(completion_tokens=None, first_output_seconds=None) + data['summary'] = http.summarize(data['requests'], 3, 5, 120) + self.change_report(mutate) + report = self.compare() + self.assertEqual(len(report['issues']), 3) + self.assertIsNone(report['cells'][0]['aggregate_completion_tokens_per_second']['median']) + + def test_warmup_failure_and_no_overwrite(self): + def fail(**kwargs): + rows, _ = self.fake_run(**kwargs) + rows[0]['success'] = False + return rows, http.summarize(rows, 1) + with patch.object(http, 'run', side_effect=fail) as run, contextlib.redirect_stdout(io.StringIO()): + self.assertEqual(baseline.collect(self.spec, self.workload, self.sha, + 'colibri', 1, self.results), 1) + self.assertEqual(run.call_count, 1) + with self.assertRaisesRegex(ValueError, 'already'): + baseline.collect(self.spec, self.workload, self.sha, 'colibri', 1, self.results) + data = json.loads((self.results / 'colibri-r1-c1.json').read_text()) + self.assertEqual(data['status'], 'warmup_failed') + self.assertEqual(data['requests'], []) + + def test_secret_is_not_saved(self): + with patch.dict('os.environ', {'OPENAI_API_KEY': 'fixture-secret'}), \ + patch.object(http, 'run', side_effect=self.fake_run) as run, \ + contextlib.redirect_stdout(io.StringIO()): + baseline.collect(self.spec, self.workload, self.sha, 'colibri', 1, self.results) + self.assertEqual(run.call_args.kwargs['key'], 'fixture-secret') + for path in self.results.glob('*.json'): + self.assertNotIn('fixture-secret', path.read_text()) + + def test_cli_collects_real_http_and_compare_reports_missing_cells(self): + fixture = http_fixture.BenchmarkTest() + fixture.setUp() + self.addCleanup(fixture.tearDown) + for config in self.spec['engines'].values(): + config['base_url'] = fixture.url.removesuffix('/chat/completions') + self.spec['matrix'].update(rounds=1) + self.save_manifest() + command = [sys.executable, str(Path(baseline.__file__).resolve())] + common = ['--manifest', str(self.manifest), '--results', str(self.results)] + for engine in baseline.ENGINES: + result = subprocess.run(command + ['run', '--engine', engine, '--round', '1'] + common, + capture_output=True, text=True, timeout=15) + self.assertEqual(result.returncode, 0, result.stderr) + result = subprocess.run(command + ['compare'] + common, capture_output=True, text=True, timeout=15) + self.assertEqual(result.returncode, 0, result.stderr) + report = json.loads(result.stdout) + self.assertEqual(report['status'], 'protocol_complete') + self.assertEqual(len(report['sources']), 6) + self.assertEqual(len(fixture.server.payloads), 18) + (self.results / 'vllm-r1-c2.json').unlink() + result = subprocess.run(command + ['compare'] + common, capture_output=True, text=True, timeout=15) + self.assertEqual(result.returncode, 1, result.stderr) + self.assertIn('missing vllm-r1-c2', json.loads(result.stdout)['issues']) diff --git a/c/tests/test_benchmark_http_serving.py b/c/tests/test_benchmark_http_serving.py new file mode 100644 index 000000000..bc143b28c --- /dev/null +++ b/c/tests/test_benchmark_http_serving.py @@ -0,0 +1,518 @@ +"""HTTP benchmark contract tests; no model, GPU or third-party packages.""" +import io +import json +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +import time +import unittest +from unittest.mock import patch +from concurrent.futures import Future +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from tools import benchmark_http_serving as bench +from openai_server import APIServer + + +def event(value): + return "data: " + (value if isinstance(value, str) else json.dumps(value)) + "\n\n" + + +def stream(delta=None, usage=3, finish=True, done=True): + text = event({"choices": [{"index": 0, "delta": {"role": "assistant"}}]}) + text += event({"choices": [{"index": 0, "delta": delta or {}}]}) + if finish: + text += event({"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}) + if usage is not None: + text += event({"choices": [], "usage": {"completion_tokens": usage}}) + if done: + text += event("[DONE]") + return text.encode() + + +class Handler(BaseHTTPRequestHandler): + def log_message(self, *_args): + pass + + def do_POST(self): + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + with self.server.lock: + self.server.payloads.append(body) + self.server.active += 1 + self.server.peak = max(self.server.peak, self.server.active) + try: + if self.server.barrier: + self.server.barrier.wait(timeout=5) + time.sleep(self.server.delay) + self.send_response(self.server.status) + self.send_header("Content-Type", self.server.content_type) + if self.server.status == 302: + self.send_header("Location", "/must-not-follow") + self.end_headers() + self.wfile.write(self.server.body) + finally: + with self.server.lock: + self.server.active -= 1 + + +class BenchmarkTest(unittest.TestCase): + def setUp(self): + self.server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + self.server.lock = threading.Lock() + self.server.payloads = [] + self.server.active = self.server.peak = 0 + self.server.status = 200 + self.server.content_type = "text/event-stream" + self.server.body = stream({"content": "hello"}) + self.server.barrier = None + self.server.delay = 0 + self.thread = threading.Thread(target=self.server.serve_forever, kwargs={"poll_interval": .01}) + self.thread.start() + self.url = f"http://127.0.0.1:{self.server.server_port}/v1/chat/completions" + self.workload = [{"messages": [{"role": "user", "content": "hello"}]}] + + def tearDown(self): + self.server.shutdown() + self.server.server_close() + self.thread.join() + + def request(self): + return bench.request_one(self.url, self.workload[0], "", 2, 0, time.perf_counter()) + + def test_success_and_first_output(self): + for delta in ({"content": "a"}, {"reasoning_content": "b"}, + {"reasoning": "c"}, {"tool_calls": [{"function": {"arguments": "{}"}}]}): + with self.subTest(delta=delta): + self.server.body = stream(delta) + row = self.request() + self.assertTrue(row["success"]) + self.assertEqual(row["completion_tokens"], 3) + self.assertLessEqual(row["first_output_seconds"], row["duration_seconds"]) + + def test_legacy_function_call_output_and_validation(self): + for function, valid, output in (({"name": "lookup"}, True, True), + ({"arguments": "{}"}, True, True), + ({"arguments": ""}, True, False), + ("lookup", False, False), + ({"arguments": 42}, False, False)): + with self.subTest(function=function): + chunk = {"choices": [{"index": 0, "delta": {"function_call": function}, + "finish_reason": "function_call"}]} + self.server.body = (event(chunk) + event("[DONE]")).encode() + row = self.request() + self.assertEqual(row["success"], valid) + if valid: + self.assertEqual(row["first_output_seconds"] is not None, output) + summary = bench.summarize([row], 1, slo_first_output=5) + self.assertEqual(summary["latency_slo"]["requests_met"], int(output)) + + def test_empty_output_is_not_first_output(self): + self.server.body = stream({"content": "", "tool_calls": [{"id": "id", "function": {}}]}, usage=0) + row = self.request() + self.assertTrue(row["success"]) + self.assertIsNone(row["first_output_seconds"]) + self.assertEqual(row["completion_tokens"], 0) + + def test_failures_are_not_successes(self): + bodies = [stream(done=False), stream(finish=False), b"data: {bad}\n\n", + event({"error": {"message": "private diagnostic"}}).encode(), + stream(usage=-1), stream(usage=True), b"data: [DONE]"] + for body in bodies: + with self.subTest(body=body): + self.server.body = body + row = self.request() + self.assertFalse(row["success"]) + self.assertIsNotNone(row["error"]) + self.assertNotIn("private diagnostic", json.dumps(row)) + + def test_invalid_choices_do_not_count_as_success(self): + choices = [ + [{"index": 0, "delta": {"content": 123}, "finish_reason": "stop"}], + [{"index": 0, "delta": {"content": "ok", "reasoning": []}, "finish_reason": "stop"}], + [{"index": 0, "delta": {}, "finish_reason": False}], + [{"index": 0, "delta": {}, "finish_reason": ""}], + [{"index": 0, "delta": {}, "finish_reason": "error"}], + [{"index": False, "delta": {}, "finish_reason": "stop"}], + [{"delta": {}, "finish_reason": "stop"}], + [{"index": 0, "delta": {}, "finish_reason": "stop"}] * 2, + [{"index": 0, "delta": {"tool_calls": [{"function": {"arguments": 7}}]}, "finish_reason": "tool_calls"}], + ] + for malformed in choices: + with self.subTest(choices=malformed): + self.server.body = (event({"choices": malformed}) + event("[DONE]")).encode() + row = self.request() + self.assertFalse(row["success"], row) + self.assertEqual(bench.summarize([row], 1, slo_duration=1)["latency_slo"]["requests_met"], 0) + + def test_choice_after_finish_is_not_successful(self): + terminal = {"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]} + for delta, finish in (({"content": "late output"}, None), ({}, "length")): + with self.subTest(delta=delta, finish=finish): + extra = {"choices": [{"index": 0, "delta": delta, "finish_reason": finish}]} + self.server.body = (event(terminal) + event(extra) + + event({"choices": [], "usage": {"completion_tokens": 999}}) + + event("[DONE]")).encode() + row = self.request() + self.assertFalse(row["success"]) + self.assertEqual(row["error"], "choice_after_finish") + self.assertEqual(row["finish_reason"], "stop") + summary = bench.summarize([row], 1, slo_duration=10) + self.assertEqual(summary["reported_successful_completion_tokens"], 0) + self.assertEqual(summary["latency_slo"]["requests_met"], 0) + + def test_supported_finish_reasons_and_null_delta(self): + for finish in ("stop", "length", "tool_calls", "function_call", "content_filter"): + with self.subTest(finish=finish): + self.server.body = (event({"choices": [{"index": 0, "delta": None, + "finish_reason": finish}]}) + event("[DONE]")).encode() + row = self.request() + self.assertTrue(row["success"], row) + self.assertIsNone(row["first_output_seconds"]) + + def test_http_errors_and_redirects(self): + for status in (429, 500, 302): + self.server.status = status + row = self.request() + self.assertFalse(row["success"]) + self.assertEqual(row["http_status"], status) + self.assertEqual(len(self.server.payloads), 3) + + def test_content_type(self): + self.server.content_type = "application/json" + self.assertFalse(self.request()["success"]) + + def test_missing_usage_disables_token_rate(self): + self.server.body = stream({"content": "a"}, usage=None) + row = self.request() + summary = bench.summarize([row], 1) + self.assertEqual(summary["succeeded"], 1) + self.assertEqual(summary["successful_requests_with_usage"], 0) + self.assertIsNone(summary["successful_completion_tokens_per_second"]) + + def test_partial_failure_accounting(self): + success = self.request() + failure = dict(success, success=False, completion_tokens=999) + summary = bench.summarize([success, failure], 2) + self.assertEqual(summary["failure_rate"], .5) + self.assertEqual(summary["successful_completion_tokens_per_second"], 1.5) + self.assertEqual(summary["successful_duration_seconds"]["count"], 1) + self.assertIsNone(bench.summarize([failure], 1)["successful_completion_tokens_per_second"]) + + def test_concurrency_and_payload(self): + self.server.barrier = threading.Barrier(2) + rows, summary = bench.run(self.url, self.workload, "test-model", 2, 4, 8, 0, "", 2) + self.assertEqual(summary["succeeded"], 4) + self.assertEqual(self.server.peak, 2) + self.assertEqual([r["index"] for r in rows], list(range(4))) + for body in self.server.payloads: + self.assertEqual(body, dict(self.workload[0], model="test-model", stream=True, + stream_options={"include_usage": True}, max_tokens=8, + temperature=0, n=1)) + + def test_fixed_rate_keeps_client_wait_visible(self): + self.server.delay = .04 + rows, summary = bench.run(self.url, self.workload, "fixture", 1, 3, 8, 0, "", 2, + request_rate=1000) + self.assertEqual(summary["succeeded"], 3) + self.assertEqual(self.server.peak, 1) + self.assertEqual([r["scheduled_seconds"] for r in rows], [0, .001, .002]) + self.assertGreater(rows[-1]["dispatch_delay_seconds"], .05) + for row in rows: + self.assertAlmostEqual(row["arrival_duration_seconds"], + row["duration_seconds"] + row["dispatch_delay_seconds"]) + self.assertAlmostEqual(row["arrival_first_output_seconds"], + row["first_output_seconds"] + row["dispatch_delay_seconds"]) + self.assertEqual(summary["arrival_timing"]["dispatch_delay_seconds"]["count"], 3) + + def test_cli_warmup_is_separate_and_failure_skips_measurement(self): + with tempfile.TemporaryDirectory() as directory: + workload = Path(directory) / "prompts.jsonl" + output = Path(directory) / "report.json" + prompts = self.workload + [{"messages": [{"role": "user", "content": "second"}]}] + workload.write_text("".join(json.dumps(p) + "\n" for p in prompts)) + command = [sys.executable, bench.__file__, "--base-url", self.url.rsplit("/", 1)[0], + "--model", "fixture", "--workload", str(workload), "--output", str(output), + "--warmup-requests", "3", "--concurrency", "2", "--repeats", "2", + "--request-rate", "100", "--slo-duration", "5"] + for status, expected_exit in ((200, 0), (503, 1)): + with self.subTest(status=status): + self.server.status = status + self.server.payloads.clear() + completed = subprocess.run(command, capture_output=True, text=True, timeout=10) + self.assertEqual(completed.returncode, expected_exit, completed.stderr) + report = json.loads(output.read_text()) + warmup = report["warmup"] + self.assertEqual(warmup["summary"]["requests"], 3) + self.assertIsNone(warmup["summary"]["arrival_timing"]) + self.assertIsNone(warmup["summary"]["latency_slo"]) + self.assertLessEqual(self.server.peak, 2) + self.assertCountEqual([p["messages"] for p in self.server.payloads[:3]], + [prompts[i % 2]["messages"] for i in range(3)]) + if expected_exit: + self.assertEqual(report["status"], "warmup_failed") + self.assertIsNone(report["summary"]) + self.assertEqual(report["requests"], []) + self.assertEqual(len(self.server.payloads), 3) + else: + self.assertEqual(report["status"], "measured") + self.assertEqual(len(self.server.payloads), 7) + self.assertEqual(report["summary"]["requests"], 4) + self.assertEqual(report["summary"]["reported_successful_completion_tokens"], 12) + self.assertEqual(report["summary"]["latency_slo"]["requests_met"], 4) + self.assertEqual([r["scheduled_seconds"] for r in report["requests"]], + [0, .01, .02, .03]) + completed = subprocess.run(command + ["--warmup-requests", "-1"], + capture_output=True, text=True, timeout=10) + self.assertEqual(completed.returncode, 2) + + def test_cli_report_and_failure_exit(self): + with tempfile.TemporaryDirectory() as directory: + workload = Path(directory) / "prompts.jsonl" + output = Path(directory) / "report.json" + workload.write_text(json.dumps(self.workload[0]) + "\n", encoding="utf-8") + command = [sys.executable, bench.__file__, "--base-url", self.url.rsplit("/", 1)[0], + "--model", "fixture", "--workload", str(workload), "--output", str(output), + "--slo-first-output", "5", "--slo-duration", "5", "--request-rate", "10"] + for status, exit_code in ((200, 0), (503, 1)): + self.server.status = status + completed = subprocess.run(command, capture_output=True, text=True, timeout=10) + self.assertEqual(completed.returncode, exit_code, completed.stderr) + report = json.loads(output.read_text()) + self.assertEqual(report["summary"]["succeeded"], 1 - exit_code) + self.assertEqual(len(report["config"]["workload_sha256"]), 64) + self.assertNotIn("messages", report["config"]) + self.assertEqual(report["config"]["slo_duration_seconds"], 5) + self.assertEqual(report["config"]["load_model"], "fixed_rate") + self.assertEqual(report["config"]["request_rate"], 10) + self.assertEqual(report["summary"]["latency_slo"]["timing_basis"], "scheduled_arrival") + self.assertEqual(report["summary"]["latency_slo"]["requests_met"], 1 - exit_code) + + + def test_poisson_cli_reports_seed_and_requires_rate(self): + with tempfile.TemporaryDirectory() as directory: + workload = Path(directory) / "workload.jsonl" + output = Path(directory) / "report.json" + workload.write_text('{"messages":[{"role":"user","content":"hello"}]}\n') + argv = [bench.__file__, "--base-url", self.url.removesuffix("/chat/completions"), + "--model", "fixture", "--workload", str(workload), "--output", str(output), + "--arrival-distribution", "poisson", "--seed", "42"] + with patch.object(sys, "argv", argv), patch.object(sys, "stderr", io.StringIO()), \ + self.assertRaises(SystemExit) as error: + bench.main() + self.assertEqual(error.exception.code, 2) + self.assertFalse(output.exists()) + with patch.object(sys, "argv", argv + ["--request-rate", "1000"]), \ + patch.object(sys, "stdout", io.StringIO()): + self.assertEqual(bench.main(), 0) + report = json.loads(output.read_text()) + self.assertEqual(report["config"]["load_model"], "poisson") + self.assertEqual(report["config"]["arrival_distribution"], "poisson") + self.assertEqual(report["config"]["arrival_seed"], 42) + self.assertEqual(report["requests"][0]["scheduled_seconds"], 0) + + +class ColibriIntegrationTest(unittest.TestCase): + def test_real_gateway_with_fake_engine(self): + class Engine: + fail = False + + def generate(self, prompt, maximum, temperature, top_p, on_text, + cache_slot=0, cancelled=None, **kwargs): + kwargs["on_accept"]({"prompt_tokens": 7}) + on_text("Hello") + if self.fail: + raise RuntimeError("private engine diagnostic") + return {"prompt_tokens": 7, "completion_tokens": 1, "length_limited": False} + + server = APIServer(("127.0.0.1", 0), Engine(), "fixture", api_key="test-secret") + thread = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": .01}) + thread.start() + try: + rows, summary = bench.run( + f"http://127.0.0.1:{server.server_port}/v1/chat/completions", + [{"messages": [{"role": "user", "content": "Hi"}]}], + "fixture", 1, 2, 8, 0, "test-secret", 2) + self.assertEqual(summary["succeeded"], 2, rows) + self.assertEqual(summary["reported_successful_completion_tokens"], 2) + self.assertEqual(summary["successful_first_output_seconds"]["count"], 2) + self.assertNotIn("test-secret", json.dumps(rows)) + server.engine.fail = True + failed = bench.request_one( + f"http://127.0.0.1:{server.server_port}/v1/chat/completions", + dict(messages=[{"role": "user", "content": "Hi"}], model="fixture", stream=True), + "test-secret", 2, 2, time.perf_counter()) + self.assertFalse(failed["success"], failed) + self.assertIsNotNone(failed["first_output_seconds"]) + self.assertIsNone(failed["finish_reason"]) + self.assertNotIn("private engine diagnostic", json.dumps(failed)) + summary = bench.summarize(rows + [failed], 1, slo_duration=5) + self.assertEqual((summary["succeeded"], summary["failed"]), (2, 1)) + self.assertEqual(summary["reported_successful_completion_tokens"], 2) + self.assertEqual(summary["latency_slo"]["requests_met"], 2) + server.engine.fail = False + recovered, _ = bench.run( + f"http://127.0.0.1:{server.server_port}/v1/chat/completions", + [{"messages": [{"role": "user", "content": "Hi again"}]}], + "fixture", 1, 1, 8, 0, "test-secret", 2) + self.assertTrue(recovered[0]["success"], recovered) + finally: + server.shutdown() + server.server_close() + thread.join() + + + +class ParsingTest(unittest.TestCase): + def test_sse_multiline_comments_crlf_and_partial_eof(self): + raw = b': ping\r\nevent: message\r\ndata: {"choices":\r\ndata: []}\r\n\r\ndata: truncated' + self.assertEqual(list(bench.sse_events(io.BytesIO(raw))), ['{"choices":\n[]}']) + + def test_endpoint(self): + self.assertEqual(bench.endpoint("http://localhost:8000/v1/"), "http://localhost:8000/v1/chat/completions") + for url in ("file:///v1", "http://key@host/v1", "http://host/v1?key=secret", "http://host/v1#x"): + with self.assertRaises(ValueError): + bench.endpoint(url) + + def test_workload_validation(self): + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "workload.jsonl" + for text in ('', '{}', '{"messages":[]}', '{"messages":[{"role":"user","content":3}]}', + '{"messages":[{"role":"user","content":"ok"}],"temperature":1}'): + path.write_text(text) + with self.assertRaises(ValueError): + bench.load_workload(path) + + def test_latency_slo_counts_all_attempts_in_denominator(self): + def row(first, duration, success=True): + return {"success": success, "first_output_seconds": first, + "duration_seconds": duration, "completion_tokens": None} + rows = [row(.5, 2), row(.6, 1), row(.2, 3), row(None, 1), row(.1, 1, False)] + summary = bench.summarize(rows, 10, slo_first_output=.5, slo_duration=2) + slo = summary["latency_slo"] + self.assertEqual(slo["requests_met"], 1) + self.assertEqual(slo["fraction_of_attempts"], .2) + self.assertEqual(slo["goodput_requests_per_second"], .1) + self.assertIsNone(summary["successful_completion_tokens_per_second"]) + self.assertEqual(bench.summarize(rows, 10, slo_duration=2)["latency_slo"]["requests_met"], 3) + self.assertEqual(bench.summarize(rows, 10, slo_first_output=.5)["latency_slo"]["requests_met"], 2) + self.assertIsNone(bench.summarize(rows, 10)["latency_slo"]) + + def test_latency_slo_no_successes(self): + row = {"success": False, "first_output_seconds": .1, + "duration_seconds": .2, "completion_tokens": 10} + slo = bench.summarize([row], 1, slo_duration=1)["latency_slo"] + self.assertEqual(slo["requests_met"], 0) + self.assertEqual(slo["goodput_requests_per_second"], 0) + + def test_fixed_rate_slo_includes_client_backlog(self): + row = {"success": True, "first_output_seconds": .1, "duration_seconds": .2, + "completion_tokens": None, "scheduled_seconds": 0, + "dispatch_delay_seconds": 2, "arrival_first_output_seconds": 2.1, + "arrival_duration_seconds": 2.2} + summary = bench.summarize([row], 3, slo_first_output=1, slo_duration=1) + self.assertEqual(summary["latency_slo"]["requests_met"], 0) + self.assertEqual(summary["latency_slo"]["timing_basis"], "scheduled_arrival") + row.update(success=False, arrival_first_output_seconds=None) + self.assertEqual(bench.summarize([row], 3)["arrival_timing"]["successful_first_output_seconds"]["count"], 0) + + def test_arrival_schedule_uses_absolute_deadlines(self): + now = [100.0] + sleeps = [] + def sleep(delay): + sleeps.append(delay) + now[0] += delay + .01 # deterministic late wakeup + def submit(_fn, _url, _payload, _key, _timeout, index, origin): + future = Future() + future.set_result({"index": index, "start_seconds": now[0] - origin, + "success": True, "first_output_seconds": .01, + "duration_seconds": .02, "completion_tokens": 1}) + now[0] += .02 + return future + with patch.object(bench.time, "perf_counter", side_effect=lambda: now[0]), \ + patch.object(bench.time, "sleep", side_effect=sleep), \ + patch.object(bench.concurrent.futures, "ThreadPoolExecutor") as executor: + executor.return_value.__enter__.return_value.submit.side_effect = submit + rows, _ = bench.run("unused", [{"messages": []}], "fixture", 1, 3, 8, 0, "", 2, + request_rate=10) + self.assertEqual(len(sleeps), 2) + self.assertAlmostEqual(sleeps[0], .08) + self.assertAlmostEqual(sleeps[1], .07) + for row, expected in zip(rows, [0, .11, .21]): + self.assertAlmostEqual(row["start_seconds"], expected) + + def test_poisson_schedule_is_seeded_and_keeps_backlog_in_slo(self): + def measure(seed, service_time): + now = [100.0] + def submit(_fn, _url, _payload, _key, _timeout, index, origin): + future = Future() + future.set_result({"index": index, "start_seconds": now[0] - origin, + "success": True, "first_output_seconds": .01, + "duration_seconds": .02, "completion_tokens": 1}) + now[0] += service_time + return future + def sleep(delay): + now[0] += delay + with patch.object(bench.time, "perf_counter", side_effect=lambda: now[0]), \ + patch.object(bench.time, "sleep", side_effect=sleep), \ + patch.object(bench.concurrent.futures, "ThreadPoolExecutor") as executor: + executor.return_value.__enter__.return_value.submit.side_effect = submit + return bench.run("unused", [{"messages": []}], "fixture", 1, 5, 8, 0, "", 2, + request_rate=10, arrival_distribution="poisson", seed=seed, + slo_duration=.1) + fast, fast_summary = measure(42, .001) + slow, slow_summary = measure(42, 1) + other, _ = measure(43, .001) + expected = [0, .1020060287274801, .104538912631754, .1367013190392506, + .161959937606262] + for row, deadline in zip(fast, expected): + self.assertAlmostEqual(row["scheduled_seconds"], deadline) + self.assertEqual([r["scheduled_seconds"] for r in fast], + [r["scheduled_seconds"] for r in slow]) + self.assertNotEqual([r["scheduled_seconds"] for r in fast], + [r["scheduled_seconds"] for r in other]) + self.assertEqual(fast_summary["latency_slo"]["requests_met"], 5) + self.assertEqual(slow_summary["latency_slo"]["requests_met"], 1) + self.assertAlmostEqual(slow[1]["arrival_duration_seconds"], 1 - expected[1] + .02) + + def test_warmup_time_is_excluded_from_measured_throughput(self): + now = [0.0] + calls = [] + def request(_url, _payload, _key, _timeout, index, origin): + start = now[0] + duration = 100 if not calls else 1 + calls.append(origin) + now[0] += duration + return {"index": index, "start_seconds": start - origin, "success": True, + "first_output_seconds": duration, "duration_seconds": duration, + "completion_tokens": 3} + with tempfile.TemporaryDirectory() as directory: + workload = Path(directory) / "workload.jsonl" + output = Path(directory) / "report.json" + workload.write_text('{"messages":[{"role":"user","content":"hello"}]}\n') + argv = [bench.__file__, "--base-url", "http://unused/v1", "--model", "fixture", + "--workload", str(workload), "--output", str(output), + "--warmup-requests", "1", "--repeats", "2", "--slo-duration", "2"] + with patch.object(sys, "argv", argv), patch.object(sys, "stdout", io.StringIO()), \ + patch.object(bench.time, "perf_counter", side_effect=lambda: now[0]), \ + patch.object(bench, "request_one", side_effect=request): + self.assertEqual(bench.main(), 0) + report = json.loads(output.read_text()) + self.assertEqual(calls, [0, 100, 100]) + self.assertEqual(report["warmup"]["summary"]["wall_seconds"], 100) + self.assertEqual(report["summary"]["wall_seconds"], 2) + self.assertEqual(report["summary"]["successful_completion_tokens_per_second"], 3) + self.assertEqual(report["summary"]["latency_slo"]["goodput_requests_per_second"], 1) + + def test_request_rate_must_be_positive_and_finite(self): + for value in ("0", "-1", "nan", "inf"): + with self.subTest(value=value), self.assertRaises(bench.argparse.ArgumentTypeError): + bench.positive_float(value) + + def test_nearest_rank_distribution(self): + self.assertEqual(bench.distribution(list(range(1, 101)))["p95"], 95) + self.assertIsNone(bench.distribution([])["p95"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_brio_api.py b/c/tests/test_brio_api.py index e95c4cb35..ec2aea9c8 100644 --- a/c/tests/test_brio_api.py +++ b/c/tests/test_brio_api.py @@ -96,9 +96,42 @@ def test_single_form_still_answers_and_never_generates(self): self.assertEqual(out["answer"], "request changes") self.assertEqual(out["usage"]["completion_tokens"], 0) self.assertTrue(all(c["max_tokens"] == 0 for c in self.engine.calls)) - # one photograph, of the whole prefix; the options are never pinned - self.assertEqual(len(self.pins()), 1) - self.assertTrue(self.pins()[0].endswith("Answer:")) + # Two photographs and no more: the shared state on its own, then the + # whole prefix. The options are never pinned. The state photograph is + # the return point that lets a later question on the same document + # reuse the snapshot instead of re-reading it (asserted below). + pins = self.pins() + self.assertEqual(len(pins), 2, pins) + self.assertEqual(pins[0], f"Context:\n{STATE}\n\n") + self.assertTrue(pins[1].startswith(pins[0])) + self.assertTrue(pins[1].endswith("Answer:")) + + def test_options_form_pins_the_shared_state_for_reuse_across_requests(self): + # The web asks one question per request on the same document. The state + # must be photographed on its own each time, so the engine has a strict + # prefix to restore and every question after the first pays only its own + # tokens instead of re-reading the whole document -- the "read once" the + # mode exists for. A handler that folded the state into the question + # prefix only would still answer correctly while re-reading it, so the + # pins are asserted, not just the answers. + self.serve({"merge": -3.0, "request changes": -0.2, "close": -4.0, + "yes": -0.1, "no": -2.0}) + self.post({"model": "test-model", "state": STATE, + "question": "What should the reviewer do?", + "options": ["merge", "request changes", "close"]}) + self.assertEqual([p for p in self.pins() if p == f"Context:\n{STATE}\n\n"], + [f"Context:\n{STATE}\n\n"]) + self.post({"model": "test-model", "state": STATE, + "question": "Does it need tests?", "options": ["yes", "no"]}) + # both requests photographed the same shared state prefix: the return + # point exists for the second question, not only the first + state_pins = [p for p in self.pins() if p == f"Context:\n{STATE}\n\n"] + self.assertEqual(len(state_pins), 2, self.pins()) + # and every question prefix extends that shared state + question_pins = [p for p in self.pins() if p.endswith("Answer:")] + self.assertEqual(len(question_pins), 2, self.pins()) + for pin in question_pins: + self.assertTrue(pin.startswith(f"Context:\n{STATE}\n\n")) # ---- questions: many on one state ---------------------------------------- def test_questions_share_one_state_photograph(self): diff --git a/c/tests/test_build_config_stamp.py b/c/tests/test_build_config_stamp.py new file mode 100644 index 000000000..faa38efbc --- /dev/null +++ b/c/tests/test_build_config_stamp.py @@ -0,0 +1,59 @@ +"""c/Makefile records the build-affecting flags in .build-config (#306). + +The write used $(file ...), which GNU Make 3.81 -- /usr/bin/make on macOS -- +does not have: the stamp was never written and every build relinked (#1732). +Make 3.x now writes it through printf. This drives the parse with +MAKE_VERSION forced to 3.81 and to the running make's own version, and checks +that both write the stamp and that an unchanged configuration leaves it alone. +The previous .build-config is restored afterwards. +""" +import os +import shutil +import subprocess +import time +import unittest +from pathlib import Path + +C_DIR = Path(__file__).resolve().parent.parent +MAKE = shutil.which("make") +STAMP = C_DIR / ".build-config" + + +@unittest.skipUnless(MAKE and os.name == "posix", "make and a POSIX shell are required") +class BuildConfigStampTest(unittest.TestCase): + def setUp(self): + self.saved = STAMP.read_bytes() if STAMP.exists() else None + + def tearDown(self): + if self.saved is None: + STAMP.unlink(missing_ok=True) + else: + STAMP.write_bytes(self.saved) + + def parse(self, *variables): + result = subprocess.run([MAKE, "-s", *variables, ".build-config"], cwd=C_DIR, + text=True, capture_output=True, timeout=120) + self.assertEqual(result.returncode, 0, result.stderr) + + def check(self, *version): + marker = f"-DCOLI_STAMP_TEST_{time.monotonic_ns()}" + self.parse(*version, f"EXTRA_CFLAGS={marker}") + self.assertTrue(STAMP.exists(), "the stamp was not written") + text = STAMP.read_text() + self.assertIn(marker, text) + before = STAMP.stat().st_mtime_ns + time.sleep(0.05) + self.parse(*version, f"EXTRA_CFLAGS={marker}") + self.assertEqual(STAMP.stat().st_mtime_ns, before, + "an unchanged configuration rewrote the stamp") + self.assertEqual(STAMP.read_text(), text) + + def test_make_381_writes_the_stamp(self): + self.check("MAKE_VERSION=3.81") + + def test_current_make_writes_the_stamp(self): + self.check() + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_check_data_logprob_gaps.py b/c/tests/test_check_data_logprob_gaps.py new file mode 100644 index 000000000..0acee63b0 --- /dev/null +++ b/c/tests/test_check_data_logprob_gaps.py @@ -0,0 +1,1157 @@ +"""tests/check_data_logprob_gaps.py must accept only a complete, well-formed +raw engine-stdout transcript for one opted-in request and reject every +other input: an unparsed or mismatched startup preamble, a legacy 3-field +DATA frame hiding a dropped logprob channel anywhere in the generation, a +numeric tail with the wrong top-k count or an out-of-range/duplicate token +id, malformed framing, a missing or duplicated ACCEPT/DONE, and every other +gap the module's docstring claims to catch. + +Checks enumerated from the source (`check_data_logprob_gaps.py`, read in +full before writing this module) and covered below: + +- `_uint`, `_c17g`, `_fixed_metric`: the noncanonical-ASCII-integer grammar, + the exact finite `%.17g` spelling, and the fixed-decimal-place grammar + with its inclusive bounds. +- `_header_fields` / `_global_header`: non-printable-ASCII and stray-space + rejection on an ordinary protocol header, and the exact field grammar + for every recognized mux-global record (BANNER/LOADED preamble, READY, + STAT, HWINFO, TIERS, EMAP, HITS, PROF). +- `parse_frames`: byte-exact DATA/ECHO payload framing, so a single + malformed frame cannot desynchronize the parse. +- `capture_mode`: the four explicit transcript shapes it names. +- `_numeric_tail` / `check`: every check enumerated in the module + docstring, including the legacy 3-field DATA gap this checker exists to + catch, the `min(topk, vocab)` top-k cap, and out-of-range/duplicate + token ids. + +Every fixture here is a literal transcript built by hand from the module's +documented wire grammar -- no expected value is produced by calling the +checker under test. The transcript-shaped fixtures (`_GapFixture`'s BANNER/ +LOADED/VALID/echo/transcript/telemetry/full_capture/complete_capture +helpers) and the `GapCheckerEvidenceTests` methods that exercise them are +carried over unchanged from the checker's own evidence-consumer suite, +excluding the one method that requires a compiled fixture binary. +""" +import ast +import collections +import importlib.util +import inspect +import pathlib +import random +import re +import subprocess +import sys +import tempfile +import unittest + +_HERE = pathlib.Path(__file__).resolve().parent +_spec = importlib.util.spec_from_file_location( + "check_data_logprob_gaps", _HERE / "check_data_logprob_gaps.py") +GAPS = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(GAPS) + + +class _GapFixture(unittest.TestCase): + BANNER = ( + b"== GLM C engine (glm_moe_dsa), cache=64 experts/layer | " + b"compute experts@4-bit dense@8-bit | idot: neon-i8mm ==\n") + LOADED = ( + b"loaded in 1.00s | resident dense: 1.00 MB | " + b"layers=78 experts=256 | MTP ACTIVE (draft=1)\n") + VALID = ( + b"ACCEPT 7 1\n" + b"ECHO 7 1 0 nan 0\n" + b"x\n" + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\n" + b"x\n" + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n" + ) + + @staticmethod + def echo(pos): + if pos == 0: + return b"ECHO 7 1 0 nan 0\nx\n" + return (f"ECHO 7 1 {pos} -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\n" + "x\n").encode("ascii") + + @classmethod + def transcript(cls, prompt=1, positions=None, emitted=1, done_prompt=None, + tps="0.10", hit="100.0", rss="10.00", flag=0): + if positions is None: + positions = list(range(prompt)) + if done_prompt is None: + done_prompt = prompt + parts = [f"ACCEPT 7 {prompt}\n".encode("ascii")] + parts.extend(cls.echo(pos) for pos in positions) + parts.extend([ + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n", + (f"DONE 7 STAT {emitted} {tps} {hit} {rss} " + f"{done_prompt} {flag}\n").encode("ascii"), + ]) + return b"".join(parts) + + def problems(self, blob, topk=2, vocab=3): + frames, framing = GAPS.parse_frames(blob) + data, echo, problems, _mode, _forms = GAPS.check( + frames, framing, "7", topk, vocab) + return data, echo, problems + + def full_problems(self, blob, topk=2, vocab=3): + """Like problems(), but also returns capture_mode and the numeric + form(s) observed -- for tests that need those two new signals.""" + frames, framing = GAPS.parse_frames(blob) + return GAPS.check(frames, framing, "7", topk, vocab) + + @staticmethod + def telemetry(cpu=b"AMD Ryzen 9", cores=b"16"): + return b"".join(( + b"HWINFO " + cores + b" 128.0 64.0 2 48.0 " + cpu + + b"|CUDA device x2\n", + b"TIERS 128 256 1024 48.00 32.50\n", + b"EMAP 2 2 00010203\n", + b"HITS 2 2 0f\n", + b"PROF 1.250 1 1 0.100 0.200 0.300 0.400 0.500 7\n", + )) + + @classmethod + def full_capture(cls): + startup = ( + b"\x01\x01READY\x01\x01\n" + b"STAT 0 0.00 0.0 10.00\n" + b"HWINFO 16 128.0 64.0 2 48.0 AMD Ryzen 9|CUDA device x2\n" + b"TIERS 128 256 1024 48.00 32.50\n" + b"EMAP 2 2 00010203\n" + ) + other = ( + b"ACCEPT 8 1\n" + b"ECHO 8 1 0 nan 0\nq\n" + b"DATA 8 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nq\n" + + cls.telemetry(cpu=b"AMD Ryzen 9") + + b"DONE 8 STAT 1 0.10 100.0 10.00 1 0\n" + ) + target = ( + b"ACCEPT 7 1\n" + b"ECHO 7 1 0 nan 0\nx\n" + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n" + + cls.telemetry(cpu=b"AMD Ryzen 9") + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n" + ) + return startup + other + target + + @classmethod + def complete_capture(cls, loaded=None): + return cls.BANNER + (loaded if loaded is not None else cls.LOADED) + cls.full_capture() + + +class GapCheckerEvidenceTests(_GapFixture): + """Ported unchanged from the checker's own evidence-consumer suite, + excluding the one method that scores a compiled fixture binary's + `%.17g` corpus (needs a build, not available here).""" + + def test_complete_request_passes(self): + for prompt in (1, 3): + with self.subTest(prompt=prompt): + data, echo, problems = self.problems(self.transcript(prompt)) + self.assertEqual((data, echo), (1, prompt)) + self.assertEqual(problems, []) + + def test_full_raw_mux_capture_passes_without_filtering(self): + data, echo, problems = self.problems(self.full_capture()) + self.assertEqual((data, echo), (1, 1)) + self.assertEqual(problems, []) + frames, _ = GAPS.parse_frames(self.full_capture()) + self.assertEqual(GAPS.capture_mode(frames), "ready-suffix") + + def test_complete_process_capture_passes_without_filtering(self): + data, echo, problems = self.problems(self.complete_capture()) + self.assertEqual((data, echo), (1, 1)) + self.assertEqual(problems, []) + frames, _ = GAPS.parse_frames(self.complete_capture()) + self.assertEqual(GAPS.capture_mode(frames), "full-process") + + for state, draft in ((b"ACTIVE", 0), (b"ACTIVE", 1), + (b"absent", 0), (b"absent", 2), + (b"DISABLED (multiplexed serve)", 0)): + loaded = ( + b"loaded in 1.00s | resident dense: 1.00 MB | " + b"layers=78 experts=256 | MTP " + state + + b" (draft=" + str(draft).encode("ascii") + b")\n") + _, _, problems = self.problems(self.complete_capture(loaded)) + self.assertEqual(problems, [], (state, draft, problems)) + + def test_complete_process_preamble_lifecycle_is_exact(self): + base = self.complete_capture() + suffix = self.full_capture() + cases = ( + suffix, + base.replace(self.BANNER, b"", 1), + base.replace(self.LOADED, b"", 1), + self.LOADED + self.BANNER + suffix, + self.BANNER + self.BANNER + self.LOADED + suffix, + self.BANNER + self.LOADED + self.LOADED + suffix, + b"unknown preamble\n" + base, + base.replace(b"idot: neon-i8mm", b"idot: fabricated", 1), + base.replace(b"loaded in 1.00s", b"loaded in 1.0s", 1), + ) + # The first case is the separately supported READY-suffix mode. + self.assertEqual(self.problems(cases[0])[2], []) + for blob in cases[1:]: + with self.subTest(blob=blob[:180]): + _, _, problems = self.problems(blob) + self.assertTrue(problems) + + def test_load_state_ranges_and_metrics_are_exact_in_full_capture(self): + base = self.LOADED + overflow = b"9" * 400 + b".00" + cases = ( + base.replace(b"ACTIVE (draft=1)", + b"DISABLED (multiplexed serve) (draft=1)"), + base.replace(b"ACTIVE", b"fabricated"), + base.replace(b"draft=1", b"draft=64"), + base.replace(b"layers=78", b"layers=0"), + base.replace(b"layers=78", b"layers=129"), + base.replace(b"experts=256", b"experts=0"), + base.replace(b"experts=256", b"experts=4097"), + base.replace(b"loaded in 1.00s", b"loaded in -0.01s"), + base.replace(b"loaded in 1.00s", b"loaded in nan s"), + base.replace(b"loaded in 1.00s", b"loaded in infs"), + base.replace(b"loaded in 1.00s", b"loaded in " + overflow + b"s"), + base.replace(b"loaded in 1.00s", b"loaded in 1.0s"), + base.replace(b"resident dense: 1.00 MB", + b"resident dense: -0.01 MB"), + base.replace(b"resident dense: 1.00 MB", + b"resident dense: nan MB"), + base.replace(b"resident dense: 1.00 MB", + b"resident dense: " + overflow + b" MB"), + base.replace(b"resident dense: 1.00 MB", + b"resident dense: 1.0 MB"), + ) + for loaded in cases: + with self.subTest(loaded=loaded): + _, _, problems = self.problems(self.complete_capture(loaded)) + self.assertTrue(any("malformed LOADED" in p for p in problems), + problems) + + for layers, experts in ((1, 1), (128, 4096)): + loaded = base.replace(b"layers=78", f"layers={layers}".encode()) + loaded = loaded.replace(b"experts=256", f"experts={experts}".encode()) + self.assertEqual(self.problems(self.complete_capture(loaded))[2], []) + + def test_emap_and_hits_producer_domains_are_exact(self): + for byte in (b"00", b"20", b"40", b"60", b"80", b"a0"): + _, problems = GAPS._global_header(b"EMAP 1 1 " + byte, 0) + self.assertEqual(problems, [], byte) + for byte in (b"21", b"61", b"a1"): + _, problems = GAPS._global_header(b"EMAP 1 1 " + byte, 0) + self.assertTrue(any("heat" in p for p in problems), problems) + for byte in (b"c0", b"ff"): + _, problems = GAPS._global_header(b"EMAP 1 1 " + byte, 0) + self.assertTrue(any("tier" in p for p in problems), problems) + + _, problems = GAPS._global_header(b"EMAP 0 2 ", 0) + self.assertEqual(problems, []) + _, problems = GAPS._global_header(b"EMAP 0 2 00", 0) + self.assertTrue(any("payload length" in p for p in problems), problems) + + for line in (b"HITS 2 2 0f", b"HITS 1 8 ff"): + _, problems = GAPS._global_header(line, 0) + self.assertEqual(problems, [], line) + for line in (b"HITS 2 2 1f", b"HITS 2 2 ff"): + _, problems = GAPS._global_header(line, 0) + self.assertTrue(any("padding" in p for p in problems), problems) + + def test_global_numeric_fields_never_collide_with_request_id(self): + capture = self.full_capture().replace( + b"HWINFO 16 128.0", b"HWINFO 7 128.0") + capture = capture.replace( + b"TIERS 128 256 1024", b"TIERS 7 256 1024") + capture = capture.replace(b"EMAP 2 2 00010203", b"EMAP 7 1 00010203040506") + capture = capture.replace(b"HITS 2 2 0f", b"HITS 7 1 7f") + data, echo, problems = self.problems(capture) + self.assertEqual((data, echo), (1, 1)) + self.assertEqual(problems, []) + + def test_malformed_or_misordered_global_records_never_pass(self): + base = self.full_capture() + cases = ( + base.replace(b"\x01\x01READY\x01\x01", b"READY", 1), + base.replace(b"STAT 0 0.00 0.0 10.00", b"STAT 0 0.0 0.0 10.00", 1), + base.replace(b"HWINFO 16 128.0", b"HWINFO 016 128.0", 1), + base.replace(b"HWINFO 16 128.0", b"HWINFO 16 128.0", 1), + base.replace(b"TIERS 128 256", b"TIERS 1_28 256", 1), + base.replace(b"EMAP 2 2 00010203", b"EMAP 2 2 0001020F", 1), + base.replace(b"HITS 2 2 0f", b"HITS 2 2 000f", 1), + base.replace(b"PROF 1.250 1 1", b"PROF 1.25 1 1", 1), + base.replace(b"PROF 1.250 1 1", b"PROF 1.250 01 1", 1), + base.replace(b"PROF 1.250 1 1", b"PROF 1.250 1 1_0", 1), + base.replace(b"\x01\x01READY\x01\x01\n", b"", 1), + base.replace(b"STAT 0 0.00 0.0 10.00\n", b"", 1), + base.replace( + b"\x01\x01READY\x01\x01\nSTAT 0 0.00 0.0 10.00\n", + b"STAT 0 0.00 0.0 10.00\n\x01\x01READY\x01\x01\n", 1), + base.replace( + b"\x01\x01READY\x01\x01\n", + b"\x01\x01READY\x01\x01\n\x01\x01READY\x01\x01\n", 1), + base.replace( + b"STAT 0 0.00 0.0 10.00\n", + b"STAT 0 0.00 0.0 10.00\nSTAT 0 0.00 0.0 10.00\n", 1), + ) + for blob in cases: + with self.subTest(blob=blob[:100]): + _, _, problems = self.problems(blob) + self.assertTrue(problems) + + def test_request_id_zero_refuses_before_record_matching(self): + frames, framing = GAPS.parse_frames(self.full_capture()) + data, echo, problems, mode, forms = GAPS.check( + frames, framing, "0", 2, 3) + self.assertEqual((data, echo), (0, 0)) + self.assertEqual(problems, ["requested id must be positive"]) + self.assertEqual(forms, frozenset()) + + with tempfile.TemporaryDirectory() as tmp: + capture = pathlib.Path(tmp) / "capture.raw" + capture.write_bytes(self.full_capture()) + proc = subprocess.run( + [sys.executable, str(pathlib.Path(GAPS.__file__)), str(capture), + "--id", "0", "--topk", "2", "--vocab", "3"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=False) + self.assertEqual(proc.returncode, 2) + self.assertIn(b"requested id must be positive", proc.stderr) + + def test_uint_primitive_and_coordinated_underscore_bite(self): + for token, value in ((b"0", 0), (b"1", 1), (b"10", 10), + (b"2147483647", 2147483647)): + with self.subTest(token=token): + self.assertEqual(GAPS._uint(token, "fixture"), value) + for token in (b"+1", b"01", b"1_0", b" 1", b"1 ", b"1\t0", + "١".encode("utf-8")): + with self.subTest(token=token): + with self.assertRaises(ValueError): + GAPS._uint(token, "fixture") + + parts = [b"ACCEPT 7 1_0\n"] + parts.extend(self.echo(pos) for pos in range(10)) + parts.extend(( + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n", + b"DONE 7 STAT 1 0.10 100.0 10.00 1_0 0\n", + )) + data, echo, problems = self.problems(b"".join(parts)) + self.assertEqual((data, echo), (1, 10)) + self.assertTrue(any("invalid ACCEPT prompt length" in p for p in problems)) + self.assertTrue(any("malformed DONE stats" in p for p in problems)) + self.assertFalse(any("ECHO positions" in p or "DONE prompt count" in p + for p in problems), problems) + + def test_malformed_framing_never_passes(self): + bad = [ + b"ACCEPT 7 1\nDATA 7 nope -2.7000000000000002 0\n", + b"ACCEPT 7 1\nDATA 7 5 -2.7000000000000002 0\nx\n", + b"ACCEPT 7 1\nDATA 7 1 -2.7000000000000002 0\nx", + b"ACCEPT 7 1", # missing header newline + ] + for blob in bad: + with self.subTest(blob=blob): + _, framing_problems = GAPS.parse_frames(blob) + self.assertTrue(framing_problems) + + def test_invalid_topk_tables_never_pass(self): + headers = [ + b"DATA 7 1 -2.7000000000000002 2 -1 -2.7000000000000002 1 -11.123455999999999", # out-of-range id + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 0 -11.123455999999999", # duplicate id + b"DATA 7 1 0.125 2 0 -2.7000000000000002 1 -11.123455999999999", # positive target lp + b"DATA 7 1 -2.7000000000000002 2 0 0.125 1 -11.123455999999999", # positive top-k lp + b"DATA 7 1 nan 2 0 -2.7000000000000002 1 -11.123455999999999", # pending nonfinite policy + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 3 -11.123455999999999", # id == vocab + ] + for header in headers: + blob = (b"ACCEPT 7 1\n" + self.echo(0) + header + b"\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + with self.subTest(header=header): + _, _, problems = self.problems(blob) + self.assertTrue(problems) + + def test_missing_lifecycle_or_partial_denominator_never_passes(self): + no_accept = self.VALID.split(b"\n", 1)[1] + no_done = self.VALID.rsplit(b"DONE", 1)[0] + wrong_done = self.VALID.replace(b"DONE 7 STAT 1", b"DONE 7 STAT 2") + error = self.VALID.replace( + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0", b"ERROR 7 ENGINE") + cases = ( + (no_accept, "expected exactly one ACCEPT"), + (no_done, "expected exactly one DONE"), + (wrong_done, "DONE emitted 2 != observed DATA 1"), + (error, "target request returned ERROR"), + ) + for blob, expected in cases: + with self.subTest(blob=blob): + _, _, problems = self.problems(blob) + self.assertTrue(any(expected in problem for problem in problems), problems) + + def test_unknown_targeted_kind_and_echo_after_data_never_pass(self): + unknown = self.VALID.replace( + b"DONE 7 STAT", b"MYSTERY 7 value\nDONE 7 STAT") + echo_after = ( + b"ACCEPT 7 1\n" + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\n" + b"x\n" + b"ECHO 7 1 0 nan 0\n" + b"x\n" + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n" + ) + for blob in (unknown, echo_after): + with self.subTest(blob=blob): + _, _, problems = self.problems(blob) + self.assertTrue(problems) + + def test_echo_denominator_and_order_are_exact(self): + cases = ([], [0], [0, 1], [0, 2], [0, 1, 1], [1, 0, 2], [0, 1, 2, 3]) + for positions in cases: + with self.subTest(positions=positions): + _, _, problems = self.problems(self.transcript(3, positions)) + self.assertTrue(any("ECHO positions" in p for p in problems), problems) + + def test_done_denominator_and_prompt_joins_bite_independently(self): + zero_data = self.VALID.replace( + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n", b"").replace( + b"DONE 7 STAT 1", b"DONE 7 STAT 0") + data, _, problems = self.problems(zero_data) + self.assertEqual(data, 0) + self.assertEqual( + [p for p in problems if p.startswith("no DATA frames")], + ["no DATA frames found for this id -- wrong id, or the run produced nothing"]) + _, _, problems = self.problems(self.transcript(1, emitted=2)) + self.assertEqual([p for p in problems if p.startswith("DONE emitted")], + ["DONE emitted 2 != observed DATA 1"]) + _, _, problems = self.problems(self.transcript(1, done_prompt=2)) + self.assertEqual([p for p in problems if p.startswith("DONE prompt")], + ["DONE prompt count 2 != ACCEPT 1"]) + + def test_done_domains_and_fixed_grammar_bite_independently(self): + cases = ( + {"tps": "-0.10"}, {"hit": "-0.1"}, {"hit": "100.1"}, + {"rss": "-0.01"}, {"tps": "0.1"}, {"hit": "0.00"}, + {"rss": "10.0"}, {"tps": "+0.10"}, {"tps": "01.00"}, + {"tps": "1e+00"}, {"tps": "nan"}, {"tps": "inf"}, + {"done_prompt": 0}, {"flag": 2}, + ) + for kwargs in cases: + with self.subTest(kwargs=kwargs): + _, _, problems = self.problems(self.transcript(1, **kwargs)) + matches = [p for p in problems if p.startswith("malformed DONE stats")] + self.assertEqual(len(matches), 1, problems) + + _, _, problems = self.problems(self.transcript( + 1, tps="-0.00", hit="-0.0", rss="-0.00")) + self.assertEqual(problems, []) + + def test_noncanonical_ascii_numeric_grammar_never_passes(self): + base = self.VALID + cases = ( + base.replace(b"ACCEPT 7 1", b"ACCEPT 7 +1"), + base.replace(b"ACCEPT 7 1", b"ACCEPT 7 1_0"), + base.replace(b"ECHO 7 1 0", b"ECHO 7 1 +0"), + base.replace(b"DATA 7 1", b"DATA 7 +1"), + base.replace(b" -2.7000000000000002 2 ", b" -0_125 2 ", 1), + base.replace(b" 1 -11.123455999999999", b" 01 -11.123455999999999"), + base.replace(b"DATA 7 1 ", b"DATA 7 1 "), + base.replace(b"DATA 7 1 ", b"DATA\t7 1 "), + base.replace(b"1 -11.123455999999999\nx\n", b"1 -11.123455999999999 \nx\n"), + base.replace(b"-2.7000000000000002 2", b"-1e-9999 2", 1), + base.replace(b"-2.7000000000000002 2", b"-1.00000000000000000 2", 1), + base.replace(b"-2.7000000000000002 2", b"-1e-9 2", 1), + base.replace(b"-2.7000000000000002 2", b"-1e--09 2", 1), + base.replace(b"DONE 7 STAT 1", b"DONE 7 STAT 01"), + base.replace(b" 1 0\n", b" +1 0\n"), + base.replace(b" 1 0\n", b" 1 00\n"), + base.replace(b"DATA 7 1", "DATA 7 ١".encode("utf-8")), + base.replace(b"DATA 7 1 -2.7000000000000002", b"DATA 7 1 -2.7000000000000002junk"), + ) + for blob in cases: + with self.subTest(blob=blob): + _, _, problems = self.problems(blob) + self.assertTrue(problems) + + +class PreambleGateFailsLoudTests(_GapFixture): + """The preamble gate never silently passes an unparsed BANNER/LOADED + line: `parse_engine_preamble` raising `PreambleError`, or the module's + own defensive `is None` branch, both surface as a named problem that + quotes the offending line.""" + + def test_unparsed_banner_is_a_named_failure_quoting_the_line(self): + bad_banner = self.BANNER.replace(b"idot: neon-i8mm", b"idot: bogus", 1) + kind, problems = GAPS._global_header(bad_banner.rstrip(b"\n"), 0) + self.assertEqual(kind, [b"BANNER"]) + self.assertEqual(len(problems), 1) + self.assertIn("malformed BANNER preamble", problems[0]) + self.assertIn(repr(bad_banner.rstrip(b"\n")), problems[0]) + + def test_unparsed_loaded_is_a_named_failure_quoting_the_line(self): + bad_loaded = self.LOADED.replace(b"MTP ACTIVE", b"MTP fabricated", 1) + kind, problems = GAPS._global_header(bad_loaded.rstrip(b"\n"), 0) + self.assertEqual(kind, [b"LOADED"]) + self.assertEqual(len(problems), 1) + self.assertIn("malformed LOADED preamble", problems[0]) + self.assertIn(repr(bad_loaded.rstrip(b"\n")), problems[0]) + + def test_preamble_returning_none_is_defensively_named_not_silently_passed(self): + # `_global_header` treats a None `parse_engine_preamble` result the + # same as a raised PreambleError: it is unreachable through the + # real BANNER/LOADED prefixes (they always parse or raise), so this + # pins the defensive branch directly by forcing that return value. + original = GAPS.parse_engine_preamble + GAPS.parse_engine_preamble = lambda text: None + try: + kind, problems = GAPS._global_header( + self.BANNER.rstrip(b"\n"), 0) + finally: + GAPS.parse_engine_preamble = original + self.assertEqual(kind, [b"BANNER"]) + self.assertEqual(len(problems), 1) + self.assertIn("malformed BANNER preamble", problems[0]) + self.assertIn(repr(self.BANNER.rstrip(b"\n")), problems[0]) + + def test_full_capture_with_unparsed_preamble_fails_the_whole_check(self): + bogus = self.complete_capture().replace( + b"idot: neon-i8mm", b"idot: bogus", 1) + _, _, problems = self.problems(bogus) + self.assertTrue(any("malformed BANNER preamble" in p for p in problems), + problems) + + +class DispatchPrefixCouplingTests(_GapFixture): + """`_global_header` independently re-tests the same two dispatch + prefixes (`"== GLM C engine"`, `"loaded in"`) that + `engine_evidence.parse_engine_preamble` already dispatches on + internally, instead of deriving them from the module -- see that + module's own docstring on the "loaded index" collision it accepts as + a documented characteristic because no engine anywhere in this tree + currently emits a colliding line. + + The two copies are byte-identical today, but nothing else enforces + that: a future edit to either literal alone would silently desync + "is this line an owned-preamble candidate" between this checker and + the module it is supposed to defer to, and this tool hard-aborts the + whole check on anything it decides is a malformed candidate. + + This test pins the coupling from the consumer side: it extracts the + prefixes from `_global_header`'s LIVE source (never retyped here) and + asserts the module still treats a genuine, fully well-formed line + built from each one as an owned-preamble candidate -- a non-None + result of the CORRECT kind, not merely "raised or didn't return + None". An earlier version of this test fed the bare prefix alone (an + incomplete line no real grammar accepts) and accepted a raised + `PreambleError` as proof of recognition; that is one-directional -- + a module whose dispatch is gutted to unconditionally raise + `PreambleError` (or unconditionally return some non-None value) for + EVERY input, recognizing nothing about either prefix specifically, + would still pass it. Confirmed by mutation: deleting the module's own + prefix dispatch outright left that version green (see the commit + history / worker report for the failing-then-fixed proof). Using a + genuine, complete, well-formed line and requiring the CORRECT typed + result closes that gap in both directions: a module that stops + recognizing the checker's literal returns `None` or the wrong `kind` + and this fails; a module that recognizes something it should not + would show up in the "vice versa" direction, which is out of scope + for a consumer-side pin (the module owns its own grammar). + `engine_evidence.py`/`test_engine_evidence.py` are not modified by + this fix -- it is entirely a P3-local test change. Contrast + `engine_evidence.py`'s own `IDOT_KERNELS` test, where a frozen + literal duplicate of the roster IS the mechanism because its whole + job is to catch drift in the module it pins; here the duplication + was incidental until this test made it intentional.""" + + @staticmethod + def _extract_dispatch_prefixes(): + # Match ONLY the leading dispatch condition + # (`if (line.startswith(b"...") or line.startswith(b"...")):`), + # not the later `kind = ... if line.startswith(b"==") ...` line + # that merely disambiguates a label within that branch. + source = inspect.getsource(GAPS._global_header) + match = re.search( + r'if\s*\(\s*line\.startswith\(b"([^"]*)"\)\s*or\s*' + r'line\.startswith\(b"([^"]*)"\)\s*\)\s*:', + source) + assert match is not None, ( + f"could not find _global_header's leading dispatch " + f"`if (line.startswith(b\"...\") or line.startswith(b\"...\"))`" + f" condition -- update this test's extraction to match its " + f"current shape before it can keep pinning the coupling:\n" + f"{source}") + return match.groups() + + def test_checker_dispatch_prefixes_stay_recognized_by_the_module(self): + banner_prefix, loaded_prefix = self._extract_dispatch_prefixes() + banner_line = self.BANNER.rstrip(b"\n").decode("ascii") + loaded_line = self.LOADED.rstrip(b"\n").decode("ascii") + + # Sanity: the fixture's own known-good lines must actually start + # with the literals just extracted, or the pairing below (banner + # literal <-> BANNER fixture, loaded literal <-> LOADED fixture) + # would be testing the wrong thing. + self.assertTrue(banner_line.startswith(banner_prefix), + (banner_prefix, banner_line)) + self.assertTrue(loaded_line.startswith(loaded_prefix), + (loaded_prefix, loaded_line)) + + banner_result = GAPS.parse_engine_preamble(banner_line) + self.assertIsNotNone( + banner_result, + f"engine_evidence.parse_engine_preamble no longer recognizes " + f"a genuine BANNER line built from _global_header's own " + f"{banner_prefix!r} literal (returned None) -- the two " + f"independently maintained copies of this prefix have " + f"diverged") + self.assertEqual(banner_result["kind"], "BANNER") + + loaded_result = GAPS.parse_engine_preamble(loaded_line) + self.assertIsNotNone( + loaded_result, + f"engine_evidence.parse_engine_preamble no longer recognizes " + f"a genuine LOADED line built from _global_header's own " + f"{loaded_prefix!r} literal (returned None) -- the two " + f"independently maintained copies of this prefix have " + f"diverged") + self.assertEqual(loaded_result["kind"], "LOADED") + + +class LegacyDataFrameGapTests(_GapFixture): + """A legacy 3-field DATA frame anywhere in the generation is the exact + gap this checker exists to catch, and it must fail loudly even when + it is not the last generated token.""" + + def test_legacy_three_field_data_frame_mid_generation_fails(self): + blob = ( + b"ACCEPT 7 1\n" + b"ECHO 7 1 0 nan 0\nx\n" + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n" # token 1: full channel + b"DATA 7 1\nx\n" # token 2: legacy gap + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n" # token 3: full channel + b"DONE 7 STAT 3 0.10 100.0 10.00 1 0\n" + ) + data, echo, problems = self.problems(blob) + self.assertEqual(data, 3) + self.assertTrue(any( + "GAP: legacy 3-field DATA frame #2" in p for p in problems), + problems) + + def test_wholly_dropped_data_frame_mid_generation_fails_via_done_count(self): + # The severer form of the same defect class: one generated token's + # DATA frame never reached stdout at all (not even degraded to 3 + # fields). Only the DONE-emitted-vs-observed-DATA cross-check can + # name this -- a checker with no concept of a DONE frame would + # pass it silently. + blob = ( + b"ACCEPT 7 1\n" + b"ECHO 7 1 0 nan 0\nx\n" + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n" # token 1 + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 1 -11.123455999999999\nx\n" # token 3 (token 2 missing) + b"DONE 7 STAT 3 0.10 100.0 10.00 1 0\n" + ) + data, echo, problems = self.problems(blob) + self.assertEqual(data, 2) + self.assertIn("DONE emitted 3 != observed DATA 2", problems) + + +class TopkCapAndTokenIdBoundsTests(_GapFixture): + """The numeric tail enforces k == min(--topk, --vocab) and + 0 <= tid < vocab, reporting the first offending frame.""" + + def test_k_above_expected_but_within_wire_cap_never_passes(self): + # vocab=40, --topk=20 => expected_k = min(20, 40) = 20; the frame + # advertises k=25, which is <=32 (the wire's own hard cap) but + # above the expected value for this request. + pairs = " ".join(f"{i} -2.7000000000000002" for i in range(25)) + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + f"DATA 7 1 -2.7000000000000002 25 {pairs}\n".encode("ascii") + b"x\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + _, _, problems = self.problems(blob, topk=20, vocab=40) + self.assertTrue(any( + "DATA top-k 25 != expected 20" in p for p in problems), problems) + + def test_k_above_wire_hard_cap_of_32_never_passes(self): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -2.7000000000000002 33 0 -0.1\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + _, _, problems = self.problems(blob, topk=32, vocab=40) + self.assertTrue(any( + "malformed DATA numeric fields" in p for p in problems), problems) + + def test_token_id_equal_to_vocab_never_passes(self): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -2.7000000000000002 2 0 -2.7000000000000002 3 -0.25\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + _, _, problems = self.problems(blob, topk=2, vocab=3) + self.assertTrue(any( + "token id 3 outside [0,3)" in p for p in problems), problems) + + def test_token_id_well_above_vocab_never_passes(self): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -2.7000000000000002 1 999 -2.7000000000000002\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + _, _, problems = self.problems(blob, topk=1, vocab=40) + self.assertTrue(any( + "token id 999 outside [0,40)" in p for p in problems), problems) + + +class FrameOrderLifecycleGateTests(_GapFixture): + """`check()`'s full-process startup order gate (the `required` tuple + binding BANNER/LOADED/READY/STAT/HWINFO/TIERS/EMAP to frames 0..6) is + exercised in every existing full-process test, but always alongside + several other simultaneous defects or via a bare `assertTrue(problems)` + -- so no existing test's pass/fail status depends on any ONE of the + seven `required` entries individually. Ablating any single entry from + `required` (deleting its `(kind, index)` pair) left the pre-fix-round + 105-test suite green: the same vacuous-gate shape as `IDOT_KERNELS` + (a gate whose expectation and its only check come from the same, + un-anchored place). + + Fix shape, matching that precedent: an independent anchor (a frozen + copy of the expected order, written here rather than derived from + `check()`'s own construction, checked against what `check()` actually + contains) plus per-ordinal coverage (one test per entry, each proving + that ENTRY's own specific "expected ... as frame N" message -- unique + text per kind, so only that entry's check can produce it -- actually + fires). `c/colibri.c`'s own emission order is NOT consulted here: that + cross-check (transcribed wire format vs C source) is a separate, + cross-cutting, pre-existing gap (flagged elsewhere) affecting every + literal in both new Python modules, not something to fix piecemeal in + this one gate. + """ + + # Independent anchor: written by hand, not copied from `check()`'s own + # construction of `required`. `test_required_order_matches_the_frozen_anchor` + # cross-checks this against the live tuple `check()` actually uses, so a + # future reorder/rename in the code is caught here even before any + # per-ordinal test runs. + EXPECTED_ORDER = ( + (b"BANNER", 0), (b"LOADED", 1), (b"READY", 2), (b"STAT", 3), + (b"HWINFO", 4), (b"TIERS", 5), (b"EMAP", 6), + ) + + @staticmethod + def _extract_required_tuple(): + """Parse the literal `required = (...)` tuple out of `check()`'s + live source by counting parens to the matching close -- not a + regex snippet, so it does not care about internal formatting -- + then `ast.literal_eval` it. This is reading the code's actual + current behavior, not retyping an assumption about it.""" + source = inspect.getsource(GAPS.check) + marker = "required = (" + start = source.index(marker) + paren_start = start + len("required = ") + depth = 0 + i = paren_start + while True: + if source[i] == "(": + depth += 1 + elif source[i] == ")": + depth -= 1 + if depth == 0: + end = i + 1 + break + i += 1 + return ast.literal_eval(source[paren_start:end]) + + def test_required_order_matches_the_frozen_anchor(self): + self.assertEqual(self._extract_required_tuple(), self.EXPECTED_ORDER) + + def _assert_drop_bites(self, drop_line, kind, expected_index, template): + blob = self.complete_capture().replace(drop_line, b"", 1) + _, _, problems = self.problems(blob) + needle = f"expected {template} {kind} as frame {expected_index}" + self.assertTrue(any(needle in p for p in problems), + f"{needle!r} not found in {problems!r}") + + def test_dropping_banner_breaks_its_own_ordinal_check(self): + self._assert_drop_bites(self.BANNER, "BANNER", 0, "one") + + def test_dropping_loaded_breaks_its_own_ordinal_check(self): + self._assert_drop_bites(self.LOADED, "LOADED", 1, "one") + + def test_dropping_ready_breaks_its_own_ordinal_check(self): + self._assert_drop_bites(b"\x01\x01READY\x01\x01\n", "READY", 2, "one") + + def test_dropping_stat_breaks_its_own_ordinal_check(self): + self._assert_drop_bites(b"STAT 0 0.00 0.0 10.00\n", "STAT", 3, "one") + + def test_dropping_hwinfo_breaks_its_own_ordinal_check(self): + self._assert_drop_bites( + b"HWINFO 16 128.0 64.0 2 48.0 AMD Ryzen 9|CUDA device x2\n", + "HWINFO", 4, "startup") + + def test_dropping_tiers_breaks_its_own_ordinal_check(self): + self._assert_drop_bites( + b"TIERS 128 256 1024 48.00 32.50\n", "TIERS", 5, "startup") + + def test_dropping_emap_breaks_its_own_ordinal_check(self): + self._assert_drop_bites(b"EMAP 2 2 00010203\n", "EMAP", 6, "startup") + + +class CaptureModeLiteralTests(unittest.TestCase): + """`capture_mode` names the four explicit transcript shapes.""" + + def test_empty_transcript_is_request_only(self): + self.assertEqual(GAPS.capture_mode([]), "request-only") + + def test_accept_first_frame_is_request_only(self): + frames, _ = GAPS.parse_frames(b"ACCEPT 7 1\n") + self.assertEqual(GAPS.capture_mode(frames), "request-only") + + def test_banner_first_frame_is_full_process(self): + frames, _ = GAPS.parse_frames( + b"== GLM C engine (glm_moe_dsa), cache=64 experts/layer | " + b"compute experts@4-bit dense@8-bit | idot: neon-i8mm ==\n" + b"ACCEPT 7 1\n") + self.assertEqual(GAPS.capture_mode(frames), "full-process") + + def test_ready_first_frame_is_ready_suffix(self): + frames, _ = GAPS.parse_frames( + b"\x01\x01READY\x01\x01\nACCEPT 7 1\n") + self.assertEqual(GAPS.capture_mode(frames), "ready-suffix") + + def test_global_after_a_non_global_first_frame_is_invalid(self): + frames, _ = GAPS.parse_frames( + b"ACCEPT 7 1\n\x01\x01READY\x01\x01\n") + self.assertEqual(GAPS.capture_mode(frames), "invalid") + + +class FixedMetricAndUintBoundaryTests(unittest.TestCase): + """`_fixed_metric`'s lower/upper bounds and `_uint`'s maximum + argument are literal, inclusive boundaries.""" + + def test_fixed_metric_lower_bound_is_inclusive(self): + self.assertEqual(GAPS._fixed_metric(b"0.0", 1, "x", lower=0.0), 0.0) + with self.assertRaises(ValueError): + GAPS._fixed_metric(b"-0.1", 1, "x", lower=0.0) + + def test_fixed_metric_upper_bound_is_inclusive(self): + self.assertEqual( + GAPS._fixed_metric(b"100.0", 1, "x", upper=100.0), 100.0) + with self.assertRaises(ValueError): + GAPS._fixed_metric(b"100.1", 1, "x", upper=100.0) + + def test_fixed_metric_place_count_is_exact(self): + with self.assertRaises(ValueError): + GAPS._fixed_metric(b"1.00", 1, "x") + with self.assertRaises(ValueError): + GAPS._fixed_metric(b"1.0", 2, "x") + + def test_uint_maximum_is_inclusive(self): + self.assertEqual(GAPS._uint(b"32", "x", 32), 32) + with self.assertRaises(ValueError): + GAPS._uint(b"33", "x", 32) + + + + +class NumericGrammarFormTests(_GapFixture): + """The numeric grammar accepts both the fixed %.6f form dev's engine + prints today and the earlier %.17g form + (plus an exact nan/inf/-inf spelling), pinned with non-dyadic literal + values (-0.3, -2.7, -11.123456) so the two encodings are actually + distinguishable by their token spelling and not merely by the float + each happens to parse to.""" + + def test_fixed6_form_is_accepted_and_reported(self): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -0.300000 2 0 -2.700000 1 -11.123456\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + data, echo, problems, mode, forms = self.full_problems(blob) + self.assertEqual(problems, []) + self.assertEqual(forms, collections.Counter({"fixed6": 3})) + + def test_c17g_form_is_accepted_and_reported(self): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -2.7000000000000002 2 0 -11.123455999999999 " + b"1 -0.29999999999999999\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + data, echo, problems, mode, forms = self.full_problems(blob) + self.assertEqual(problems, []) + self.assertEqual(forms, collections.Counter({"c17g": 3})) + + def test_a_token_matching_both_grammars_is_ambiguous_and_still_accepted(self): + # '%.17g' % -1.234567 == '-1.234567', which is ALSO the exact + # six-decimal spelling of that same double: there is no way to + # tell, from the token alone, which engine format produced it. + # An earlier version of this check treated a tail containing both + # an unambiguous fixed6 token and an unambiguous c17g token as a + # "mixed forms" error; that rule was removed because per-token + # classification is inherently ambiguous and the rule produced + # false rejections on real %.17g transcripts (some of whose + # tokens are, coincidentally, exact six-decimal spellings too). + # Nothing about mixing forms is rejected anymore -- only a token + # matching NEITHER grammar is. + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -1.234567 2 0 -2.7000000000000002 1 -0.300000\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + data, echo, problems, mode, forms = self.full_problems(blob) + self.assertEqual(problems, []) + self.assertEqual(forms, collections.Counter( + {"ambiguous": 1, "c17g": 1, "fixed6": 1})) + + def test_special_nan_inf_tokens_are_syntactically_valid_then_flagged(self): + # "nan"/"inf"/"-inf" are exact libc renderings either format's + # snprintf can emit for a non-finite double: syntactically valid + # under both grammars, but still fail the finite/non-positive + # semantic check that applies regardless of which form carried it. + for special in (b"nan", b"inf", b"-inf"): + with self.subTest(special=special): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 " + special + b" 1 0 -0.300000\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + _, _, problems, _, forms = self.full_problems(blob, topk=1) + self.assertTrue(any( + "target logprob is not finite/non-positive" in p + for p in problems), problems) + self.assertIn("special", forms) + + def test_token_matching_neither_form_never_passes(self): + blob = ( + b"ACCEPT 7 1\n" + self.echo(0) + + b"DATA 7 1 -0.3 2 0 -2.7 1 -11.123456\nx\n" + + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + _, _, problems = self.problems(blob) + self.assertTrue(any( + "malformed DATA numeric fields" in p for p in problems), problems) + + +class ByteOffsetTests(unittest.TestCase): + """Every reported problem cites the byte offset of + the OFFENDING frame's own header line, not the frame that follows it, + and numeric-tail problems (previously offset-less) now carry one too. + """ + + def test_malformed_global_offset_cites_its_own_line_not_the_next_one(self): + first = b"\x01\x01READY\x01\x01\n" + bad_stat = b"STAT 0 0.0 0.0 10.00\n" # "0.0" should be "0.00" + blob = first + bad_stat + b"ACCEPT 7 1\n" + _, problems = GAPS.parse_frames(blob) + self.assertEqual(len(problems), 1) + self.assertIn(f"byte {len(first)}", problems[0]) + self.assertNotIn(f"byte {len(first) + len(bad_stat)}", problems[0]) + + def test_header_fields_offset_cites_its_own_line_not_the_next_one(self): + first = b"ACCEPT 7 1\n" + bad = b" ACCEPT 8 1\n" # leading space: noncanonical header + blob = first + bad + _, problems = GAPS.parse_frames(blob) + self.assertEqual(len(problems), 1) + self.assertIn(f"byte {len(first)}", problems[0]) + self.assertNotIn(f"byte {len(first) + len(bad)}", problems[0]) + + def test_numeric_tail_problem_cites_the_data_frames_own_header(self): + prefix = b"ACCEPT 7 1\n" + b"ECHO 7 1 0 nan 0\nx\n" + # positive target lp: a semantic failure inside _numeric_tail. + bad_data = b"DATA 7 1 0.300000 2 0 -0.300000 1 -0.300000\n" + blob = prefix + bad_data + b"x\n" + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n" + frames, framing = GAPS.parse_frames(blob) + _, _, problems, _, _ = GAPS.check(frames, framing, "7", 2, 3) + offset = len(prefix) + self.assertTrue(any( + f"byte {offset}" in p and + "target logprob is not finite/non-positive" in p + for p in problems), problems) + + def test_offset_of_a_non_first_offending_frame_is_hand_verified(self): + # Three frames precede the offending one: ACCEPT (11 bytes: the + # 10-character line "ACCEPT 7 1" plus its newline), an ECHO at + # position 0 (17 bytes: "ECHO 7 1 0 nan 0" plus its newline), and + # that ECHO's one-byte "x" payload plus its terminator (2 bytes). + # 11 + 17 + 2 = 30 is where the malformed DONE header starts -- + # a hand count, not a value read back from the module under test. + line0 = b"ACCEPT 7 1\n" + line1 = b"ECHO 7 1 0 nan 0\n" + payload1 = b"x\n" + self.assertEqual(len(line0), 11) + self.assertEqual(len(line1), 17) + self.assertEqual(len(payload1), 2) + bad_done = b"DONE 7 STAT 1 0.10 100.0 10.00 1\n" # missing 1 field + blob = line0 + line1 + payload1 + bad_done + frames, framing = GAPS.parse_frames(blob) + self.assertEqual(len(frames), 3) + self.assertEqual(frames[2][2], 30) + _, _, problems, _, _ = GAPS.check(frames, framing, "7", 2, 3) + self.assertTrue(any( + "malformed DONE frame at byte 30" in p for p in problems), problems) + + +class PreambleGateWiredIntoMainTests(unittest.TestCase): + """capture_mode() is actually wired into check(), so + a transcript opening with an unrecognized (garbage) line fails through + main()'s own CLI path, not only via a function called directly; and + the resolved capture mode is always reported, never silent.""" + + def test_garbage_preamble_fails_via_main_cli_path(self): + blob = ( + b"GARBAGE NOT A REAL PREAMBLE\n" + b"ACCEPT 7 1\n" + b"ECHO 7 1 0 nan 0\nx\n" + b"DATA 7 1 -0.300000 2 0 -0.300000 1 -0.300000\nx\n" + b"DONE 7 STAT 1 0.10 100.0 10.00 1 0\n") + with tempfile.TemporaryDirectory() as tmp: + capture = pathlib.Path(tmp) / "capture.raw" + capture.write_bytes(blob) + proc = subprocess.run( + [sys.executable, str(pathlib.Path(GAPS.__file__)), str(capture), + "--id", "7", "--topk", "2", "--vocab", "3"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=False) + self.assertEqual(proc.returncode, 1) + self.assertIn(b"unrecognized frame opens the transcript", proc.stdout) + + def test_capture_mode_and_numeric_form_are_reported_in_main_summary(self): + with tempfile.TemporaryDirectory() as tmp: + capture = pathlib.Path(tmp) / "capture.raw" + capture.write_bytes(_GapFixture.VALID) + proc = subprocess.run( + [sys.executable, str(pathlib.Path(GAPS.__file__)), str(capture), + "--id", "7", "--topk", "2", "--vocab", "3"], + stdout=subprocess.PIPE, stderr=subprocess.PIPE, check=False) + self.assertEqual(proc.returncode, 0) + self.assertIn(b"capture_mode=request-only", proc.stdout) + self.assertIn(b"numeric form tally: c17g=3", proc.stdout) + + def test_invalid_capture_mode_is_named_via_check(self): + # A stray global record with no BANNER/READY lead frame at all. + blob = b"STAT 0 0.00 0.0 10.00\nACCEPT 7 1\n" + frames, framing = GAPS.parse_frames(blob) + self.assertEqual(GAPS.capture_mode(frames), "invalid") + _, _, problems, mode, _ = GAPS.check(frames, framing, "7", 2, 3) + self.assertEqual(mode, "invalid") + self.assertTrue(any("invalid capture mode" in p for p in problems), problems) + + + +class EngineFormatCorpusTests(_GapFixture): + """A synthesized corpus in the engine's real wire format: every + numeric token below is spelled exactly as C's `snprintf(..., "%.6f + %d", ...)` (dev's shipped `logprob_tail()`) would spell it -- not + hand-picked round numbers, but 210 distinct, non-dyadic values, so no + single lucky literal is doing the proving. It must pass at head, and + every one of its numeric tokens is independently confirmed to be + something the strict %.17g-only grammar this checker's numeric + parsing predates would have rejected outright. + """ + + N_FRAMES = 210 + VOCAB = 300 + TOPK = 2 + + @classmethod + def _corpus(cls, seed=20260904): + rng = random.Random(seed) + parts = [b"ACCEPT 7 1\n", b"ECHO 7 1 0 nan 0\nx\n"] + tokens = [] + for _ in range(cls.N_FRAMES): + target_lp = -abs(rng.uniform(1e-6, 20.0)) + target_token = f"{target_lp:.6f}".encode("ascii") + tokens.append(target_token) + pair_ids = rng.sample(range(cls.VOCAB), cls.TOPK) + pieces = [target_token, str(cls.TOPK).encode("ascii")] + for tid in pair_ids: + token_lp = -abs(rng.uniform(1e-6, 20.0)) + token_lp_token = f"{token_lp:.6f}".encode("ascii") + tokens.append(token_lp_token) + pieces.append(str(tid).encode("ascii")) + pieces.append(token_lp_token) + parts.append(b"DATA 7 1 " + b" ".join(pieces) + b"\nx\n") + parts.append( + f"DONE 7 STAT {cls.N_FRAMES} 0.10 100.0 10.00 1 0\n".encode("ascii")) + return b"".join(parts), tokens + + def test_engine_format_corpus_of_210_frames_passes(self): + blob, tokens = self._corpus() + self.assertGreaterEqual(len(tokens), self.N_FRAMES) + frames, framing = GAPS.parse_frames(blob) + data, echo, problems, mode, forms = GAPS.check( + frames, framing, "7", self.TOPK, self.VOCAB) + self.assertEqual(problems, []) + self.assertEqual(data, self.N_FRAMES) + # Every token here is a genuine %.6f emission, so none can be + # classified as c17g-only; some nonetheless land on an exact + # %.17g spelling too (the false-positive "mixed forms" case a + # since-removed rejection used to misfire on) and are tallied + # "ambiguous" rather than "fixed6" -- informational only, and + # this test's PASS/problems assertions above already prove that + # tally has no bearing on the verdict. + self.assertEqual(forms["c17g"], 0) + self.assertGreater(forms["ambiguous"], 0) + self.assertEqual(sum(forms.values()), len(tokens)) + + def test_engine_format_corpus_values_are_non_dyadic(self): + # A dyadic value (an exact binary fraction, e.g. 0.125) would make + # the fixed6/c17g distinction uninteresting for that token, since + # both forms could spell it exactly. Confirm the corpus avoids + # that by construction: none of its %.6f tokens round-trips + # through Python's own dyadic-fraction check. + _, tokens = self._corpus() + dyadic = 0 + for token in tokens: + value = float(token) + # A double is dyadic (exactly binary-fraction-representable at + # 6 decimal places) iff multiplying by 1e6 and rounding loses + # nothing AND the resulting numerator's lowest set bits divide + # evenly -- simpler and just as decisive here: a dyadic value + # would print identically under %.17g and %.6f once trailing + # zeros are accounted for, which none of these do (see the + # next test) -- this test instead confirms none is a "clean" + # few-bits-of-mantissa value like *.0, *.5, *.25, *.125. + frac = abs(value) - int(abs(value)) + eighths = frac * 8 + if abs(eighths - round(eighths)) < 1e-9: + dyadic += 1 + self.assertEqual(dyadic, 0) + + def test_engine_format_corpus_would_fail_a_c17g_only_grammar(self): + # Every token in the corpus is a %.6f spelling. Whether any ONE + # such token also happens to be its own double's shortest %.17g + # round-trip spelling is unpredictable per-token (it depends on + # that double's neighborhood, not on the fact that it came from + # %.6f) -- but the module docstring's actual claim is about the + # TRANSCRIPT, not any single token: one rejected token anywhere + # is enough to fail the whole check, since `_numeric_tail` bails + # out of the entire frame the moment its numeric parse raises. + # Confirm most tokens are rejected outright by the standalone + # %.17g parser this module still carries (`_c17g`, kept for the + # earlier engine's own emitted form), and that at least one + # rejection lands inside a real DATA frame of the corpus -- + # which is what actually dooms the whole transcript under a + # %.17g-only grammar. + blob, tokens = self._corpus() + rejected = 0 + for token in tokens: + try: + GAPS._c17g(token, "corpus token") + except ValueError: + rejected += 1 + self.assertGreater(rejected, len(tokens) // 2, (rejected, len(tokens))) + + frame_rejected = False + for line in blob.split(b"\n"): + if not line.startswith(b"DATA "): + continue + fields = line.split(b" ") + for field in fields[3:]: + try: + GAPS._c17g(field, "corpus token") + except ValueError: + frame_rejected = True + break + if frame_rejected: + break + self.assertTrue( + frame_rejected, + "expected at least one DATA frame in the corpus to contain a " + "token the %.17g-only parser refuses") + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_convert_gguf_to_olmoe.py b/c/tests/test_convert_gguf_to_olmoe.py new file mode 100644 index 000000000..4bb9d31a7 --- /dev/null +++ b/c/tests/test_convert_gguf_to_olmoe.py @@ -0,0 +1,338 @@ +#!/usr/bin/env python3 +"""tools/convert_gguf_to_olmoe.py: end-to-end conversion of a tiny synthetic GGUF. + +Builds a minimal OLMoE-shaped GGUF (all F16/F32, no K-quants) with the smallest +dims the engine accepts, runs the real CLI, and asserts the container c/olmoe.c +expects: config.json keys, tokenizer.json reconstruction, the dense name +mapping, and per-expert merged_weight (I8) + .qs (F32) at the exact sizes the +engine requires. Also covers --dry-run (writes nothing, creates no dir) and the +refusals for a non-OLMoE GGUF / unknown tensor; --remove-source-file removes +the source only after full success. +""" + +import json +import struct +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +import importlib + +try: + import numpy as np + for _dependency in ("safetensors", "safetensors.numpy"): + importlib.import_module(_dependency) +except ImportError as exc: + raise unittest.SkipTest("numpy/safetensors not installed: %s" % exc) + +TESTDIR = Path(__file__).resolve().parent +TOOLS = TESTDIR.parent / "tools" +sys.path.insert(0, str(TOOLS)) +sys.path.insert(0, str(TESTDIR)) + +from gguf_fixture import ( # noqa: E402 (encoder helpers reused verbatim) + METADATA_ARRAY, + METADATA_BOOL, + METADATA_FLOAT32, + METADATA_STRING, + METADATA_UINT32, + _encode_string, + _encode_value, +) + +import convert_gguf_to_olmoe # noqa: E402 + +# tiny but engine-valid dims +HIDDEN, INTER, N_EXPERTS, LAYERS, VOCAB, HEADS = 8, 4, 2, 1, 16, 2 + + +def _align(value, alignment): + return (value + alignment - 1) // alignment * alignment + + +def _f16_bytes(values): + return np.asarray(values, dtype=np.float16).tobytes() + + +def _f32_bytes(values): + return np.asarray(values, dtype=np.float32).tobytes() + + +def build_tiny_olmoe_gguf(path, arch="olmoe", extra_tensor=None, override=None): + rng = np.random.default_rng(20260825) + override = override or {} + + def m2d(hf_shape): + return list(reversed(hf_shape)) + + tensors = [] + # HF shape -> GGUF dims (reversed). Dense matrices F16, 1-D norms F32. + def add_gguf(name, gguf_dims, ggml_type, payload): + tensors.append({"name": name, "dims": gguf_dims, + "ggml_type": ggml_type, "payload": override.get(name, payload)}) + + add_gguf("token_embd.weight", m2d([VOCAB, HIDDEN]), 1, + _f16_bytes(rng.normal(size=VOCAB * HIDDEN))) + add_gguf("output.weight", m2d([VOCAB, HIDDEN]), 1, + _f16_bytes(rng.normal(size=VOCAB * HIDDEN))) + add_gguf("output_norm.weight", [HIDDEN], 0, _f32_bytes(rng.normal(size=HIDDEN))) + for layer in range(LAYERS): + add_gguf("blk.%d.attn_norm.weight" % layer, [HIDDEN], 0, + _f32_bytes(rng.normal(size=HIDDEN))) + add_gguf("blk.%d.ffn_norm.weight" % layer, [HIDDEN], 0, + _f32_bytes(rng.normal(size=HIDDEN))) + for w in ("q", "k", "v", "output"): + add_gguf("blk.%d.attn_%s.weight" % (layer, w), m2d([HIDDEN, HIDDEN]), 1, + _f16_bytes(rng.normal(size=HIDDEN * HIDDEN))) + add_gguf("blk.%d.attn_q_norm.weight" % layer, [HIDDEN], 0, + _f32_bytes(rng.normal(size=HIDDEN))) + add_gguf("blk.%d.attn_k_norm.weight" % layer, [HIDDEN], 0, + _f32_bytes(rng.normal(size=HIDDEN))) + add_gguf("blk.%d.ffn_gate_inp.weight" % layer, m2d([N_EXPERTS, HIDDEN]), 1, + _f16_bytes(rng.normal(size=N_EXPERTS * HIDDEN))) + add_gguf("blk.%d.ffn_gate_exps.weight" % layer, m2d([N_EXPERTS, INTER, HIDDEN]), 1, + _f16_bytes(rng.normal(size=N_EXPERTS * INTER * HIDDEN))) + add_gguf("blk.%d.ffn_up_exps.weight" % layer, m2d([N_EXPERTS, INTER, HIDDEN]), 1, + _f16_bytes(rng.normal(size=N_EXPERTS * INTER * HIDDEN))) + add_gguf("blk.%d.ffn_down_exps.weight" % layer, m2d([N_EXPERTS, HIDDEN, INTER]), 1, + _f16_bytes(rng.normal(size=N_EXPERTS * INTER * HIDDEN))) + if extra_tensor: + tensors.append(extra_tensor) + + metadata = [ + (METADATA_STRING, "general.architecture", arch), + (METADATA_UINT32, "general.alignment", 32), + (METADATA_UINT32, "olmoe.embedding_length", HIDDEN), + (METADATA_UINT32, "olmoe.block_count", LAYERS), + (METADATA_UINT32, "olmoe.attention.head_count", HEADS), + (METADATA_UINT32, "olmoe.attention.head_count_kv", HEADS), + (METADATA_UINT32, "olmoe.expert_count", N_EXPERTS), + (METADATA_UINT32, "olmoe.expert_used_count", 2), + (METADATA_UINT32, "olmoe.feed_forward_length", INTER), + (METADATA_FLOAT32, "olmoe.rope.freq_base", 10000.0), + (METADATA_FLOAT32, "olmoe.attention.layer_norm_rms_epsilon", 1e-5), + (METADATA_ARRAY, "tokenizer.ggml.tokens", + (METADATA_STRING, ["aa", "bb", "", "Ġt"])), + (METADATA_ARRAY, "tokenizer.ggml.token_type", + (METADATA_UINT32, [1, 1, 3, 4])), + (METADATA_ARRAY, "tokenizer.ggml.merges", + (METADATA_STRING, ["Ġ t"])), + (METADATA_UINT32, "tokenizer.ggml.bos_token_id", 2), + (METADATA_UINT32, "tokenizer.ggml.eos_token_id", 2), + (METADATA_BOOL, "tokenizer.ggml.add_bos_token", False), + (METADATA_BOOL, "tokenizer.ggml.add_eos_token", False), + ] + return write_gguf(path, metadata, tensors) + + +def write_gguf(path, metadata, tensors): + alignment = 32 + out = b"GGUF" + out += struct.pack("": 2, "Ġt": 3}) + + present = set() + from safetensors import safe_open + for shard in sorted(self.out.glob("model-*.safetensors")): + with safe_open(str(shard), framework="np") as handle: + present.update(handle.keys()) + + self.assertEqual(present, expected_names()) + + with safe_open(str(next(self.out.glob("model-*.safetensors"))), framework="np") as handle: + merged = handle.get_tensor(base_merge_name(0, 0)) + qs = handle.get_tensor("model.layers.0.mlp.experts.0.qs") + self.assertEqual(merged.dtype, np.int8) + self.assertEqual(qs.dtype, np.float32) + self.assertEqual(merged.size, INTER * HIDDEN * 2 + HIDDEN * INTER) + self.assertEqual(qs.size, INTER + INTER + HIDDEN) + + def test_dry_run_creates_nothing(self): + result = run_converter(self.gguf, self.out, "--dry-run") + self.assertEqual(result.returncode, 0, result.stderr) + self.assertFalse(self.out.exists()) + + def test_refuses_wrong_architecture(self): + wrong = self.dir / "wrong.gguf" + build_tiny_olmoe_gguf(str(wrong), arch="llama") + result = run_converter(wrong, self.out) + self.assertEqual(result.returncode, 1) + self.assertIn("general.architecture", result.stderr) + self.assertFalse(self.out.exists()) + + def test_refuses_unknown_tensor(self): + weird = self.dir / "weird.gguf" + build_tiny_olmoe_gguf(str(weird), extra_tensor={ + "name": "blk.0.not_a_real_tensor", + "dims": [HIDDEN], + "ggml_type": 0, + "payload": _f32_bytes(np.zeros(HIDDEN)), + }) + result = run_converter(weird, self.out) + self.assertEqual(result.returncode, 1) + self.assertIn("unrecognized tensors", result.stderr) + + def test_remove_source_only_on_success(self): + result = run_converter(self.gguf, self.out, "--remove-source-file") + self.assertEqual(result.returncode, 0, result.stderr) + self.assertFalse(self.gguf.exists()) + self.assertTrue((self.out / "config.json").is_file()) + + def test_qk_left_as_stored_by_default(self): + rng = np.random.default_rng(99) + hf_q_proj = rng.normal(size=(HIDDEN, HIDDEN)).astype(np.float16) + permuted_q_proj = _rope_permute(hf_q_proj, HEADS) + self.assertFalse(np.array_equal(permuted_q_proj, hf_q_proj)) + build_tiny_olmoe_gguf(str(self.gguf), override={ + "blk.0.attn_q.weight": permuted_q_proj.tobytes(), + }) + + result = run_converter(self.gguf, self.out) + self.assertEqual(result.returncode, 0, result.stderr) + got = self._read("model.layers.0.self_attn.q_proj.weight") + # no flag: the stored (permuted) layout is passed through unchanged + np.testing.assert_array_equal(got, permuted_q_proj) + + def test_qk_permuted_flag_undoes_rope_permutation(self): + rng = np.random.default_rng(99) + hf_q_proj = rng.normal(size=(HIDDEN, HIDDEN)).astype(np.float16) + hf_k_proj = rng.normal(size=(HIDDEN, HIDDEN)).astype(np.float16) + build_tiny_olmoe_gguf(str(self.gguf), override={ + "blk.0.attn_q.weight": _rope_permute(hf_q_proj, HEADS).tobytes(), + "blk.0.attn_k.weight": _rope_permute(hf_k_proj, HEADS).tobytes(), + }) + + result = run_converter(self.gguf, self.out, "--qk-permuted") + self.assertEqual(result.returncode, 0, result.stderr) + np.testing.assert_array_equal( + self._read("model.layers.0.self_attn.q_proj.weight"), hf_q_proj) + np.testing.assert_array_equal( + self._read("model.layers.0.self_attn.k_proj.weight"), hf_k_proj) + + def _read(self, name): + from safetensors import safe_open + for shard in sorted(self.out.glob("model-*.safetensors")): + with safe_open(str(shard), framework="np") as handle: + if name in handle.keys(): + return handle.get_tensor(name) + raise KeyError(name) + + def test_overwrite_rewrites_existing_container(self): + self.assertEqual(run_converter(self.gguf, self.out).returncode, 0) + (self.out / "model-99999.safetensors").write_bytes(b"stale") + result = run_converter(self.gguf, self.out, "--overwrite") + self.assertEqual(result.returncode, 0, result.stderr) + self.assertFalse((self.out / "model-99999.safetensors").exists()) + + def test_refuses_existing_container_without_flag(self): + self.assertEqual(run_converter(self.gguf, self.out).returncode, 0) + before = {p.name: p.read_bytes() for p in self.out.glob("model-*.safetensors")} + result = run_converter(self.gguf, self.out) + self.assertEqual(result.returncode, 1) + self.assertIn("already contains a converted container", result.stderr) + self.assertIn("--overwrite", result.stderr) + after = {p.name: p.read_bytes() for p in self.out.glob("model-*.safetensors")} + self.assertEqual(before, after) + + def test_dense_storage_dtypes(self): + self.assertEqual( + convert_gguf_to_olmoe._dense_to_storage(np.zeros(HIDDEN, np.float32)).dtype, + np.float32) + self.assertEqual( + convert_gguf_to_olmoe._dense_to_storage(np.zeros((HIDDEN, HIDDEN), np.float32)).dtype, + np.float16) + + +if __name__ == "__main__": + unittest.main() \ No newline at end of file diff --git a/c/tests/test_convert_routing.py b/c/tests/test_convert_routing.py index 825b88b56..2b255146c 100644 --- a/c/tests/test_convert_routing.py +++ b/c/tests/test_convert_routing.py @@ -105,6 +105,18 @@ def test_a_flash_checkpoint_reaches_its_own_converter(self): self.assertTrue(calls, "no converter was launched at all") self.assertEqual(self.script_of(calls[0]), "convert_glm53.py") + def test_olmoe_checkpoint_reaches_convert_olmoe_merged(self): + """OLMoE\'s converter is convert_olmoe_merged.py (not GLM-5.2\'s + convert_fp8_to_int4.py). The converter takes no precision flags; + the command must carry only --repo and --outdir.""" + calls = self.run_convert("olmoe") + self.assertTrue(calls, "no converter was launched at all") + self.assertEqual(self.script_of(calls[0]), "convert_olmoe_merged.py") + for flag in ("--ebits", "--io-bits", "--xbits", "--group-size"): + self.assertNotIn(flag, calls[0], + f"{flag} was passed to convert_olmoe_merged.py " + f"but that converter does not accept it") + def test_the_flash_command_carries_no_precision_flags(self): """convert_glm53.py has no --ebits/--io-bits/--xbits: it keeps the dense weights and the embedding wide and the engine picks the precision at @@ -231,10 +243,14 @@ def test_every_declared_converter_states_what_it_accepts(self): self.assertTrue( family.converter.endswith(".py"), f"{family.id}: converter must be a script under tools/") - self.assertTrue( - family.converter_accepts, - f"{family.id}: names {family.converter} but says nothing about " - f"which coli options it takes, so every one of them is dropped") + # An explicit empty tuple is a valid declaration when the converter + # genuinely takes none of the four precision flags (ebits / io_bits + # / xbits / group_size). OLMoE\'s convert_olmoe_merged.py is one + # such case: it has --flush-every and --min-free-gb, neither of + # which is a coli-convert precision flag. + self.assertIsInstance( + family.converter_accepts, tuple, + f"{family.id}: converter_accepts must be a tuple") def test_declared_converters_exist_on_disk(self): for family in FAMILIES: diff --git a/c/tests/test_cuda_binary_engine.py b/c/tests/test_cuda_binary_engine.py index 6130cf448..5aa84114d 100644 --- a/c/tests/test_cuda_binary_engine.py +++ b/c/tests/test_cuda_binary_engine.py @@ -1,12 +1,15 @@ """#1533: cuda_binary() must inspect the engine that will run, not always the -GLM binary. On Windows a CUDA_DLL build is recognised by a marker string in -the executable plus coli_cuda.dll next to it; a qwen36 build that carries -both must pass even when colibri.exe is CPU-only or absent.""" +GLM binary. On Windows a CUDA_DLL/HIP_DLL host is recognised by the backend +basename compiled into the loader, plus that file next to the engine. The +GLM/Qwen banner is not required: Kimi K3 CUDA_DLL builds link the same +loader without printing it, and a HIP host loads coli_hip.dll.""" import importlib.machinery import importlib.util +import json import os import sys import tempfile +import types import unittest from pathlib import Path from unittest import mock @@ -48,9 +51,66 @@ def test_windows_probe_reads_the_given_engine(self): with mock.patch.object(self.m.sys, "platform", "win32"): self.assertFalse(self.m.cuda_binary(str(qwen)), "no DLL next to the engine") + def test_windows_kimi_cuda_dll_without_glm_banner(self): + """Kimi K3 CUDA_DLL=1 links backend_loader.o and prints [K3-CUDA], not + the GLM routed-experts banner. --gpu used to refuse that install.""" + with tempfile.TemporaryDirectory() as d: + eng = Path(d) / "kimi_k3.exe" + eng.write_bytes(b"MZ [K3-CUDA] MXFP4 routed experts coli_cuda.dll") + (Path(d) / "coli_cuda.dll").write_bytes(b"MZ") + with mock.patch.object(self.m.sys, "platform", "win32"): + self.assertTrue(self.m.cuda_binary(str(eng))) + (Path(d) / "coli_cuda.dll").unlink() + with mock.patch.object(self.m.sys, "platform", "win32"): + self.assertFalse(self.m.cuda_binary(str(eng))) + + def test_windows_hip_host_accepts_coli_hip_dll(self): + """backend_loader.c under HIP_DLL=1 loads coli_hip.dll. Requiring + coli_cuda.dll refused a working HIP host that doctor already accepted.""" + with tempfile.TemporaryDirectory() as d: + eng = Path(d) / "colibri.exe" + eng.write_bytes(b"MZ [CUDA] mode: routed experts coli_hip.dll") + (Path(d) / "coli_hip.dll").write_bytes(b"MZ") + with mock.patch.object(self.m.sys, "platform", "win32"): + self.assertTrue(self.m.cuda_binary(str(eng))) + (Path(d) / "coli_cuda.dll").write_bytes(b"MZ") + (Path(d) / "coli_hip.dll").unlink() + with mock.patch.object(self.m.sys, "platform", "win32"): + self.assertFalse(self.m.cuda_binary(str(eng)), + "a stray CUDA DLL must not satisfy a HIP host") + def test_missing_engine_is_false(self): self.assertFalse(self.m.cuda_binary("/no/such/engine")) + def test_kimi_gpu_flag_sets_k3_cuda(self): + """--gpu writes COLI_CUDA=1. Kimi reads K3_CUDA, so the flag used to + pass the build check and still leave experts on the CPU.""" + with tempfile.TemporaryDirectory() as d: + model = Path(d) + (model / "config.json").write_text( + json.dumps({"model_type": "kimi_linear"}), encoding="utf-8") + a = types.SimpleNamespace( + model=str(model), gpu="0", vram=None, ram=0, ctx=None, + ngen=None, temp=None, cap=None, auto_tier=False) + with mock.patch.dict(os.environ, {}, clear=True), \ + mock.patch.object(self.m, "cuda_binary", return_value=True): + env = self.m.env_for_engine(a, "kimi") + self.assertEqual(env["COLI_CUDA"], "1") + self.assertEqual(env["K3_CUDA"], "1") + + def test_kimi_gpu_none_clears_k3_cuda(self): + with tempfile.TemporaryDirectory() as d: + model = Path(d) + (model / "config.json").write_text( + json.dumps({"model_type": "kimi_linear"}), encoding="utf-8") + a = types.SimpleNamespace( + model=str(model), gpu="none", vram=None, ram=0, ctx=None, + ngen=None, temp=None, cap=None, auto_tier=False) + with mock.patch.dict(os.environ, {"K3_CUDA": "1"}, clear=True): + env = self.m.env_for_engine(a, "kimi") + self.assertEqual(env["COLI_CUDA"], "0") + self.assertEqual(env["K3_CUDA"], "0") + if __name__ == "__main__": unittest.main() diff --git a/c/tests/test_cuda_fmt_guard.c b/c/tests/test_cuda_fmt_guard.c index 056d68535..85e5fee60 100644 --- a/c/tests/test_cuda_fmt_guard.c +++ b/c/tests/test_cuda_fmt_guard.c @@ -36,21 +36,27 @@ static int fails = 0; #define CHECK(c) do{ if(!(c)){ printf("FAIL %s:%d: %s\n", __FILE__, __LINE__, #c); fails++; } }while(0) int main(void) { - /* Decodable by weight_at's explicit branches. */ + /* Decodable by weight_at's explicit branches. fmt=8 (fp8-e4m3-b128) joined + * this set when the absorb path gained its block-scale decode (weight_at's + * fmt==8 branch + absorb_scale's per-128x128-block branch); its decode + * additionally needs the e4m3 LUT live on the device, which is an + * UPLOAD-time gate (coli_cuda_tensor_upload refuses fmt=8 until + * coli_cuda_fp8_set_lut has run), deliberately not part of this truth + * table -- see the predicate's own caveat note in backend_cuda.h. */ CHECK(coli_cuda_weight_at_supported(0)); /* f32 */ CHECK(coli_cuda_weight_at_supported(1)); /* int8-row */ CHECK(coli_cuda_weight_at_supported(2)); /* int4 nibbles */ CHECK(coli_cuda_weight_at_supported(3)); /* int2 */ CHECK(coli_cuda_weight_at_supported(4)); /* grouped int4 */ + CHECK(coli_cuda_weight_at_supported(8)); /* fp8-e4m3-b128 */ /* NOT decodable. Each of these has a real in-tree meaning, and each used to * be read as int2 by the fall-through. fmt=5 and fmt=6 are the ones that - * could already reach a CUDA tensor; fmt=8 is the one this branch is about; - * fmt=7 (MXFP4) has its own quant_matmul branch and never routes here. */ + * could already reach a CUDA tensor; fmt=7 (MXFP4) has its own quant_matmul + * branch and never routes here. */ CHECK(!coli_cuda_weight_at_supported(5)); /* int3-g64 */ CHECK(!coli_cuda_weight_at_supported(6)); /* E8/IQ3 */ CHECK(!coli_cuda_weight_at_supported(7)); /* MXFP4 */ - CHECK(!coli_cuda_weight_at_supported(8)); /* fp8-e4m3-b128 */ /* Out of range in both directions. The negative cases are the ones the * previous `fmt <= 4` host gate admitted: a descriptor whose fmt field is @@ -62,10 +68,10 @@ int main(void) { CHECK(!coli_cuda_weight_at_supported(100)); /* PRIVATE ORDINAL BLOCK base */ CHECK(!coli_cuda_weight_at_supported(1 << 30)); - /* The set is exactly {0,1,2,3,4} and nothing else in a wide sweep -- so a + /* The set is exactly {0,1,2,3,4,8} and nothing else in a wide sweep -- so a * later edit that widens the predicate has to change this line too. */ for (int fmt = -2048; fmt <= 2048; fmt++) { - int expect = (fmt >= 0 && fmt <= 4); + int expect = (fmt >= 0 && fmt <= 4) || fmt == 8; if (!!coli_cuda_weight_at_supported(fmt) != expect) { printf("FAIL fmt=%d: supported=%d, expected %d\n", fmt, coli_cuda_weight_at_supported(fmt), expect); diff --git a/c/tests/test_cuda_fmt_trap_cuda.cu b/c/tests/test_cuda_fmt_trap_cuda.cu index 07b2372f1..78e22691e 100644 --- a/c/tests/test_cuda_fmt_trap_cuda.cu +++ b/c/tests/test_cuda_fmt_trap_cuda.cu @@ -31,7 +31,7 @@ * WHAT MAKES THIS TEST BITE RATHER THAN MERELY PASS. Three controls, because an * exit code that says "the child failed" is worthless if the child fails no * matter what: - * 1. IN-PROCESS control: every supported fmt (0,1,2,3,4) is launched in the + * 1. IN-PROCESS control: every supported fmt (0,1,2,3,4,8) is launched in the * parent and must complete cleanly. If weight_at trapped on those, the * trap would be firing on valid input and this test would fail here. * 2. CHILD-HARNESS control: the probe list always contains SUPPORTED formats @@ -56,13 +56,24 @@ * landing in either order: whatever the predicate promises, the silicon must * deliver, and whatever it refuses, the silicon must refuse. * - * The fmt set probed covers both sides of today's truth table (0,3 supported; - * 5,6,7,8 unsupported container-carriable formats) plus the out-of-range values - * a corrupt descriptor could present (-1, 9, 1<<30) -- the ones the previous + * The fmt set probed covers both sides of today's truth table (0,3,8 supported + * -- 8 since the fp8-e4m3 absorb decode widened the predicate; 5,6,7 + * unsupported container-carriable formats) plus the out-of-range values a + * corrupt descriptor could present (-1, 9, 1<<30) -- the ones the previous * `fmt <= 4` host gate admitted. * + * fmt=8 gets one extra obligation on top of the derived DECODED verdict: its + * decoded VALUE is checked against an independent arithmetic e4m3 reference + * (a6_e4m3_ref below), because a fresh child process starts with a + * zero-initialized c_e4m3 table -- a decode that "returns a number" from a + * zero LUT is exactly the fabricated-numbers failure mode this file exists to + * refuse. The child publishes the LUT through the real + * coli_cuda_init/coli_cuda_fp8_set_lut path first, the same way any real + * fmt=8 caller must before upload. + * * make -C c cuda-test CUDA_ARCH=native (runs this among the CUDA tests) */ +#include #include #include #include @@ -95,6 +106,8 @@ #define A6_EXIT_TRAPPED 42 /* the launch aborted -- the refusal fired */ #define A6_EXIT_DECODED 43 /* the kernel returned a value -- no refusal */ #define A6_EXIT_HARNESS 44 /* could not run the probe at all */ +#define A6_EXIT_MISMATCH 45 /* decoded without trapping, but not to the * + * value the independent reference expects */ #define A6_PROBE_FLAG "--a6-probe" @@ -103,6 +116,38 @@ * visible number rather than something that could be mistaken for "no output". */ #define A6_FILL 0xA5 +/* Independent arithmetic e4m3 decoder (sign/exp/mantissa, OCP E4M3-FN: no + * infinities, only 0x7F/0xFF are NaN) -- same construction as t8_e4m3_ref in + * tests/test_backend_cuda.cu and e4m3_ref in tests/test_fp8_cuda.cu. Scope + * of the check, stated exactly: the device table is PUBLISHED FROM this + * reference (a6_publish_lut), so the comparison proves the table is nonzero + * and that weight_at's c_e4m3[base[i]] indexing is correct -- it cannot + * catch a wrong reference (a wrong table built from it would cancel), and + * the engine's own E4M3_LUT (quant.h) is not exercised by this harness at + * all. A6_FILL (0xA5) is not a NaN byte pattern. */ +static float a6_e4m3_ref(uint8_t b) { + int s = b >> 7, e = (b >> 3) & 15, m = b & 7; + if (e == 15 && m == 7) return NAN; /* E4M3-FN: only NaN, no inf */ + float v = e ? ldexpf(1.f + m/8.f, e-7) : ldexpf(m/8.f, -6); + return s ? -v : v; +} + +/* Publish the e4m3 LUT through the engine's own path, so weight_at's fmt=8 + * branch reads a real table instead of context-fresh zeros. Harmless for every + * other fmt (only the fmt==8 branch reads c_e4m3). coli_cuda_fp8_set_lut walks + * the engine's context table (g_nctx/g_ctx), which nothing else in this file + * populates -- the rest of the file talks to the device directly via the raw + * CUDA runtime API, deliberately, to stay independent of the engine's + * device-selection plumbing -- so coli_cuda_init(device 0) is called for this + * one dependency only. Returns 0 on harness failure. */ +static int a6_publish_lut(void) { + int dev0 = 0; + if (!coli_cuda_init(&dev0, 1)) return 0; + float lut[256]; + for (int i = 0; i < 256; i++) lut[i] = a6_e4m3_ref((uint8_t)i); + return coli_cuda_fp8_set_lut(lut); +} + /* The deliberate bad launch. weight_at is file-static device code; this is the * only caller in this TU, and it passes fmt straight through as a runtime * argument so nvcc cannot constant-fold the dispatch away. */ @@ -116,6 +161,16 @@ static int a6_probe_child(int fmt) { void *w = NULL; float *out = NULL; + /* Fresh execv'd process, fresh CUDA context: c_e4m3 starts zero-initialized + * here regardless of what the parent published. Published unconditionally + * rather than gated on fmt==8, so this child's setup mirrors a real + * caller's -- the engine publishes the LUT once at boot for every process + * that might touch fmt=8, not per tensor. */ + if (!a6_publish_lut()) { + printf(" [child fmt=%d] coli_cuda_init/coli_cuda_fp8_set_lut failed\n", fmt); + return A6_EXIT_HARNESS; + } + if (cudaMalloc(&w, 256) != cudaSuccess) { printf(" [child fmt=%d] cudaMalloc(weights) failed\n", fmt); return A6_EXIT_HARNESS; @@ -155,6 +210,24 @@ static int a6_probe_child(int fmt) { return A6_EXIT_TRAPPED; } printf(" [child fmt=%d] weight_at RETURNED %g\n", fmt, (double)v); + + /* fmt=8's extra obligation (see the file header): the decoded value must + * match the independent e4m3 reference. Without this, a zero-LUT decode + * (or any wrong LUT/indexing) would still count as "DECODED" -- a wrong + * number is not meaningfully different from a fabricated one, which is + * the exact failure mode this file's __trap() backstop exists to refuse. + * The other DECODED fmts in the probe list (0, 3) are already pinned + * bit-exact elsewhere (inprocess_supported_control here, the dense-matmul + * oracle in tests/test_backend_cuda.cu). */ + if (fmt == 8) { + float want = a6_e4m3_ref((uint8_t)A6_FILL); + if (v != want) { + printf(" [child fmt=%d] MISMATCH: weight_at(A6_FILL=0x%02X) = %g, " + "independent e4m3 reference says %g\n", fmt, A6_FILL, + (double)v, (double)want); + return A6_EXIT_MISMATCH; + } + } return A6_EXIT_DECODED; } @@ -210,7 +283,11 @@ static void expect_child(const char *self, int fmt, int want, const char *why) { printf("ok fmt=%-6d %s (child exit %d)\n", fmt, why, got); return; } - if (got == A6_EXIT_DECODED && want == A6_EXIT_TRAPPED) { + if (got == A6_EXIT_MISMATCH && want == A6_EXIT_DECODED) { + printf("FAIL fmt=%-6d %s -- decoded without trapping, but not to the " + "value the independent e4m3 reference expects (see child output " + "above)\n", fmt, why); + } else if (got == A6_EXIT_DECODED && want == A6_EXIT_TRAPPED) { printf("FAIL fmt=%-6d weight_at DECODED a format the host predicate " "refuses -- the device-side __trap() is not firing (dispatch " "disagrees with coli_cuda_weight_at_supported)\n", fmt); @@ -247,7 +324,14 @@ static void inprocess_supported_control(void) { return; } (void)cudaGetLastError(); - for (int fmt = 0; fmt <= 4; fmt++) { + /* The full predicate-true set, fmt=8 included -- its LUT is published by + * main() before this control runs (a6_publish_lut), so its in-process + * decode reads a real table, and its value is checked against the + * independent reference right here (the child probes re-check it in a + * fresh process, where the LUT must be re-published). */ + static const int supported[] = {0, 1, 2, 3, 4, 8}; + for (size_t fi = 0; fi < sizeof supported / sizeof supported[0]; fi++) { + int fmt = supported[fi]; if (!coli_cuda_weight_at_supported(fmt)) { /* keeps the two in step */ printf("FAIL fmt=%d is in this loop but the host predicate rejects it\n", fmt); fails++; @@ -267,6 +351,12 @@ static void inprocess_supported_control(void) { fails++; return; /* the context is poisoned; nothing after this is valid */ } + if (fmt == 8 && v != a6_e4m3_ref((uint8_t)A6_FILL)) { + printf("FAIL fmt=8 decoded in-process to %g, independent e4m3 " + "reference says %g (wrong LUT or wrong indexing)\n", + (double)v, (double)a6_e4m3_ref((uint8_t)A6_FILL)); + fails++; + } printf("ok fmt=%-6d supported: decoded in-process, weight_at = %g\n", fmt, (double)v); } @@ -288,6 +378,13 @@ int main(int argc, char **argv) { return 1; } + if (!a6_publish_lut()) { + printf("cuda fmt trap test: could not publish the e4m3 LUT " + "(coli_cuda_init/coli_cuda_fp8_set_lut failed) -- the fmt=8 " + "control cannot run\n"); + return 1; + } + printf("-- control: supported formats decode (in-process)\n"); inprocess_supported_control(); diff --git a/c/tests/test_cuda_init_failure.cu b/c/tests/test_cuda_init_failure.cu new file mode 100644 index 000000000..855fbb732 --- /dev/null +++ b/c/tests/test_cuda_init_failure.cu @@ -0,0 +1,74 @@ +/* Two logical devices mapped to one real GPU, with real stream lifetimes. + * This tests initialization rollback, not physical multi-GPU execution. */ +#include "../backend_gpu_compat.h" +#include +static int attempts, live_streams, fail_stream; +static int fail_select=-1, fail_props=-1; +static cudaError_t real_count(int *n) { return cudaGetDeviceCount(n); } +static cudaError_t test_count(int *n) { *n=2;return cudaSuccess; } +static cudaError_t test_select(int device) { + return device==fail_select ? cudaErrorMemoryAllocation : cudaSetDevice(0); +} +static cudaError_t test_props(cudaDeviceProp *p,int device) { + return device==fail_props ? cudaErrorMemoryAllocation : cudaGetDeviceProperties(p,0); +} +static cudaError_t test_create(cudaStream_t *s,unsigned flags) { + if(++attempts==fail_stream) return cudaErrorMemoryAllocation; + cudaError_t e=cudaStreamCreateWithFlags(s,flags);if(e==cudaSuccess)live_streams++;return e; +} +static cudaError_t test_destroy(cudaStream_t s) { + cudaError_t e=cudaStreamDestroy(s);if(e==cudaSuccess)live_streams--;return e; +} +#undef cudaGetDeviceCount +#undef cudaSetDevice +#undef cudaGetDeviceProperties +#undef cudaStreamCreateWithFlags +#undef cudaStreamDestroy +#define cudaGetDeviceCount test_count +#define cudaSetDevice test_select +#define cudaGetDeviceProperties test_props +#define cudaStreamCreateWithFlags test_create +#define cudaStreamDestroy test_destroy +#include "../backend_cuda.cu" +int main(void) { + int count=0;if(real_count(&count)!=cudaSuccess||!count){puts("SKIP: no GPU");return 0;} + int ids[]={0,1}, duplicate[]={0,0},invalid[]={0,2}; + for(int round=0;round<2;round++){ + attempts=0; + if(coli_cuda_init(duplicate,2)||attempts||live_streams||coli_cuda_device_count()) return 1; + if(coli_cuda_init(invalid,2)||attempts||live_streams||coli_cuda_device_count()) return 2; + for(fail_stream=1;fail_stream<=2;fail_stream++){ + attempts=0; + if(coli_cuda_init(ids,2)||live_streams||coli_cuda_device_count()) return 3; + } + fail_stream=0; + for(int device=0;device<2;device++){ + g_current_device=-1; /* force the runtime selection fault past the cache */ + fail_select=device; + if(coli_cuda_init(ids,2)||live_streams||coli_cuda_device_count()) return 11; + fail_select=-1; + fail_props=device; + if(coli_cuda_init(ids,2)||live_streams||coli_cuda_device_count()) return 12; + fail_props=-1; + } + attempts=0; + if(!coli_cuda_init(ids,2)||live_streams!=2||coli_cuda_device_count()!=2) return 4; + /* Rejected input must leave an existing backend reachable. */ + if(coli_cuda_init(invalid,2)||live_streams!=2||coli_cuda_device_count()!=2) return 5; + if(coli_cuda_init(duplicate,2)||live_streams!=2||coli_cuda_device_count()!=2) return 6; + if(cudaMalloc(&g_ctx[0].x,256)!=cudaSuccess) return 10; + float *scratch=g_ctx[0].x; + cudaStream_t original=g_ctx[0].stream; + int before=attempts; + if(!coli_cuda_init(ids,2)||attempts!=before||live_streams!=2||g_ctx[0].stream!=original||g_ctx[0].x!=scratch){ + puts("FAIL: repeated init replaced live contexts");return 8; + } + int reversed[]={1,0}; + if(coli_cuda_init(ids,1)||coli_cuda_init(reversed,2)||attempts!=before|| + live_streams!=2||g_ctx[0].stream!=original||g_ctx[0].x!=scratch||coli_cuda_device_count()!=2){ + puts("FAIL: device-list change modified active contexts");return 9; + } + coli_cuda_shutdown();if(live_streams||coli_cuda_device_count()) return 7; + } + puts("CUDA initialization rollback: PASS");return 0; +} diff --git a/c/tests/test_cuda_lut_gate.c b/c/tests/test_cuda_lut_gate.c new file mode 100644 index 000000000..ab6a1d5c3 --- /dev/null +++ b/c/tests/test_cuda_lut_gate.c @@ -0,0 +1,142 @@ +/* fmt=8 LUT-gate state machine, pinned on the host with no CUDA toolchain. + * + * Why this exists: the gate that stops a fmt=8 tensor reaching a kernel whose + * device has an unwritten e4m3 table is implemented in backend_cuda.cu, which + * compiles only under nvcc/hipcc. Every test that exercised it therefore ran + * nowhere in CPU CI, and a rebase was able to change the surrounding + * init semantics while three sites in the tree went on asserting the old ones. + * The two decisions the gate rests on are pure predicates in backend_cuda.h; + * backend_cuda.cu calls them at the real sites, so what is pinned below is the + * engine's own logic, not a restatement of it. + * + * What this does NOT pin: that backend_cuda.cu calls the predicates in the + * right order, or at all. That needs a CUDA toolchain (tests/test_fp8_cuda.cu + * Phase 3 covers it there). What it does pin is the decision table itself and + * the one invariant the whole gate rests on: + * + * no live context set => the LUT flag is clear + * + * which holds because shutdown is the only writer that clears the flag and it + * clears g_nctx in the same breath, and because init refuses to rebuild + * contexts underneath a live set. + */ +#include +#include "../backend_cuda.h" + +static int fails; + +static void ck(int cond, const char *what) { + if (!cond) { printf("FAIL %s\n", what); fails++; } +} + +/* A model of the three public transitions, written in terms of the SAME + * predicates the engine calls. `nctx`/`ready` are the engine's two globals. */ +typedef struct { int nctx; int ready; int dev[8]; } Gate; + +static int g_init(Gate *g, const int *want, int count) { + int d = coli_cuda_init_disposition(g->nctx, count, want, g->dev); + if (d == COLI_CUDA_INIT_REFUSE) return 0; /* touches nothing */ + if (d == COLI_CUDA_INIT_ACCEPT) return 1; /* touches nothing */ + for (int i = 0; i < count; i++) g->dev[i] = want[i]; + g->nctx = count; /* flag NOT written here */ + return 1; +} +static void g_set_lut(Gate *g) { if (g->nctx > 0) g->ready = 1; } +static void g_shutdown(Gate *g) { g->nctx = 0; g->ready = 0; } +static int g_upload(Gate *g, int fmt) { + return g->nctx > 0 && coli_cuda_fp8_gate_admits(fmt, g->ready); +} + +int main(void) { + /* --- the disposition table, directly ------------------------------- */ + { + int live1[1] = {0}, want1[1] = {0}, want2[2] = {0, 1}, wantB[1] = {1}; + ck(coli_cuda_init_disposition(0, 1, want1, live1) == COLI_CUDA_INIT_BUILD, + "nothing live -> BUILD"); + ck(coli_cuda_init_disposition(1, 1, want1, live1) == COLI_CUDA_INIT_ACCEPT, + "same set -> ACCEPT"); + ck(coli_cuda_init_disposition(1, 2, want2, live1) == COLI_CUDA_INIT_REFUSE, + "widening set -> REFUSE"); + ck(coli_cuda_init_disposition(1, 1, wantB, live1) == COLI_CUDA_INIT_REFUSE, + "same size, different device -> REFUSE"); + int live2[2] = {0, 1}; + ck(coli_cuda_init_disposition(2, 1, want1, live2) == COLI_CUDA_INIT_REFUSE, + "narrowing set -> REFUSE"); + ck(coli_cuda_init_disposition(2, 2, want2, live2) == COLI_CUDA_INIT_ACCEPT, + "same two-device set -> ACCEPT"); + } + + /* --- the upload gate ------------------------------------------------ */ + for (int fmt = -3; fmt <= 9; fmt++) { + if (fmt == 8) continue; + ck(coli_cuda_fp8_gate_admits(fmt, 0) && coli_cuda_fp8_gate_admits(fmt, 1), + "non-fmt=8 is admitted regardless of the LUT flag"); + } + ck(!coli_cuda_fp8_gate_admits(8, 0), "fmt=8 refused while the LUT is unpublished"); + ck(coli_cuda_fp8_gate_admits(8, 1), "fmt=8 admitted once the LUT is published"); + + /* --- the three lifecycle edges, as sequences ------------------------ */ + { + Gate g = {0, 0, {0}}; + int d0[1] = {0}, d01[2] = {0, 1}; + + ck(g_init(&g, d0, 1) == 1, "first init succeeds"); + ck(!g_upload(&g, 8), "fmt=8 refused before the first publish"); + ck(g_upload(&g, 1), "fmt=1 unaffected by the gate"); + g_set_lut(&g); + ck(g_upload(&g, 8), "fmt=8 admitted after publish"); + + /* Edge 2: same-set re-init keeps the flag, because it keeps the + * contexts the table was published to. */ + ck(g_init(&g, d0, 1) == 1, "same-set re-init returns success"); + ck(g.nctx == 1, "same-set re-init leaves the context count alone"); + ck(g_upload(&g, 8), "fmt=8 still admitted after a same-set re-init"); + + /* Edge 3: different-set re-init is refused and changes nothing. */ + ck(g_init(&g, d01, 2) == 0, "widening re-init is refused"); + ck(g.nctx == 1, "refused re-init leaves the context set alone"); + ck(g_upload(&g, 8), "refused re-init leaves the LUT flag alone"); + + /* Edge 1: shutdown clears both, and the next span must republish. */ + g_shutdown(&g); + ck(g.nctx == 0 && g.ready == 0, "shutdown clears the set and the flag"); + ck(g_init(&g, d01, 2) == 1, "after shutdown a WIDER set may be built"); + ck(!g_upload(&g, 8), "fmt=8 refused on the widened set until it republishes"); + g_set_lut(&g); + ck(g_upload(&g, 8), "fmt=8 admitted on the widened set after republish"); + } + + /* --- the invariant, over every reachable transition sequence -------- */ + { + /* Exhaustive over sequences of length 6 drawn from + * {init{0}, init{0,1}, set_lut, shutdown}: no reachable state may have + * a live-free context set and a set flag, and no state may admit fmt=8 + * without a live set. */ + int d0[1] = {0}, d01[2] = {0, 1}; + int idx[6] = {0, 0, 0, 0, 0, 0}; + long checked = 0; + for (;;) { + Gate g = {0, 0, {0}}; + for (int s = 0; s < 6; s++) { + switch (idx[s]) { + case 0: g_init(&g, d0, 1); break; + case 1: g_init(&g, d01, 2); break; + case 2: g_set_lut(&g); break; + default: g_shutdown(&g); break; + } + if (g.nctx == 0 && g.ready != 0) { printf("FAIL invariant: no contexts but the LUT flag is set\n"); return 1; } + if (g.nctx == 0 && g_upload(&g, 8)) { printf("FAIL invariant: fmt=8 admitted with no live context\n"); return 1; } + } + checked++; + int p = 5; + while (p >= 0 && ++idx[p] > 3) { idx[p] = 0; p--; } + if (p < 0) break; + } + if (checked != 4096) { printf("FAIL sequence enumeration covered %ld, expected 4096\n", checked); return 1; } + printf("lut-gate invariant holds over %ld transition sequences\n", checked); + } + + if (fails) { printf("cuda lut-gate tests: %d FAILED\n", fails); return 1; } + printf("cuda lut-gate tests: ok\n"); + return 0; +} diff --git a/c/tests/test_deepseek_v4.c b/c/tests/test_deepseek_v4.c index 55224b82f..7dd6b829f 100644 --- a/c/tests/test_deepseek_v4.c +++ b/c/tests/test_deepseek_v4.c @@ -1063,6 +1063,163 @@ static int test_expert_store_miss_scaling(void) { "(44/104/208 slots=1 probe each)"); return 0; } + +/* REAP-style layout: per-matrix scale/weight pairs so contiguous_group() + * fails and build_record() sets per_matrix=1. Payload is padded past 4 KiB + * so an O_DIRECT window has a real aligned pread. */ +static int write_per_matrix_fixture(const char *path, int expert_count) { + static const char *matrix_names[3] = {"w1", "w2", "w3"}; + if (expert_count < 1) return -1; + size_t header_capacity = 256 + (size_t)expert_count * 1024; + char *header = malloc(header_capacity); + size_t payload_size = 8192; + unsigned char *payload = malloc(payload_size); + if (!header || !payload) { + free(payload); + free(header); + return -1; + } + size_t used = (size_t)snprintf(header, header_capacity, "{"); + size_t cursor = 0; + for (int expert = 0; expert < expert_count; expert++) { + for (int matrix = 0; matrix < 3; matrix++) { + size_t scale = cursor; + size_t weight = cursor + 1; + int count = snprintf( + header + used, header_capacity - used, + "%s\"layers.0.ffn.experts.%d.%s.scale\":{" + "\"dtype\":\"F8_E8M0\",\"shape\":[1,1]," + "\"data_offsets\":[%zu,%zu]}", + used > 1 ? "," : "", expert, matrix_names[matrix], + scale, scale + 1); + if (count < 0 || (size_t)count >= header_capacity - used) { + free(payload); free(header); return -1; + } + used += (size_t)count; + count = snprintf( + header + used, header_capacity - used, + ",\"layers.0.ffn.experts.%d.%s.weight\":{" + "\"dtype\":\"I8\",\"shape\":[1,16]," + "\"data_offsets\":[%zu,%zu]}", + expert, matrix_names[matrix], weight, weight + 16); + if (count < 0 || (size_t)count >= header_capacity - used) { + free(payload); free(header); return -1; + } + used += (size_t)count; + cursor += 17; + } + } + if (used + 1 >= header_capacity) { + free(payload); free(header); return -1; + } + header[used++] = '}'; + header[used] = '\0'; + for (size_t i = 0; i < payload_size; i++) + payload[i] = (unsigned char)i; + uint64_t header_length = used; + int fd = open(path, O_CREAT | O_TRUNC | O_WRONLY | COMPAT_O_BINARY, 0600); + if (fd < 0) { + free(payload); free(header); return -1; + } + int result = write_all(fd, &header_length, sizeof(header_length)) || + write_all(fd, header, (size_t)header_length) || + write_all(fd, payload, payload_size); + close(fd); + free(payload); + free(header); + return result ? -1 : 0; +} + +static int expect_per_matrix_expert(const ColiExpertView *view, int expert) { + int base = expert * 51; + return view->gate.format == COLI_TENSOR_FP4_NATIVE_BLOCK && + view->gate.rows == 1 && view->gate.columns == 32 && + view->gate.data_bytes == 16 && view->gate.scale_bytes == 1 && + view->down.data_bytes == 16 && view->up.data_bytes == 16 && + ((const unsigned char *)view->gate.scales)[0] == + (unsigned char)base && + ((const unsigned char *)view->down.scales)[0] == + (unsigned char)(base + 17) && + ((const unsigned char *)view->up.scales)[0] == + (unsigned char)(base + 34) && + ((const unsigned char *)view->gate.data)[0] == + (unsigned char)(base + 1) && + ((const unsigned char *)view->down.data)[0] == + (unsigned char)(base + 18) && + ((const unsigned char *)view->up.data)[0] == + (unsigned char)(base + 35); +} + +static int test_expert_store_per_matrix_direct(void) { + /* Would fail this test: per_matrix always increments v4_direct_fallbacks + * and never calls v4_read_direct_window even when a direct twin exists. */ + char directory[] = "colibri-v4-reap-XXXXXX"; + char path[256], error[256]; + setenv("COLI_V4_AUTOPIN", "0", 1); + setenv("COLI_V4_SAVE_USAGE", "0", 1); + setenv("COLI_V4_ROWS16", "0", 1); + setenv("COLI_V4_PREWARM", "0", 1); + unsetenv("COLI_V4_DIRECT"); + if (!mkdtemp(directory)) { perror("mkdtemp per_matrix"); return 1; } + snprintf(path, sizeof(path), "%s/model.safetensors", directory); + if (write_per_matrix_fixture(path, 2) != 0) { + perror("write_per_matrix_fixture"); + rmdir(directory); + return 1; + } + + ColiDeepSeekV4ExpertStoreOptions options = { + directory, 1, 2, 102, -1, 0, 0 + }; + ColiExpertStore *store = NULL; + if (coli_deepseek_v4_expert_store_open(&options, &store, + error, sizeof(error)) != 0) { + fprintf(stderr, "%s\n", error); + unlink(path); rmdir(directory); + return 1; + } + if (coli_v4_test_force_streaming_direct(store) != 0) { + fprintf(stderr, "per_matrix direct: no streaming-direct fd\n"); + store->ops->destroy(store); + unlink(path); rmdir(directory); + return 1; + } + coli_v4_test_reset_direct_io_stats(); + ColiExpertView view; + if (coli_expert_lookup(store, (ColiExpertKey){0, 0}, &view) != 0) { + fprintf(stderr, "per_matrix lookup failed\n"); + store->ops->destroy(store); + unlink(path); rmdir(directory); + return 1; + } + if (!expect_per_matrix_expert(&view, 0)) { + fprintf(stderr, + "per_matrix slab mismatch: scales=%u/%u/%u weights=%u/%u/%u\n", + ((const unsigned char *)view.gate.scales)[0], + ((const unsigned char *)view.down.scales)[0], + ((const unsigned char *)view.up.scales)[0], + ((const unsigned char *)view.gate.data)[0], + ((const unsigned char *)view.down.data)[0], + ((const unsigned char *)view.up.data)[0]); + coli_expert_release(store, &view); + store->ops->destroy(store); + unlink(path); rmdir(directory); + return 1; + } + coli_expert_release(store, &view); + uint64_t reads = coli_v4_test_direct_reads(); + uint64_t fallbacks = coli_v4_test_direct_fallbacks(); + store->ops->destroy(store); + if (scratch_remove(directory, path)) return 1; + if (reads < 1 || fallbacks != 0) { + fprintf(stderr, + "per_matrix direct path unused: reads=%llu fallbacks=%llu\n", + (unsigned long long)reads, (unsigned long long)fallbacks); + return 1; + } + puts("DeepSeek-V4 ExpertStore per_matrix direct: ok"); + return 0; +} /* ==== end test_deepseek_v4_expert_store.c ==== */ /* ==== begin test_deepseek_v4_kv_cache.c ==== */ @@ -1531,6 +1688,10 @@ int main(int argc, char **argv) { fprintf(stderr, "FAIL: test_expert_store_miss_scaling\n"); return 1; } + if (test_expert_store_per_matrix_direct() != 0) { + fprintf(stderr, "FAIL: test_expert_store_per_matrix_direct\n"); + return 1; + } if (test_kv_cache() != 0) { fprintf(stderr, "FAIL: test_kv_cache\n"); return 1; diff --git a/c/tests/test_deepseek_v4_brio.py b/c/tests/test_deepseek_v4_brio.py new file mode 100644 index 000000000..3fb771042 --- /dev/null +++ b/c/tests/test_deepseek_v4_brio.py @@ -0,0 +1,289 @@ +#!/usr/bin/env python3 +"""The numeric channel on the DeepSeek V4 serve path (SUBMIT logprobs=k / pin=1). + +This is what `POST /v1/brio` speaks (docs/brio.md), and #1648 found that this +engine accepted the header and then refused the request. The properties under +test are the exact ones a closed-set caller relies on: + +- a read-only request (max_tokens=0, logprobs>0) answers with one ECHO frame + per prompt position after the first, and generates nothing; +- the ECHO at position p is the head over prompt[:p]: its best token is the + token a greedy generation from that same prefix produces; +- a pinned prompt gives every prompt that extends it the predictor of its + first fresh token, from the kept scores, and those scores are the ones a + cold prefill computes; +- with the channel open, generation is unchanged and every DATA frame carries + the tail of the token it holds. + +Speaks SUBMIT/ECHO/DATA/DONE directly, as tests/test_deepseek_v4_prefix.py does. +""" +from __future__ import annotations + +import argparse +import json +import math +import os +import re +import subprocess +import sys +from pathlib import Path + + +def token_prompt(ids: list[int]) -> str: + return "".join(f"" for token in ids) + + +def parse_tokens(text: str) -> list[int]: + return [int(match) for match in re.findall(r"", text)] + + +class Reply: + def __init__(self) -> None: + self.accepted = -1 + self.echoes: dict[int, dict] = {} # position -> {"token", "lp", "top": {tid: tlp}} + self.data: list[tuple[bytes, dict | None]] = [] + self.reuse = -1 + self.completion = -1 + + @property + def text(self) -> str: + return b"".join(piece for piece, _ in self.data).decode("utf-8", "replace") + + +def parse_tail(fields: list[str]) -> dict | None: + """` [tid tlp]*k` -> {"lp": float, "top": {tid: tlp}}; None for no tail.""" + if not fields: + return None + lp = None if fields[0] == "nan" else float(fields[0]) + k = int(fields[1]) + pairs = fields[2:2 + 2 * k] + top = {int(pairs[i]): float(pairs[i + 1]) for i in range(0, len(pairs), 2)} + return {"lp": lp, "top": top} + + +class Serve: + """One persistent `SERVE=1` engine process.""" + + def __init__(self, binary: Path, model: Path, ctx: str = "128") -> None: + env = dict(os.environ, SERVE="1", SNAP=str(model), CTX=ctx) + self.process = subprocess.Popen( + [str(binary)], stdin=subprocess.PIPE, stdout=subprocess.PIPE, + stderr=subprocess.PIPE, env=env, + ) + self.counter = 0 + + def submit(self, prompt: str, max_tokens: int, logprobs: int = 0, + pin: bool = False) -> Reply: + self.counter += 1 + request_id = f"r{self.counter}" + payload = prompt.encode("utf-8") + header = f"SUBMIT {request_id} 0 {len(payload)} {max_tokens} 0.0 1.0 0" + if logprobs: + header += f" logprobs={logprobs}" + if pin: + header += " pin=1" + assert self.process.stdin is not None + self.process.stdin.write(header.encode("ascii") + b"\n" + payload + b"\n") + self.process.stdin.flush() + + reply = Reply() + stream = self.process.stdout + assert stream is not None + while True: + line = stream.readline() + if not line: + raise AssertionError( + f"engine closed stdout during {request_id}; " + f"stderr:\n{self._drain_stderr()}") + fields = line.decode("utf-8", "replace").split() + if not fields: + continue + kind = fields[0] + if kind == "ERROR": + raise AssertionError(f"engine ERROR: {line!r}") + if kind == "ACCEPT": + reply.accepted = int(fields[2]) + elif kind == "DATA": + size = int(fields[2]) + piece = stream.read(size) + stream.read(1) + reply.data.append((piece, parse_tail(fields[3:]))) + elif kind == "ECHO": + # ECHO [tid tlp]*k, then n bytes + LF + size = int(fields[2]) + piece = stream.read(size) + stream.read(1) + tail = parse_tail(fields[4:]) + assert tail is not None + reply.echoes[int(fields[3])] = {"token": piece.decode("utf-8", "replace"), **tail} + elif kind == "DONE": + # DONE STAT [] + reply.completion = int(fields[3]) + if len(fields) >= 10: + reply.reuse = int(fields[9]) + break + return reply + + def _drain_stderr(self) -> str: + assert self.process.stderr is not None + try: + return self.process.stderr.read(8192).decode("utf-8", "replace") + except Exception: # pragma: no cover + return "" + + def close(self) -> None: + try: + if self.process.stdin: + self.process.stdin.close() + self.process.wait(timeout=30) + except Exception: # pragma: no cover + self.process.kill() + + +def best_of(echo: dict) -> int: + return max(echo["top"], key=lambda tid: echo["top"][tid]) + + +def check_read_only(binary: Path, model: Path, ids: list[int]) -> None: + """max_tokens=0 with logprobs: one ECHO per position after the first, no DATA.""" + serve = Serve(binary, model) + try: + reply = serve.submit(token_prompt(ids), 0, logprobs=3) + finally: + serve.close() + n = len(ids) + if reply.accepted != n: + raise AssertionError(f"ACCEPT said {reply.accepted} tokens for a {n}-token prompt") + if reply.data: + raise AssertionError(f"a read-only request generated {len(reply.data)} DATA frames") + if reply.completion != 0: + raise AssertionError(f"DONE reported {reply.completion} completion tokens") + if sorted(reply.echoes) != list(range(1, n)): + raise AssertionError(f"ECHO positions {sorted(reply.echoes)}, expected 1..{n - 1}") + for position, echo in reply.echoes.items(): + if parse_tokens(echo["token"]) != [ids[position]]: + raise AssertionError(f"ECHO at {position} names {echo['token']!r}, prompt has {ids[position]}") + if echo["lp"] is None or echo["lp"] > 1e-6: + raise AssertionError(f"ECHO at {position}: log-probability {echo['lp']} is not <= 0") + if len(echo["top"]) != 3: + raise AssertionError(f"ECHO at {position}: asked top-3, got {len(echo['top'])}") + mass = sum(math.exp(v) for v in echo["top"].values()) + if mass > 1.0 + 1e-4: + raise AssertionError(f"ECHO at {position}: top-3 mass {mass} exceeds 1") + if ids[position] in echo["top"] and abs(echo["top"][ids[position]] - echo["lp"]) > 1e-5: + raise AssertionError(f"ECHO at {position}: the token's own lp differs from its top-k entry") + print(f"PASS read-only: {n - 1} ECHO frames, zero generated tokens") + + +def check_echo_is_the_head(binary: Path, model: Path, ids: list[int]) -> None: + """The best token of the ECHO at p is what greedy decoding emits after prompt[:p].""" + serve = Serve(binary, model) + try: + full = serve.submit(token_prompt(ids), 0, logprobs=4) + for position in sorted({1, len(ids) // 2, len(ids) - 1}): + cold = Serve(binary, model) + try: + gen = cold.submit(token_prompt(ids[:position]), 1) + finally: + cold.close() + produced = parse_tokens(gen.text) + expected = best_of(full.echoes[position]) + if produced != [expected]: + raise AssertionError( + f"position {position}: ECHO's best token is {expected}, " + f"greedy generation from the same prefix gave {produced}") + finally: + serve.close() + print("PASS echo is the head: best ECHO token == greedy token from the same prefix") + + +def check_pin(binary: Path, model: Path, ids: list[int], a: int, b: int) -> None: + """Pin the prompt, ask two options: both start from the pin with the first token's predictor.""" + n = len(ids) + serve = Serve(binary, model) + try: + pinned = serve.submit(token_prompt(ids), 0, logprobs=2, pin=True) + if pinned.completion != 0 or pinned.data: + raise AssertionError("the pinned read generated tokens") + first = serve.submit(token_prompt(ids + [a]), 0, logprobs=2) + second = serve.submit(token_prompt(ids + [b]), 0, logprobs=2) + finally: + serve.close() + for name, reply in (("first option", first), ("second option", second)): + if reply.reuse != n: + raise AssertionError(f"{name}: reused {reply.reuse} tokens, the pin holds {n}") + if sorted(reply.echoes) != [n]: + raise AssertionError(f"{name}: ECHO positions {sorted(reply.echoes)}, expected [{n}]") + if first.echoes[n]["top"] != second.echoes[n]["top"]: + raise AssertionError("the two options saw different predictors for the same pinned prompt") + if parse_tokens(first.echoes[n]["token"]) != [a] or parse_tokens(second.echoes[n]["token"]) != [b]: + raise AssertionError("the ECHO at the option position does not name the option token") + # the kept scores are what a cold prefill computes at that position + cold = Serve(binary, model) + try: + fresh = cold.submit(token_prompt(ids + [a]), 0, logprobs=2) + finally: + cold.close() + if fresh.reuse != 0: + raise AssertionError(f"cold engine reported reuse={fresh.reuse}") + warm_top, cold_top = first.echoes[n]["top"], fresh.echoes[n]["top"] + if warm_top.keys() != cold_top.keys() or any(abs(warm_top[t] - cold_top[t]) > 1e-4 for t in warm_top): + raise AssertionError(f"pinned predictor {warm_top} differs from a cold prefill's {cold_top}") + if abs(first.echoes[n]["lp"] - fresh.echoes[n]["lp"]) > 1e-4: + raise AssertionError("the option's log-probability differs between the pin and a cold prefill") + print(f"PASS pin: both options reused {n} tokens and read the pinned predictor, equal to a cold prefill") + + +def check_generation_with_tail(binary: Path, model: Path, ids: list[int], max_new: int) -> None: + """The channel does not change greedy output, and every DATA carries its token's tail.""" + plain = Serve(binary, model) + try: + without = plain.submit(token_prompt(ids), max_new) + finally: + plain.close() + channel = Serve(binary, model) + try: + with_lp = channel.submit(token_prompt(ids), max_new, logprobs=2) + finally: + channel.close() + if without.text != with_lp.text: + raise AssertionError(f"logprobs changed the output:\n plain: {without.text!r}\n channel: {with_lp.text!r}") + if not with_lp.data: + raise AssertionError("no DATA frames with the channel open") + for piece, tail in with_lp.data: + if tail is None: + raise AssertionError(f"DATA {piece!r} carries no logprob tail") + token = parse_tokens(piece.decode("utf-8", "replace")) + if len(token) != 1: + raise AssertionError(f"DATA frame is not one token: {piece!r}") + if len(tail["top"]) != 2: + raise AssertionError(f"asked top-2, got {len(tail['top'])}") + # greedy: the emitted token IS the best one, so its lp is the top entry + if best_of(tail) != token[0] or abs(tail["top"][token[0]] - tail["lp"]) > 1e-5: + raise AssertionError(f"DATA {token}: tail {tail} does not name it as the best token") + print(f"PASS generation: {len(with_lp.data)} tokens unchanged, each DATA frame with its tail") + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n", 1)[0]) + parser.add_argument("--binary", type=Path, required=True) + parser.add_argument("--fixture", type=Path, required=True) + arguments = parser.parse_args() + binary = arguments.binary.resolve() + reference = json.loads((arguments.fixture / "ref.json").read_text("utf-8")) + case = reference["cases"]["short"] + ids = list(case["prompt_ids"]) + greedy = list(case["greedy_new_ids"]) + # two options: the token greedy decoding would pick, and one it would not + a = greedy[0] + b = next(t for t in ids if t != a) + check_read_only(binary, arguments.fixture, ids) + check_echo_is_the_head(binary, arguments.fixture, ids) + check_pin(binary, arguments.fixture, ids, a, b) + check_generation_with_tail(binary, arguments.fixture, ids, 4) + print("PASS DeepSeek V4 numeric channel: all checks completed") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/c/tests/test_deepseek_v4_build_flags.py b/c/tests/test_deepseek_v4_build_flags.py new file mode 100644 index 000000000..850dc20dd --- /dev/null +++ b/c/tests/test_deepseek_v4_build_flags.py @@ -0,0 +1,114 @@ +"""Makefile.deepseek-v4 must rebuild its objects when the build flags change. + +The unit objects are named after the unit (COLI_V4_UNIT_MATH.o), not after the +flags they were compiled with, and three builds share those names in c/: the +engine (`CUDA=1` adds -DCOLI_V4_GPU_TIER), the parent's `make test-c` (no GPU +tier) and `make test-asan` (EXTRA_CFLAGS). Timestamps cannot tell them apart, +so without a record of the flags make reports "is up to date" and links the +other build's objects (#1702): + + make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 + make test-c + -> undefined reference to `coli_v4_gpu_matvec_grouped' + +and, the silent direction, `make test-c` followed by the CUDA=1 engine build +links CPU-only units into an engine that was asked for the GPU tier. + +The compiler here is a stand-in that only creates its -o target, so the test +needs make and sh but no toolchain, and checks exactly what make decides to +rebuild. The sources are copied, not linked, so their timestamps can be pinned +without touching the tree. +""" +import os +import shutil +import subprocess +import tempfile +import time +import unittest +from pathlib import Path + +C_DIR = Path(__file__).resolve().parent.parent +MAKE = shutil.which("make") +UNIT = "COLI_V4_UNIT_MATH.o" + +FAKE_CC = """#!/bin/sh +# Stand-in compiler: creates the -o target and compiles nothing. +while [ "$#" -gt 0 ]; do + if [ "$1" = -o ]; then : > "$2"; exit 0; fi + shift +done +exit 1 +""" + + +@unittest.skipUnless(MAKE and os.name == "posix", "make and a POSIX shell are required") +class DeepseekV4BuildFlagsTest(unittest.TestCase): + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.dir = Path(self._tmp.name) + for pattern in ("Makefile.deepseek-v4*", "deepseek_v4.c", "*.h", "*.inc"): + for src in C_DIR.glob(pattern): + shutil.copy(src, self.dir / src.name) + fake = self.dir / "fakecc.sh" + fake.write_text(FAKE_CC, encoding="utf-8") + self.cc = f"sh {fake}" + # Sources well in the past: every object built below is newer than all + # of them, so the only thing left that can make it stale is the flags. + past = time.time() - 1000 + for f in self.dir.iterdir(): + os.utime(f, (past, past)) + probe = self.make("print-v4-objs") + if probe.returncode != 0: + self.skipTest("Makefile.deepseek-v4 does not support this host: " + + probe.stderr.strip()[-200:]) + + def tearDown(self): + self._tmp.cleanup() + + def make(self, *args): + return subprocess.run( + [MAKE, "-f", "Makefile.deepseek-v4", f"CC={self.cc}", *args], + cwd=self.dir, text=True, capture_output=True, check=False, timeout=120) + + def build(self, *variables): + """Build the unit, then age everything it produced by the same amount. + + Leaves every file this build wrote at one timestamp, older than now + and newer than the sources: a later build only recompiles if make + finds a real reason to, never because two writes landed in the same + clock tick. + """ + before = {f: f.stat().st_mtime for f in self.dir.iterdir()} + result = self.make(UNIT, *variables) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + aged = time.time() - 500 + for f in self.dir.iterdir(): + if before.get(f) != f.stat().st_mtime: + os.utime(f, (aged, aged)) + return "-c deepseek_v4.c" in result.stdout + + def test_same_flags_do_not_rebuild(self): + """The guard: a fix that always recompiles would pass the tests below.""" + self.assertTrue(self.build("EXTRA_CFLAGS=-DCOLI_TEST_FLAGS_A")) + self.assertFalse(self.build("EXTRA_CFLAGS=-DCOLI_TEST_FLAGS_A"), + f"{UNIT} was recompiled with unchanged flags") + + def test_changed_flags_rebuild(self): + """EXTRA_CFLAGS is how the parent's test-asan reaches this Makefile.""" + self.assertTrue(self.build("EXTRA_CFLAGS=-DCOLI_TEST_FLAGS_A")) + self.assertTrue(self.build("EXTRA_CFLAGS=-DCOLI_TEST_FLAGS_B"), + f"{UNIT} was reused after EXTRA_CFLAGS changed: the " + "objects of one build configuration leak into another") + + def test_cuda_tier_objects_are_not_reused_by_a_cpu_build(self): + """The reported case, in both directions (#1702). Building one unit + object needs no nvcc: only backend_cuda_dsv4.o does.""" + self.assertTrue(self.build("CUDA=1")) + self.assertTrue(self.build(), + f"{UNIT} built with CUDA=1 was reused by a CPU build") + self.assertTrue(self.build("CUDA=1"), + f"{UNIT} built without the GPU tier was reused by CUDA=1") + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_deepseek_v4_cuda_arch_flags.py b/c/tests/test_deepseek_v4_cuda_arch_flags.py new file mode 100644 index 000000000..325951239 --- /dev/null +++ b/c/tests/test_deepseek_v4_cuda_arch_flags.py @@ -0,0 +1,154 @@ +"""backend_cuda_dsv4.o must be rebuilt when the nvcc command line changes. + +The object is named after its source, but its contents come from +$(NVCC) $(V4_NVCCFLAGS): ``-arch=$(CUDA_ARCH)`` (or the -gencode preset), the +-DCOLI_DSV4_NO_TC guard and the DeepGEMM defines. Timestamps cannot tell those +apart, so without a record of the command make reports "is up to date" and +reuses the previous object (#1702 is the C-side case; #1707 fixed the units, +this is the nvcc side): + + make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 CUDA_ARCH=sm_86 + make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 CUDA_ARCH=sm_80 + -> no output at all, still the sm_86 engine + +The parent Makefile records CUDA_ARCH in .build-config for this reason (#306); +the standalone build keeps its own stamp, deepseek_v4.cudaflags. + +The nvcc here is a stand-in that only creates its -o target and appends its own +command line to a log, so the test needs make and sh but no CUDA toolkit and no +GPU, and it checks exactly what make decides to recompile. The sources are +copied, not linked, so their timestamps can be pinned without touching the tree. +""" +import os +import shutil +import subprocess +import tempfile +import time +import unittest +from pathlib import Path + +C_DIR = Path(__file__).resolve().parent.parent +MAKE = shutil.which("make") +OBJ = "backend_cuda_dsv4.o" +STAMP = "deepseek_v4.cudaflags" + +FAKE_NVCC = """#!/bin/sh +# Stand-in nvcc: records the command line, creates the -o target, compiles nothing. +printf '%s\\n' "$*" >> "$FAKE_NVCC_LOG" +while [ "$#" -gt 0 ]; do + if [ "$1" = "-o" ]; then : > "$2"; exit 0; fi + shift +done +exit 1 +""" + + +@unittest.skipUnless(MAKE and os.name == "posix", "make and a POSIX shell are required") +class DeepseekV4CudaArchFlagsTest(unittest.TestCase): + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.dir = Path(self._tmp.name) + for pattern in ("Makefile.deepseek-v4*", "deepseek_v4.c", "*.h", "*.inc", "*.cu"): + for src in C_DIR.glob(pattern): + shutil.copy(src, self.dir / src.name) + self.nvcc = self.dir / "fake-nvcc.sh" + self.nvcc.write_text(FAKE_NVCC, encoding="utf-8") + self.nvcc.chmod(0o755) + self.log = self.dir / "nvcc.log" + # Sources well in the past: every object built below is newer than all of + # them, so the only thing left that can make the object stale is a flag. + past = time.time() - 1000 + for f in self.dir.iterdir(): + os.utime(f, (past, past)) + probe = self.make("print-v4-objs") + if probe.returncode != 0: + self.skipTest("Makefile.deepseek-v4 does not support this host: " + + probe.stderr.strip()[-200:]) + + def tearDown(self): + self._tmp.cleanup() + + def make(self, *args): + env = dict(os.environ, FAKE_NVCC_LOG=str(self.log)) + return subprocess.run( + [MAKE, "-f", "Makefile.deepseek-v4", f"NVCC={self.nvcc}", *args], + cwd=self.dir, text=True, capture_output=True, check=False, + timeout=120, env=env) + + def nvcc_calls(self): + return len(self.log.read_text(encoding="utf-8").split("\n")) - 1 if self.log.exists() else 0 + + def build(self, *variables): + """Build the object, then age it so only a flags change can rebuild it. + + Leaves the object and the stamp older than now but newer than the + sources, so a later build recompiles only if make finds a real reason, + never because two writes landed in the same clock tick. + """ + self.log.write_text("", encoding="utf-8") + result = self.make(OBJ, *variables) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + calls = self.nvcc_calls() + aged = time.time() - 500 + for f in self.dir.iterdir(): + if f.stat().st_mtime > aged + 400: + os.utime(f, (aged, aged)) + return calls + + def test_same_arch_does_not_rebuild(self): + """The guard: a fix that always recompiles would pass the tests below.""" + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_86"), 1) + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_86"), 0, + f"{OBJ} was recompiled with an unchanged nvcc command line") + + def test_changed_arch_rebuilds(self): + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_86"), 1) + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_80"), 1, + f"{OBJ} built for sm_86 was reused by an sm_80 build: the " + "engine links an object for the wrong architecture") + + def test_changed_gencode_preset_rebuilds(self): + """The presets change -gencode just as -arch does.""" + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_86"), 1) + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=portable"), 1, + f"{OBJ} was reused after the gencode preset changed") + + def test_changed_no_tc_rebuilds(self): + """NO_TC adds -DCOLI_DSV4_NO_TC to the nvcc line, same object name.""" + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_86", "NO_TC=0"), 1) + self.assertEqual(self.build("CUDA=1", "CUDA_ARCH=sm_86", "NO_TC=1"), 1, + f"{OBJ} was reused after NO_TC changed") + + def test_a_dry_run_writes_nothing(self): + """make -n must not create build state. The stamp is written by a recipe, + not while the Makefile is read, so a dry run only prints it.""" + dry = self.make("-n", OBJ, "CUDA=1", "CUDA_ARCH=sm_86") + self.assertEqual(dry.returncode, 0, dry.stdout + dry.stderr) + self.assertFalse((self.dir / STAMP).exists(), + f"make -n created {STAMP}") + self.assertEqual(self.nvcc_calls(), 0, "make -n ran nvcc") + clean = self.make("-n", "deepseek-v4-clean") + self.assertEqual(clean.returncode, 0, clean.stdout + clean.stderr) + self.assertFalse((self.dir / STAMP).exists(), + f"make -n deepseek-v4-clean created {STAMP}") + + def test_clean_writes_nothing(self): + """A clean recipe that removes the stamp must not recreate it first.""" + self.build("CUDA=1", "CUDA_ARCH=sm_86") + result = self.make("deepseek-v4-clean") + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertFalse((self.dir / STAMP).exists(), + f"{STAMP} survived deepseek-v4-clean") + + def test_clean_removes_the_stamp(self): + """A stale stamp after `make clean` would make the next build a no-op.""" + self.build("CUDA=1", "CUDA_ARCH=sm_86") + subprocess.run([MAKE, "-f", "Makefile.deepseek-v4", "deepseek-v4-clean"], + cwd=self.dir, text=True, capture_output=True, check=False, + timeout=120) + self.assertFalse((self.dir / STAMP).exists(), + f"{STAMP} survived deepseek-v4-clean") + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_degrade_zero.c b/c/tests/test_degrade_zero.c new file mode 100644 index 000000000..2eb5d11b3 --- /dev/null +++ b/c/tests/test_degrade_zero.c @@ -0,0 +1,383 @@ +/* test_degrade_zero.c — unit tests for the DEGRADE_ZERO miss-slot zero-fill logic. + * + * Strategy: mirrors test_ablate.c — a standalone mini-harness that reimplements + * only the routing data structures the DEGRADE_ZERO block in moe() touches. + * No colibri.c include, no Model, no weights, no disk I/O needed. + * + * The block under test (colibri.c, "DEGRADE_ZERO: zero-fill miss slots..."): + * 1. Marks cache hits as always-keep. + * 2. Per-position tau gate: a miss expert is kept if ANY position routes to it + * with weight >= tau (each position tested independently, not aggregated). + * 3. Rescue rule: if all of a position's experts would be dropped, reinstates + * the highest-gate-weight miss so no position has zero routed experts. + * 4. Rewrites idxs[]/ws[]/keff[] removing dropped experts, NO renorm — + * survivors keep their original weights; compacts uniq[]; increments + * g_degrade_dropped. + * + * Properties verified: + * P1 OFF-BY-DEFAULT — with g_degrade_zero=0, routing is byte-identical to + * the input (nothing is dropped or changed). + * P2 HITS-ARE-SAFE — experts already resident (simulated via a mock + * "resident" set) are never dropped regardless of gate weight. + * P3 TAU-GATE — cold misses with per-position weight >= tau are kept; + * those below tau are dropped and counted in g_degrade_dropped. + * P4 NO-RENORM — surviving weights are unchanged after dropping; the + * dropped mass is simply lost (not redistributed). + * P5 PER-POSITION-TAU — at S>1 a shared expert is kept as long as ONE + * position's weight meets tau; aggregate does not govern. + * P6 RESCUE-RULE — when all of a position's experts are cold misses below + * tau, the highest-weight one is reinstated; g_degrade_dropped is NOT + * inflated for the rescued expert. + * P7 PREFILL-GUARD — with S>4 (prefill batch) the block is a no-op; nothing + * is dropped even if g_degrade_zero=1 and all experts are below tau. + * P8 COUNTER — g_degrade_dropped accumulates correctly across calls. + * P9 BIT-IDENTICAL — with flag off, idxs[], ws[], keff[], and uniq[] are + * byte-identical to the input even when sub-tau experts are present. + */ + +#include +#include +#include +#include + +static int g_fails = 0; +#define CHECK(cond, msg) do { \ + if (!(cond)) { printf(" FAIL: %s\n", (msg)); g_fails++; } \ + else { printf(" ok: %s\n", (msg)); } \ +} while (0) +#define CHECKF(cond, msg, ...) do { \ + if (!(cond)) { printf(" FAIL: " msg "\n", __VA_ARGS__); g_fails++; } \ + else { printf(" ok: " msg "\n", __VA_ARGS__); } \ +} while (0) + +/* ---- mirror of the globals the block reads -------------------------------- */ +static int g_degrade_zero = 0; +static float g_degrade_tau = 0.03f; +static long long g_degrade_dropped = 0; + + +/* ---- resident-set mock: a flat array of expert ids considered "in cache" -- */ +#define MAX_RESIDENT 32 +static int g_resident[MAX_RESIDENT]; +static int g_nresident = 0; + +static int is_resident(int eid) { + for (int i = 0; i < g_nresident; i++) + if (g_resident[i] == eid) return 1; + return 0; +} + +/* ---- the drop logic extracted verbatim from colibri.c moe() --------------- * + * Parameters match the local variables in moe() at the insertion point: + * idxs[S*K], ws[S*K], keff[S], uniq[nu], nu, S, K. + * Returns the new nu after compaction. */ +static int degrade_zero_apply(int *idxs, float *ws, int *keff, + int *uniq, int nu, int S, int K) +{ + if (!g_degrade_zero || S > 4) return nu; + + /* 1. residency scan: hits always kept */ + unsigned char *dg_keep = calloc((size_t)nu, 1); + for (int j = 0; j < nu; j++) + if (is_resident(uniq[j])) dg_keep[j] = 1; + + /* 2. per-position tau gate: keep a miss expert if ANY position routes to it + * with weight >= tau (each position tested independently, not aggregated) */ + for (int s = 0; s < S; s++) + for (int kk = 0; kk < keff[s]; kk++) { + float wv = ws[s * K + kk]; + if (wv >= g_degrade_tau) { + int e = idxs[s * K + kk]; + for (int j = 0; j < nu; j++) + if (uniq[j] == e) { dg_keep[j] = 1; break; } + } + } + + /* 3. rescue: no position may end up with zero routed experts */ + /* build seen[] from current dg_keep */ + int *seen = calloc((size_t)256, sizeof(int)); /* expert ids < 256 in tests */ + for (int j = 0; j < nu; j++) if (dg_keep[j]) seen[uniq[j]] = 1; + for (int s = 0; s < S; s++) { + int alive = 0; + for (int kk = 0; kk < keff[s] && !alive; kk++) + if (seen[idxs[s * K + kk]]) alive = 1; + if (alive || keff[s] <= 0) continue; + /* reinstate highest-gate-weight miss */ + int be = -1; float bw = -1e30f; + for (int kk = 0; kk < keff[s]; kk++) { + float wv = ws[s * K + kk]; + if (wv > bw) { bw = wv; be = idxs[s * K + kk]; } + } + if (be < 0) be = idxs[s * K]; + seen[be] = 1; + for (int j = 0; j < nu; j++) + if (uniq[j] == be && !dg_keep[j]) { dg_keep[j] = 1; break; } + } + free(seen); + + /* 4. count dropped, then apply */ + int dg_dropped = 0; + for (int j = 0; j < nu; j++) if (!dg_keep[j]) dg_dropped++; + + if (dg_dropped) { + g_degrade_dropped += dg_dropped; + + /* rebuild seen[] from final dg_keep */ + int *seen2 = calloc((size_t)256, sizeof(int)); + for (int j = 0; j < nu; j++) if (dg_keep[j]) seen2[uniq[j]] = 1; + + /* no renorm: survivors keep original weights — the approximation IS the + * dropped mass; renorm would hide it and bias the output upward */ + for (int s = 0; s < S; s++) { + int w = 0; + for (int kk = 0; kk < keff[s]; kk++) { + int e = idxs[s * K + kk]; float wv = ws[s * K + kk]; + if (seen2[e]) { idxs[s * K + w] = e; ws[s * K + w] = wv; w++; } + } + if (w < keff[s]) keff[s] = w; + } + + /* compact uniq[] */ + int nu2 = 0; + for (int j = 0; j < nu; j++) if (dg_keep[j]) uniq[nu2++] = uniq[j]; + nu = nu2; + free(seen2); + } + + free(dg_keep); + return nu; +} + +/* ---- helpers -------------------------------------------------------------- */ +static float sum_weights(const float *ws, int K, const int *keff, int S) { + float s = 0; + for (int i = 0; i < S; i++) + for (int kk = 0; kk < keff[i]; kk++) + s += ws[i * K + kk]; + return s; +} + +static int expert_in_uniq(const int *uniq, int nu, int eid) { + for (int j = 0; j < nu; j++) if (uniq[j] == eid) return 1; + return 0; +} + +static int expert_in_routing(const int *idxs, const int *keff, int S, int K, int eid) { + for (int s = 0; s < S; s++) + for (int kk = 0; kk < keff[s]; kk++) + if (idxs[s * K + kk] == eid) return 1; + return 0; +} + +/* ---- tests ---------------------------------------------------------------- */ + +/* P1: g_degrade_zero=0 — block is a no-op */ +static void test_off_by_default(void) { + printf("\nP1: off-by-default\n"); + /* S=1, K=2: experts 10 (w=0.8) and 11 (w=0.01, below tau) */ + int idxs[2] = {10, 11}; + float ws[2] = {0.8f, 0.01f}; + int keff[1] = {2}; + int uniq[2] = {10, 11}; int nu = 2; + + g_degrade_zero = 0; + g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 2); + + CHECK(nu == 2, "nu unchanged when off"); + CHECK(keff[0] == 2, "keff unchanged when off"); + CHECK(g_degrade_dropped == 0, "counter unchanged when off"); +} + +/* P2: resident experts are never dropped regardless of gate weight */ +static void test_hits_are_safe(void) { + printf("\nP2: hits-are-safe\n"); + /* Expert 5 is resident with gate weight 0.001 (well below tau=0.03) */ + g_resident[0] = 5; g_nresident = 1; + /* S=1, K=2: expert 5 (resident, w=0.001) and expert 7 (miss, w=0.8) */ + int idxs[2] = {5, 7}; + float ws[2] = {0.001f, 0.8f}; + int keff[1] = {2}; + int uniq[2] = {5, 7}; int nu = 2; + + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 2); + + CHECK(expert_in_uniq(uniq, nu, 5), "resident expert 5 kept despite low gate weight"); + CHECK(expert_in_uniq(uniq, nu, 7), "above-tau miss expert 7 kept"); + CHECK(g_degrade_dropped == 0, "nothing dropped"); + + g_nresident = 0; +} + +/* P3: cold misses below tau are dropped; at/above tau are kept */ +static void test_tau_gate(void) { + printf("\nP3: tau-gate\n"); + /* S=1, K=3: expert 1 (w=0.7, above tau), expert 2 (w=0.03, exactly tau), + * expert 3 (w=0.02, below tau) — all cold misses */ + int idxs[3] = {1, 2, 3}; + float ws[3] = {0.7f, 0.03f, 0.02f}; + int keff[1] = {3}; + int uniq[3] = {1, 2, 3}; int nu = 3; + + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 3); + + CHECK(expert_in_uniq(uniq, nu, 1), "expert 1 (w=0.7) kept"); + CHECK(expert_in_uniq(uniq, nu, 2), "expert 2 (w=0.03, exactly tau) kept"); + CHECK(!expert_in_uniq(uniq, nu, 3), "expert 3 (w=0.02, below tau) dropped"); + CHECK(g_degrade_dropped == 1, "counter incremented by 1"); + CHECK(keff[0] == 2, "keff reduced to 2"); +} + +/* P4: no-renorm — surviving weights are unchanged after a drop */ +static void test_no_renorm(void) { + printf("\nP4: no-renorm\n"); + /* S=1, K=2: expert 10 (w=0.6, keep), expert 11 (w=0.02, drop). + * Survivor must keep its original weight 0.6 — not rescaled. */ + int idxs[2] = {10, 11}; + float ws[2] = {0.6f, 0.02f}; + int keff[1] = {2}; + int uniq[2] = {10, 11}; int nu = 2; + + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 2); + + CHECK(nu == 1, "uniq compacted to 1"); + CHECK(keff[0] == 1, "keff=1 after drop"); + CHECKF(fabsf(ws[0] - 0.6f) < 1e-6f, + "survivor weight unchanged (got %.6f, want 0.600000)", ws[0]); +} + +/* P5: per-position tau gate — at S=2 a shared expert is kept when ONE + * position's weight meets tau even if the other's does not */ +static void test_per_position_tau(void) { + printf("\nP5: per-position tau gate\n"); + /* S=2, K=1: both positions route to expert 20. + * Position 0: w=0.01 (below tau), position 1: w=0.04 (above tau). + * Per-position: expert 20 kept (position 1 meets tau). + * Aggregate would also keep it (0.05 >= 0.03) — so test the case where + * aggregate would DROP but per-position keeps: shared expert 21 with + * pos0=0.02, pos1=0.04. Expert 22: pos0=0.01, pos1=0.01 (both below, + * aggregate 0.02 < 0.03) — must be dropped. */ + int idxs[4] = {21, 22, /* pos 0: experts 21, 22 */ + 21, 22}; /* pos 1: experts 21, 22 */ + float ws[4] = {0.02f, 0.01f, /* pos 0 weights */ + 0.04f, 0.01f}; /* pos 1 weights */ + int keff[2] = {2, 2}; + int uniq[2] = {21, 22}; int nu = 2; + + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 2, 2); + + CHECK(expert_in_uniq(uniq, nu, 21), "expert 21 kept: pos 1 has w=0.04 >= tau"); + CHECK(!expert_in_uniq(uniq, nu, 22), "expert 22 dropped: all positions below tau"); + CHECK(g_degrade_dropped == 1, "counter=1"); +} + +/* P6: rescue rule — all experts are cold misses below tau */ +static void test_rescue_rule(void) { + printf("\nP6: rescue rule\n"); + /* S=1, K=2: both experts are cold misses below tau. + * Expert 30 has the higher gate weight — it must be rescued. */ + int idxs[2] = {30, 31}; + float ws[2] = {0.025f, 0.01f}; + int keff[1] = {2}; + int uniq[2] = {30, 31}; int nu = 2; + + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 2); + + CHECK(keff[0] >= 1, "position not left with 0 routed experts"); + CHECK(expert_in_routing(idxs, keff, 1, 2, 30), "highest-weight expert 30 rescued"); + CHECK(!expert_in_routing(idxs, keff, 1, 2, 31), "lower-weight expert 31 dropped"); + /* only expert 31 counts as dropped; rescued expert 30 does not */ + CHECK(g_degrade_dropped == 1, "only truly-dropped expert counted"); +} + +/* P7: prefill guard — S>4 means the block must be a complete no-op */ +static void test_prefill_guard(void) { + printf("\nP7: prefill guard (S=8)\n"); + /* S=8, K=1: all experts are cold misses well below tau */ + int idxs[8] = {0, 1, 2, 3, 4, 5, 6, 7}; + float ws[8] = {0.001f, 0.001f, 0.001f, 0.001f, 0.001f, 0.001f, 0.001f, 0.001f}; + int keff[8] = {1, 1, 1, 1, 1, 1, 1, 1}; + int uniq[8] = {0, 1, 2, 3, 4, 5, 6, 7}; int nu = 8; + + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 8, 1); + + CHECK(nu == 8, "uniq unchanged for prefill batch"); + CHECK(g_degrade_dropped == 0, "counter unchanged for prefill batch"); +} + +/* P8: counter accumulates correctly across two calls */ +static void test_counter_accumulates(void) { + printf("\nP8: counter accumulates across calls\n"); + g_degrade_zero = 1; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + + /* call 1: drop 1 expert */ + { + int idxs[2] = {40, 41}; float ws[2] = {0.8f, 0.01f}; + int keff[1] = {2}; int uniq[2] = {40, 41}; int nu = 2; + degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 2); + } + CHECK(g_degrade_dropped == 1, "counter=1 after first call"); + + /* call 2: drop 2 experts */ + { + int idxs[3] = {50, 51, 52}; float ws[3] = {0.8f, 0.01f, 0.01f}; + int keff[1] = {3}; int uniq[3] = {50, 51, 52}; int nu = 3; + degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 3); + } + CHECK(g_degrade_dropped == 3, "counter=3 after second call (cumulative)"); +} + +/* P9: bit-identical when off — idxs[], ws[], keff[], and uniq[] are all + * unchanged when g_degrade_zero=0, including sub-tau experts that would + * be dropped if the flag were on. */ +static void test_bit_identical_when_off(void) { + printf("\nP9: bit-identical when off\n"); + /* S=1, K=3: experts with a mix of weights, some below tau */ + int idxs[3] = {10, 20, 30}; + float ws[3] = {0.7f, 0.02f, 0.01f}; + int keff[1] = {3}; + int uniq[3] = {10, 20, 30}; int nu = 3; + + /* take copies to compare against after the call */ + int idxs_ref[3]; memcpy(idxs_ref, idxs, sizeof(idxs)); + float ws_ref[3]; memcpy(ws_ref, ws, sizeof(ws)); + int keff_ref = keff[0]; + + g_degrade_zero = 0; g_degrade_tau = 0.03f; g_degrade_dropped = 0; + nu = degrade_zero_apply(idxs, ws, keff, uniq, nu, 1, 3); + + CHECK(nu == 3, "nu unchanged"); + CHECK(keff[0] == keff_ref, "keff unchanged"); + CHECK(idxs[0]==idxs_ref[0] && idxs[1]==idxs_ref[1] && idxs[2]==idxs_ref[2], + "idxs[] byte-identical"); + CHECK(ws[0]==ws_ref[0] && ws[1]==ws_ref[1] && ws[2]==ws_ref[2], + "ws[] byte-identical"); + CHECK(g_degrade_dropped == 0, "counter unchanged"); +} + +/* ---- main ----------------------------------------------------------------- */ +int main(void) { + printf("test_degrade_zero: DEGRADE_ZERO miss-slot zero-fill logic\n"); + test_off_by_default(); + test_hits_are_safe(); + test_tau_gate(); + test_no_renorm(); + test_per_position_tau(); + test_rescue_rule(); + test_prefill_guard(); + test_counter_accumulates(); + test_bit_identical_when_off(); + printf("\n"); + if (g_fails) { + printf("test_degrade_zero: %d FAILED\n", g_fails); + return 1; + } + printf("test_degrade_zero: all tests passed\n"); + return 0; +} diff --git a/c/tests/test_doctor.py b/c/tests/test_doctor.py index 419347c49..3e2d50298 100644 --- a/c/tests/test_doctor.py +++ b/c/tests/test_doctor.py @@ -242,6 +242,22 @@ def test_windows_cpu_only_engine_is_not_a_gpu_build(self): self.assertEqual(self._win_linkage(engine), {"linked": False, "missing": False}) + def test_windows_kimi_cuda_host_without_glm_banner(self): + """Kimi K3 CUDA_DLL=1 never prints [CUDA] mode: routed experts. + The loader still compiles coli_cuda.dll into the host, and that is + the artifact doctor must require -- the same pair cuda_binary uses.""" + engine = self.root / "kimi_k3.exe" + engine.write_bytes(b"MZ [K3-CUDA] MXFP4 routed experts coli_cuda.dll") + (self.root / "coli_cuda.dll").write_bytes(b"") + self.assertEqual(self._win_linkage(engine), + {"linked": True, "missing": False}) + + def test_windows_kimi_cuda_host_missing_dll_is_missing(self): + engine = self.root / "kimi_k3.exe" + engine.write_bytes(b"MZ [K3-CUDA] MXFP4 routed experts coli_cuda.dll") + self.assertEqual(self._win_linkage(engine), + {"linked": False, "missing": True}) + def test_hip_host_with_its_backend_reports_gpu_available(self): # End-to-end: the same host that previously reported a hard error # ("GPU runtime library is missing") now passes, with #903's diff --git a/c/tests/test_dsv41_prefix_serve.py b/c/tests/test_dsv41_prefix_serve.py index 949b0f5e4..d7413e6e8 100644 --- a/c/tests/test_dsv41_prefix_serve.py +++ b/c/tests/test_dsv41_prefix_serve.py @@ -30,6 +30,55 @@ FIXTURE = Path(os.environ.get("DSV41_TINY", HERE / "dsv41_tiny")) +@unittest.skipUnless(ENGINE.exists() and (FIXTURE / "model.safetensors").exists(), + "built V4.1 engine and tiny fixture required") +class Dsv41ContextTest(unittest.TestCase): + def setUp(self): + self.engine = ServeEngine(ENGINE, ["8"], + {"SNAP": str(FIXTURE), "SERVE": "1", "CTX": "32", + "OMP_NUM_THREADS": "2", "V41_DSPARK": "0"}, reuse=False) + self.addCleanup(self.engine.p.stdout.close) + self.addCleanup(self.engine.p.stderr.close) + self.addCleanup(self.engine.close) + + def request(self, size, budget, logprobs=0): + prompt = b"a" * size # the fixture's tokenizer has one id per byte + header = f"SUBMIT ctx 0 {size} {budget} 0 1 logprobs={logprobs}\n".encode() + self.engine.p.stdin.write(header + prompt + b"\n") + self.engine.p.stdin.flush() + frames = [] + while True: + line = self.engine.p.stdout.readline() + self.assertTrue(line, "engine closed before completing the request") + fields = line.decode().split() + if fields[0] in ("DATA", "ECHO"): + self.engine.p.stdout.read(int(fields[2])) + self.engine.p.stdout.readline() + frames.append(fields) + if fields[0] in ("DONE", "ERROR"): + return frames + + def test_output_ceiling_leaves_one_token(self): + frames = self.request(31, 5000) + self.assertEqual(frames[-1][:4], ["DONE", "ctx", "STAT", "1"]) + self.assertIn(["ACCEPT", "ctx", "31"], frames) + + def test_full_context_is_valid_only_for_scoring(self): + frames = self.request(32, 0, logprobs=1) + self.assertEqual(frames[-1][:4], ["DONE", "ctx", "STAT", "0"]) + self.assertEqual(sum(f[0] == "ECHO" for f in frames), 31) + self.assertFalse(any(f[0] == "DATA" for f in frames)) + frames = self.request(32, 1) + self.assertEqual(frames[-1][:3], ["ERROR", "ctx", "CONTEXT_EXCEEDED"]) + + def test_oversized_scoring_prompt_is_not_truncated(self): + frames = self.request(33, 0, logprobs=1) + self.assertEqual(frames[-1][:3], ["ERROR", "ctx", "CONTEXT_EXCEEDED"]) + self.assertFalse(any(f[0] in ("ACCEPT", "ECHO", "DATA") for f in frames)) + # Refusal drains the request and leaves the next one usable. + self.assertEqual(self.request(4, 1)[-1][:4], ["DONE", "ctx", "STAT", "1"]) + + @unittest.skipUnless(ENGINE.exists(), "deepseek_v41 is not built") @unittest.skipUnless((FIXTURE / "model.safetensors").exists(), "tiny dsv41 fixture is absent (tools/make_dsv41_tiny.py)") diff --git a/c/tests/test_dsv41_serve_budget.c b/c/tests/test_dsv41_serve_budget.c new file mode 100644 index 000000000..6477527f4 --- /dev/null +++ b/c/tests/test_dsv41_serve_budget.c @@ -0,0 +1,28 @@ +/* The serve contract accepts a fitting prompt even if the requested output + * budget is too large, and never turns a read-only request into generation. */ +#define main dsv41_main_unused +#include "../deepseek_v41.c" +#undef main +#include + +#define CHECK(c) do { if (!(c)) { \ + fprintf(stderr, "%s:%d: %s\n", __FILE__, __LINE__, #c); return 1; \ +} } while (0) + +int main(void) { + CHECK(serve_budget(20, 5000, 4096, 0) == 4076); + CHECK(serve_budget(20, 10, 4096, 0) == 10); + CHECK(serve_budget(4095, 5000, 4096, 0) == 1); + CHECK(serve_budget(4096, 1, 4096, 0) == -1); + CHECK(serve_budget(4097, 1, 4096, 0) == -1); + CHECK(serve_budget(0, 1, 4096, 0) == -1); + CHECK(serve_budget(20, 0, 4096, 0) == 256); + CHECK(serve_budget(4095, 0, 4096, 0) == 1); + CHECK(serve_budget(4096, 0, 4096, 1) == 0); + CHECK(serve_budget(4097, 0, 4096, 1) == -1); + CHECK(serve_budget(20, 0, 4096, 1) == 0); + CHECK(serve_budget(4095, 100, 4096, 1) == 1); + CHECK(serve_budget(20, INT_MAX, 4096, 0) == 4076); + puts("dsv41 serve budget: ok"); + return 0; +} diff --git a/c/tests/test_e8x4g64_loader.c b/c/tests/test_e8x4g64_loader.c new file mode 100644 index 000000000..32fb17072 --- /dev/null +++ b/c/tests/test_e8x4g64_loader.c @@ -0,0 +1,108 @@ +/* wq-v0-class (e8x4g64: int8 spine + grouped-int4 g64 experts) END-TO-END + * loader harness -- the C-side half of the mint->load regression case. + * + * tests/test_e8x4g64_mint_load.py drives it: builds a synthetic FP8 GLM-shaped + * checkpoint (tools/glm_fp8_emit.py, the same fixture helper the fp8 + * passthrough e2e uses), mints it with the REAL tools/convert_fp8_to_int4.py + * CLI at the e8x4g64 recipe (--ebits 8 --xbits 4 --io-bits 8 --group-size 64), + * then invokes this binary against the real output directory. Sibling of + * tests/test_fp8_e2e_loader.c (fmt=8 passthrough), same division of labor: + * the Python side proves the TOOL ran for real; this side proves the REAL + * st_init/qt_from_disk (the identical functions every model load uses) + * resolve every minted tensor to the format the recipe promises. + * + * Usage: test_e8x4g64_loader [name O I wantfmt wantgs]... + * For every (name,O,I,wantfmt,wantgs) 5-tuple, calls the REAL qt_from_disk and + * asserts: + * (a) the resolved fmt equals wantfmt (1 = int8 per-row spine, 4 = grouped + * int4 experts) and, for fmt=4, the derived group size equals wantgs -- + * both resolved by qt_resolve_fmt's byte arithmetic on the tool's actual + * output, neither mocked nor stamped; + * (b) the weight and scale buffers are non-NULL; + * (c) every dequantized value is finite -- catches a scale-layout or + * packing bug a pure byte-count check wouldn't. + * + * The D-2 duplicate-name refusal is exercised through this same binary: the + * Python driver runs it a second time against a copy of the container holding + * a duplicated shard, and st_init below then refuses (exit 1, naming both + * shards) before any tuple is checked -- the positive half of the D-2 pair, + * on a tool-produced container. The clean run doubles as the negative + * control: a legitimate mint loads with no refusal. */ +#define main coli_glm_main_unused +#include "../colibri.c" +#undef main + +#include +#include +#include +#include + +int main(int argc, char **argv){ + if(argc < 2){ fprintf(stderr,"usage: %s [name O I wantfmt wantgs]...\n", argv[0]); return 2; } + if((argc-2) % 5 != 0){ + fprintf(stderr,"args after must come in (name,O,I,wantfmt,wantgs) 5-tuples (got %d)\n", argc-2); + return 2; + } + const char *dir = argv[1]; + int ntensors = (argc-2)/5; + if(ntensors == 0){ fprintf(stderr,"no tuples given -- nothing to check\n"); return 2; } + + static Model gm; memset(&gm,0,sizeof gm); + st_init(&gm.S, dir); /* D-2 duplicate-name detection lives in here (st.h) */ + + int fails = 0; + for(int i=0;i>1]; int nib=(ii&1)?(b>>4):(b&0xF); + float v=((int)nib-8)*scl[ii/t.gs]; + if(!isfinite(v)){ bad=o*(int64_t)I+ii; break; } } } + } else { + printf("FAIL %s: this harness only knows the e8x4g64 formats (1, 4); " + "wantfmt=%d is a driver bug\n", name, wantfmt); + fails++; continue; + } + if(bad >= 0){ + printf("FAIL %s: non-finite dequantized value at flat index %lld\n", name, (long long)bad); + fails++; continue; + } + printf("ok %s: fmt=%d%s O=%d I=%d, loaded through the real loader, all-finite\n", + name, t.fmt, t.fmt==4?" (g64)":"", O, I); + } + if(fails){ printf("e8x4g64 mint->load: %d/%d tensor(s) FAILED\n", fails, ntensors); return 1; } + printf("e8x4g64 mint->load: ok (%d tensor(s))\n", ntensors); + return 0; +} diff --git a/c/tests/test_e8x4g64_mint_load.py b/c/tests/test_e8x4g64_mint_load.py new file mode 100644 index 000000000..7f506950a --- /dev/null +++ b/c/tests/test_e8x4g64_mint_load.py @@ -0,0 +1,297 @@ +"""wq-v0-class regression case: REAL tools/convert_fp8_to_int4.py output fed +into the REAL C loader (st_init/qt_from_disk in colibri.c) at toy scale. + +The wq-v0 production container class is "e8x4g64": int8 per-row spine +(attention / shared expert / dense MLP / embed / lm_head -> fmt=1) with +grouped-int4 g64 routed experts (fmt=4, gs=64). This test mints that class +each cycle at toy scale with GLM-shaped tensor names and the load-bearing +real dimensions (kv_lora_rank-width kv_b contraction, group-64-divisible +expert dims) through the real converter CLI -- not a reimplementation of its +quantizers -- and then proves the REAL C loader resolves every tensor to the +format the recipe promises. Regression coverage, not new-feature proof: the +full-scale wq-v0 container is already banked and audited; what this pins is +that the TOOLING and the LOADER still agree on the class after engine +changes (the mint half of the mint->load->run regression bar; the run half +executes at full scale against the banked container, outside this suite). + +Also carried here, because it belongs to exactly this container class: the +D-2 duplicate-tensor-name pair on a TOOL-PRODUCED container. + * positive: a copy of the minted container with one shard duplicated must + refuse at st_init with the exact D-2 message (naming both shards), before + any tensor is read; + * negative control: the untouched mint loads clean through the same + binary -- the guard is duplicate-targeted, not a blanket gate. +(tests/test_dup_name_refusal.c pins the same guard on hand-built fixtures; +this is the tool-produced end of it.) + +Hermetic: checkpoint, minted output, duplicated copy, and the compiled +harness all live under temporary directories; nothing is left behind. +""" +import glob +import os +import shutil +import subprocess +import sys +import tempfile +import unittest + +try: + import torch +except ImportError as e: + raise unittest.SkipTest(f"torch not installed: {e}") + +try: + import numpy as np +except ImportError as e: + raise unittest.SkipTest(f"numpy not installed: {e}") + +HERE = os.path.dirname(os.path.abspath(__file__)) +C_DIR = os.path.normpath(os.path.join(HERE, "..")) +sys.path.insert(0, os.path.join(C_DIR, "tools")) +# reuse: the real fp8 block-quantize fixture helpers, not reimplemented. The +# quantize/dequantize pair anchors the numeric reference below to EXACTLY what +# the converter ingests (its dequant() applies the same repeat_interleave +# formula to the same saved fp8 bytes); the converter's own int8/int4 +# quantizers are NOT imported anywhere in this file -- their output is decoded +# and bounded by independent code below. +from glm_fp8_emit import save_fp8_safetensors, fp8_block_quantize, fp8_block_dequantize + + +def _cc_flags(): + """Mirror the Makefile's CFLAGS closely enough to compile colibri.c + cleanly (same arrangement as tests/test_fp8_e2e_repack_load.py).""" + cc = shutil.which("cc") or shutil.which("clang") or shutil.which("gcc") + if not cc: + return None, None, None + cflags = ["-O3", "-Wall", "-Wextra", "-Wno-unused-parameter", + "-Wno-misleading-indentation", "-Wno-unused-function"] + ldflags = ["-lm"] + if sys.platform not in ("darwin", "win32"): + cflags += ["-fopenmp"] + ldflags += ["-fopenmp"] + if sys.platform == "darwin": + # The Makefile builds this harness with OpenMP on macOS via Homebrew's + # libomp. Locating it must not fail quietly: a swallowed error here + # compiles a DIFFERENT binary from the one the suite builds, and the + # warning assertion below would then vouch for flags nobody ships. + # Report why it could not be found and let the caller decide. + try: + probe = subprocess.run(["brew", "--prefix", "libomp"], capture_output=True, + text=True, timeout=10) + prefix = probe.stdout.strip() + if probe.returncode != 0: + raise RuntimeError( + "`brew --prefix libomp` exited %d: %s" + % (probe.returncode, probe.stderr.strip() or "(no stderr)")) + except FileNotFoundError: + raise RuntimeError( + "libomp is required to build this harness on macOS the way the " + "Makefile builds it, and `brew` is not on PATH") + except (OSError, subprocess.TimeoutExpired) as exc: + raise RuntimeError("could not run `brew --prefix libomp`: %r" % (exc,)) + inc, lib = os.path.join(prefix, "include"), os.path.join(prefix, "lib") + if not (prefix and os.path.exists(os.path.join(inc, "omp.h"))): + raise RuntimeError( + "`brew --prefix libomp` gave %r but %s/omp.h does not exist; " + "install libomp (`brew install libomp`) rather than compiling " + "this harness without OpenMP" % (prefix, inc)) + cflags += ["-Xclang", "-fopenmp", "-I", inc] + ldflags += ["-L", lib, "-lomp"] + return cc, cflags, ldflags + + +# Toy scale, real geometry where it is load-bearing: +# * kv_b_proj keeps the REAL contraction width I=512 (GLM-5.2's +# kv_lora_rank) with a toy head count -> O=160; +# * routed-expert dims stay multiples of 64 (real containers are), so the +# g64 grouping has no synthetic tail the production mint never sees -- +# while D_H (o_proj/q_a rows, expert contraction) is NOT a multiple of +# 128, keeping the fp8 input's block scales on the partial-tail path; +# * no tensor sits on the fmt=1-vs-fmt=8 collision boundary +# (O == ceil(O/128)*ceil(I/128)), so byte-arithmetic resolution is +# unambiguous, as it is at full scale. +D_H = 320 # toy hidden size (multiple of 64, not of 128) +KV_B_O, KV_B_I = 160, 512 +E_M = 192 # toy expert intermediate (multiple of 64) +VOCAB = 512 + + +class E8x4g64MintLoadTest(unittest.TestCase): + """The real e8x4g64 mint recipe -> the real C loader.""" + + _harness_dir = None + _harness_bin = None + _build_stderr = None + + @classmethod + def setUpClass(cls): + cc, cflags, ldflags = _cc_flags() + if not cc: + raise unittest.SkipTest("no C compiler found on PATH") + cls._harness_dir = tempfile.TemporaryDirectory() + harness_src = os.path.join(HERE, "test_e8x4g64_loader.c") + harness_bin = os.path.join(cls._harness_dir.name, "test_e8x4g64_loader") + build = subprocess.run([cc] + cflags + [harness_src, "-o", harness_bin] + ldflags, + capture_output=True, text=True, cwd=C_DIR) + if build.returncode != 0: + raise AssertionError(f"e8x4g64 harness build failed:\n{build.stderr}") + cls._harness_bin = harness_bin + cls._build_stderr = build.stderr + + @classmethod + def tearDownClass(cls): + if cls._harness_dir: + cls._harness_dir.cleanup() + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.indir = os.path.join(self.tmp.name, "fp8src") + os.makedirs(self.indir) + self.outdir = os.path.join(self.tmp.name, "out") + self.shard = os.path.join(self.indir, "model-00001-of-00001.safetensors") + + def tearDown(self): + self.tmp.cleanup() + + def _emit_checkpoint(self): + """A GLM-shaped fp8 source checkpoint: every name below is one the real + converter's classify() routes exactly like the production checkpoint's + (kvb/attn/o -> ebits, sh/dmlp -> ebits, io -> io_bits, experts -> xbits, + norms -> f32 passthrough).""" + torch.manual_seed(11) + L = "model.layers.0" + sd = { + f"{L}.self_attn.kv_b_proj.weight": torch.randn(KV_B_O, KV_B_I) * 0.02, + f"{L}.self_attn.q_a_proj.weight": torch.randn(D_H, D_H) * 0.02, + f"{L}.self_attn.o_proj.weight": torch.randn(D_H, D_H) * 0.02, + f"{L}.mlp.shared_experts.gate_proj.weight": torch.randn(E_M, D_H) * 0.02, + f"{L}.mlp.experts.0.gate_proj.weight": torch.randn(E_M, D_H) * 0.02, + f"{L}.mlp.experts.0.up_proj.weight": torch.randn(E_M, D_H) * 0.02, + f"{L}.mlp.experts.0.down_proj.weight": torch.randn(D_H, E_M) * 0.02, + "model.embed_tokens.weight": torch.randn(VOCAB, D_H) * 0.02, + f"{L}.input_layernorm.weight": torch.randn(D_H), # f32 passthrough control + } + self.sd = sd # kept: the numeric round-trip test derives its reference from it + save_fp8_safetensors(sd, self.shard) + + def _mint(self): + """The REAL converter CLI at the wq-v0 e8x4g64 recipe: int8 spine + (--ebits 8, --io-bits 8), grouped-int4 g64 routed experts (--xbits 4, + --group-size 64).""" + tool = os.path.join(C_DIR, "tools", "convert_fp8_to_int4.py") + rc = subprocess.run([sys.executable, tool, "--indir", self.indir, + "--outdir", self.outdir, "--n-layers", "1", + "--ebits", "8", "--xbits", "4", "--io-bits", "8", + "--group-size", "64"], + capture_output=True, text=True) + self.assertEqual(rc.returncode, 0, + f"real converter failed:\nSTDOUT:\n{rc.stdout}\nSTDERR:\n{rc.stderr}") + outs = glob.glob(os.path.join(self.outdir, "out-*.safetensors")) + self.assertEqual(len(outs), 1, f"expected exactly one minted shard, got {outs}") + return outs[0] + + # 5-tuples for the C harness: (name, O, I, wantfmt, wantgs) + _EXPECT = [ + ("model.layers.0.self_attn.kv_b_proj.weight", KV_B_O, KV_B_I, 1, 0), + ("model.layers.0.self_attn.q_a_proj.weight", D_H, D_H, 1, 0), + ("model.layers.0.self_attn.o_proj.weight", D_H, D_H, 1, 0), + ("model.layers.0.mlp.shared_experts.gate_proj.weight", E_M, D_H, 1, 0), + ("model.embed_tokens.weight", VOCAB, D_H, 1, 0), + ("model.layers.0.mlp.experts.0.gate_proj.weight", E_M, D_H, 4, 64), + ("model.layers.0.mlp.experts.0.up_proj.weight", E_M, D_H, 4, 64), + ("model.layers.0.mlp.experts.0.down_proj.weight", D_H, E_M, 4, 64), + ] + + def _load(self, container_dir): + args = [self._harness_bin, container_dir] + for name, n_out, n_in, fmt, gs in self._EXPECT: + args += [name, str(n_out), str(n_in), str(fmt), str(gs)] + return subprocess.run(args, capture_output=True, text=True) + + def test_harness_builds_without_warnings(self): + """Production flags require zero warnings (pr-cycle mechanical gate).""" + self.assertEqual(self._build_stderr.strip(), "", + f"e8x4g64 harness build produced warnings:\n{self._build_stderr}") + + def test_e8x4g64_mint_loads_through_real_c_loader(self): + self._emit_checkpoint() + self._mint() + rc = self._load(self.outdir) + self.assertEqual(rc.returncode, 0, + f"loader harness failed:\nSTDOUT:\n{rc.stdout}\nSTDERR:\n{rc.stderr}") + for name, _, _, fmt, _ in self._EXPECT: + self.assertIn(f"ok {name}: fmt={fmt}", rc.stdout) + # negative D-2 control rides along: the clean mint produced NO + # duplicate-name refusal anywhere in the load + self.assertNotIn("duplicate tensor name", rc.stderr) + + def test_e8x4g64_numeric_round_trip(self): + """The minted VALUES, not just the format class (deep-audit MAJOR-1: + the fmt/finite checks alone passed a 37x scale corruption of + quant_int8 -- format class and finiteness both survive numeric + corruption). This closes DR-9(a)'s literal words, "mint ROUND-TRIP at + toy scale": dequantize the minted bytes with independent numpy code + (the converter's quantizers are never imported here) and require + agreement with the fp8-round-tripped source within the schemes' own + exact bound -- symmetric absmax rint quantization can never miss by + more than half a quantization step, so the tolerance is 0.5 steps + (+1e-3 for f32 arithmetic slop), derived from a scale RECOMPUTED from + the source, never from the minted .qs (a corrupted stored scale must + widen the error, not the bound).""" + self._emit_checkpoint() + minted = self._mint() + from safetensors.numpy import load_file + out = load_file(minted) + for name, n_out, n_in, fmt, gs in self._EXPECT: + # Reference: exactly what the converter ingested -- the fp8 + # block-quantized source, dequantized (bit-identical formula to + # the converter's dequant(); see the import comment above). + w_fp8, scale = fp8_block_quantize(self.sd[name].float()) + w_ref = fp8_block_dequantize(w_fp8, scale).numpy().astype(np.float32) + self.assertEqual(w_ref.shape, (n_out, n_in)) + q, qs = out[name], out[name + ".qs"] + if fmt == 1: + # int8 per-row: one scale per output row, q in [-127,127] + # (amax maps to qmax=127, so -128 is unreachable). + deq = q.view(np.int8).reshape(n_out, n_in).astype(np.float32) * qs.reshape(n_out, 1) + s_true = np.maximum(np.abs(w_ref).max(axis=1, keepdims=True) / 127.0, 1e-8) + step = np.broadcast_to(s_true, (n_out, n_in)) + else: + # int4 grouped g64: packed nibbles (low = even column), one + # scale per (row, 64-column group), q in [-7,7] likewise. + rb, ng = (n_in + 1) // 2, (n_in + gs - 1) // gs + b = q.reshape(n_out, rb) + deq = np.empty((n_out, n_in), np.float32) + deq[:, 0::2] = (b & 0xF).astype(np.float32)[:, : (n_in + 1) // 2] - 8.0 + deq[:, 1::2] = (b >> 4).astype(np.float32)[:, : n_in // 2] - 8.0 + deq *= np.repeat(qs.reshape(n_out, ng), gs, axis=1)[:, :n_in] + pad = np.zeros((n_out, ng * gs - n_in), np.float32) + grp = np.concatenate([np.abs(w_ref), pad], axis=1).reshape(n_out, ng, gs) + s_true = np.maximum(grp.max(axis=2) / 7.0, 1e-8) # [n_out, ng] + step = np.repeat(s_true, gs, axis=1)[:, :n_in] + worst = float((np.abs(deq - w_ref) / step).max()) + self.assertLessEqual( + worst, 0.5 + 1e-3, + f"{name}: minted values diverge from the source by {worst:.4g} " + f"quantization steps (bound: half a step) -- the converter's " + f"numeric output regressed even though the format class may " + f"still be intact") + + def test_d2_duplicate_shard_refuses_by_name(self): + """D-2 positive on the tool-produced container: duplicating the minted + shard under a second indexed name must refuse at st_init with the + exact duplicate-tensor-name message, before any tensor check runs.""" + self._emit_checkpoint() + minted = self._mint() + dupdir = os.path.join(self.tmp.name, "dup") + shutil.copytree(self.outdir, dupdir) + shutil.copy(minted, os.path.join(dupdir, "out-99999.safetensors")) + rc = self._load(dupdir) + self.assertNotEqual(rc.returncode, 0, "duplicated container must refuse to load") + self.assertIn("duplicate tensor name across indexed shards, refusing", rc.stderr) + self.assertNotIn("ok model.", rc.stdout, + "no tensor may load from a container st_init refused") + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_edge_adapters_real.c b/c/tests/test_edge_adapters_real.c index 65cec44f3..ac45247cd 100644 --- a/c/tests/test_edge_adapters_real.c +++ b/c/tests/test_edge_adapters_real.c @@ -154,6 +154,61 @@ int main(int argc, char **argv) { "tokenizer round-trip differs"); free(probe_text); probe_text = NULL; free(probe_ids); probe_ids = NULL; + + /* Qwen3.6 stores protocol markers outside model.vocab in tokenizer.json's + * added_tokens array. They must participate in both tokenization and + * detokenization just like ordinary vocabulary entries. */ + if (!strcmp(family, "qwen36")) { + static const struct { + const char *text; + int32_t id; + } added_token_cases[] = { + {"", 248058}, + {"", 248059}, + {"", 248066}, + {"", 248067}, + {"", 248068}, + {"", 248069}, + }; + + for (size_t token = 0; + token < sizeof(added_token_cases) / sizeof(added_token_cases[0]); + token++) { + const char *expected = added_token_cases[token].text; + size_t expected_bytes = strlen(expected); + int32_t encoded = -1; + size_t encoded_count = 0; + + REQUIRE(coli_edge_tokenize(edge, expected, expected_bytes, + &encoded, 1, &encoded_count, + error, sizeof(error)) == 0, + "Qwen3.6 added token encode failed"); + REQUIRE(encoded_count == 1 && + encoded == added_token_cases[token].id, + "Qwen3.6 added token encoded to the wrong ID"); + + size_t decoded_bytes = 0; + REQUIRE(coli_edge_detokenize(edge, &encoded, 1, + NULL, 0, &decoded_bytes, + error, sizeof(error)) == 0, + "Qwen3.6 added token decode sizing failed"); + REQUIRE(decoded_bytes == expected_bytes, + "Qwen3.6 added token decoded to the wrong size"); + + char decoded[64]; + REQUIRE(decoded_bytes + 1u <= sizeof(decoded), + "Qwen3.6 added token exceeds test buffer"); + REQUIRE(coli_edge_detokenize(edge, &encoded, 1, + decoded, sizeof(decoded), + &decoded_bytes, + error, sizeof(error)) == 0, + "Qwen3.6 added token decode failed"); + REQUIRE(decoded_bytes == expected_bytes && + !memcmp(decoded, expected, expected_bytes), + "Qwen3.6 added token round-trip differs"); + } + } + if (!strcmp(family, "qwen38")) { int32_t invalid_ids[] = {-1, (int32_t)edge_cap.vocab_size}; for (size_t invalid = 0; diff --git a/c/tests/test_engine_close_drain.py b/c/tests/test_engine_close_drain.py new file mode 100644 index 000000000..b45022761 --- /dev/null +++ b/c/tests/test_engine_close_drain.py @@ -0,0 +1,75 @@ +"""Engine.close drain contract: stdin EOF is the engine's graceful exit. + +``Engine.close`` must give the engine its one portable graceful path — the +serve loop reads requests from stdin, so closing that pipe lets the process +run its ``atexit`` teardown (``qt_shutdown`` writes HEAT_FILE) and exit 0. +Only a process that outlives the drain window falls through to the existing +terminate/kill ladder. + +The tests drive ``close()`` against real subprocesses of ``python -c``: +one that blocks on ``stdin.read()`` (the well-behaved engine) and one that +ignores EOF and sleeps (the hung engine). No model, GPU, or engine build is +involved — the contract under test is the *shutdown handshake*. +""" + +import os +import subprocess +import sys +import threading +import time +import unittest + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +import openai_server # noqa: E402 + +EOF_ENGINE = "import sys\nsys.stdin.read()\n" # blocks until EOF, exits 0 +HANG_ENGINE = "import time\ntime.sleep(120)\n" # ignores stdin entirely + + +def bare_engine(code): + """An Engine with only what close() touches, around a real subprocess.""" + engine = object.__new__(openai_server.Engine) + engine.closed = False + engine.pending_lock = threading.Lock() + engine._fail_pending = lambda *args, **kwargs: None + engine.dispatcher = threading.current_thread() # skip the join branch + engine.process = subprocess.Popen( + [sys.executable, "-c", code], stdin=subprocess.PIPE) + return engine + + +class EngineCloseDrain(unittest.TestCase): + def tearDown(self): + saved = openai_server._ENGINE_DRAIN_S + self.addCleanup(lambda: setattr(openai_server, "_ENGINE_DRAIN_S", saved)) + + def test_eof_exit_is_graceful(self): + openai_server._ENGINE_DRAIN_S = 10.0 + engine = bare_engine(EOF_ENGINE) + self.addCleanup(engine.process.kill) + engine.process.stdin.write(b"x") # a buffered request byte, like a turn + engine.process.stdin.flush() + started = time.perf_counter() + engine.close() + elapsed = time.perf_counter() - started + self.assertEqual(engine.process.returncode, 0, + "engine should exit 0 on stdin EOF, not by signal") + self.assertLess(elapsed, openai_server._ENGINE_DRAIN_S, + "graceful exit must not consume the drain window") + + def test_hung_engine_falls_back_to_hard_stop(self): + openai_server._ENGINE_DRAIN_S = 0.5 + engine = bare_engine(HANG_ENGINE) + self.addCleanup(engine.process.kill) + started = time.perf_counter() + engine.close() + elapsed = time.perf_counter() - started + self.assertIsNotNone(engine.process.returncode, + "fallback ladder must still stop the process") + self.assertNotEqual(engine.process.returncode, 0, + "a killed engine must not look like a clean exit") + self.assertLess(elapsed, 20, "fallback should be prompt") + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_engine_evidence.py b/c/tests/test_engine_evidence.py new file mode 100644 index 000000000..c447d9f30 --- /dev/null +++ b/c/tests/test_engine_evidence.py @@ -0,0 +1,384 @@ +"""tools/engine_evidence.py must parse only exact, in-range preamble text. + +Pins the exact banner/loaded text the module accepts, the field ranges it +enforces on both sides of each bound, and the None-vs-raise split in +parse_engine_preamble, so a future edit to the shared parser cannot +silently loosen or break any of its numeric bounds or its exact-text +matching without a local, fast failure. +""" +import unittest + +from tools.engine_evidence import ( + IDOT_KERNELS, + PreambleError, + parse_engine_banner, + parse_engine_loaded, + parse_engine_preamble, +) + +_BANNER = ( + "== GLM C engine (glm_moe_dsa), cache=8 experts/layer | " + "compute experts@4-bit dense@8-bit | idot: avx2 ==" +) +_LOADED = ( + "loaded in 12.34s | resident dense: 5678.90 MB | layers=32 experts=128 " + "| MTP ACTIVE (draft=4)" +) + + +def _banner(**subs): + text = _BANNER + for old, new in subs.items(): + assert old in text, old + text = text.replace(old, new, 1) + return text + + +def _loaded(**subs): + text = _LOADED + for old, new in subs.items(): + assert old in text, old + text = text.replace(old, new, 1) + return text + + +class ParseEngineBannerTest(unittest.TestCase): + def test_exact_banner_returns_typed_fields(self): + fields = parse_engine_banner(_BANNER) + self.assertEqual(fields, { + "kind": "BANNER", "cap": 8, "expert_bits": 4, "dense_bits": 8, + "kernel": "avx2", + }) + + def test_non_string_raises(self): + with self.assertRaises(PreambleError): + parse_engine_banner(None) + + def test_unrecognized_text_raises(self): + with self.assertRaises(PreambleError): + parse_engine_banner("not a banner at all") + + def test_unknown_kernel_raises(self): + # Negative control for the roster test below: an unlisted kernel + # name (real ISA extension, not in IDOT_KERNELS) must be refused. + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"idot: avx2": "idot: sse4"})) + + def test_trailing_text_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_BANNER + " extra") + + def test_trailing_newline_rejected(self): + # fullmatch requires consuming the WHOLE string; $ is zero-width + # and matches just before a trailing "\n" too, so a weakening to + # .match() or .search() would let a newline-terminated banner + # through here even though fullmatch correctly refuses it. A + # caller doing `for line in f:` on real engine stdout hands over + # lines WITH their trailing newline, so this is the shape a real + # caller would actually feed in, not a synthetic corner case. + with self.assertRaises(PreambleError): + parse_engine_banner(_BANNER + "\n") + + # -- cap: [1, 2**31-1] -- + + def test_cap_lower_bound_accepted(self): + fields = parse_engine_banner(_banner(**{"cache=8": "cache=1"})) + self.assertEqual(fields["cap"], 1) + + def test_cap_lower_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"cache=8": "cache=0"})) + + def test_cap_upper_bound_accepted(self): + fields = parse_engine_banner( + _banner(**{"cache=8": "cache=2147483647"})) + self.assertEqual(fields["cap"], 2147483647) + + def test_cap_upper_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"cache=8": "cache=2147483648"})) + + # -- expert_bits: [1, 16] -- + + def test_expert_bits_lower_bound_accepted(self): + fields = parse_engine_banner(_banner(**{"experts@4-bit": "experts@1-bit"})) + self.assertEqual(fields["expert_bits"], 1) + + def test_expert_bits_lower_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"experts@4-bit": "experts@0-bit"})) + + def test_expert_bits_upper_bound_accepted(self): + fields = parse_engine_banner(_banner(**{"experts@4-bit": "experts@16-bit"})) + self.assertEqual(fields["expert_bits"], 16) + + def test_expert_bits_upper_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"experts@4-bit": "experts@17-bit"})) + + # -- dense_bits: [1, 16] -- + + def test_dense_bits_lower_bound_accepted(self): + fields = parse_engine_banner(_banner(**{"dense@8-bit": "dense@1-bit"})) + self.assertEqual(fields["dense_bits"], 1) + + def test_dense_bits_lower_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"dense@8-bit": "dense@0-bit"})) + + def test_dense_bits_upper_bound_accepted(self): + fields = parse_engine_banner(_banner(**{"dense@8-bit": "dense@16-bit"})) + self.assertEqual(fields["dense_bits"], 16) + + def test_dense_bits_upper_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"dense@8-bit": "dense@17-bit"})) + + # -- leading-zero handling (no field allows a leading zero on a + # multi-digit value; a leading zero makes the whole line unrecognized, + # not merely out of range) -- + + def test_leading_zero_digit_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"cache=8": "cache=007"})) + + def test_no_leading_zero_digit_accepted(self): + fields = parse_engine_banner(_banner(**{"cache=8": "cache=7"})) + self.assertEqual(fields["cap"], 7) + + def test_oversized_field_raises_preamble_error_not_bare_value_error(self): + # _UINT_TEXT has no digit-count cap, so a field beyond Python's + # int-string conversion limit (sys.int_info.default_max_str_digits, + # 4300 by default) used to reach int()/float() raw, escaping as a + # bare ValueError instead of this module's own PreambleError -- + # breaking every caller's contract to refuse with a named error. + huge = "1" + "0" * 4300 + with self.assertRaises(PreambleError): + parse_engine_banner(_banner(**{"cache=8": f"cache={huge}"})) + + +class IdotKernelRosterTest(unittest.TestCase): + """Pin every IDOT_KERNELS entry from the engine's own source, not just + avx2 (the only entry any test previously referenced -- confirmed by + RAN mutation: reducing IDOT_KERNELS to ("avx2",) alone left this + module's OWN 45-test suite fully green, since nothing here iterated + or symbolically referenced the roster). Removing any single entry + below, or the roster itself down to fewer entries, must fail this + test -- proof-of-bite for each is in the worker report. + + Source of the 7 entries, cited by file:line: c/quant.h:582-594's + #if/#elif ladder defines IDOT_KERNEL to each of these string + literals in turn, and c/colibri.c:11243 prints it unmodified inside + the exact banner format this fixture reproduces character-for- + character (cache=%d experts/layer | compute experts@%d-bit + dense@%d-bit | idot: ). + + c/quant.h:582 "avx512-vnni" (__AVX512VNNI__ && __AVX512BW__) + c/quant.h:584 "avx-vnni" (__AVXVNNI__ && __AVX2__) + c/quant.h:586 "avx2" (__AVX2__) + c/quant.h:588 "neon-i8mm" (__ARM_NEON && __ARM_FEATURE_MATMUL_INT8) + c/quant.h:590 "neon" (__ARM_NEON, no i8mm) -- RAN + c/quant.h:592 "vsx" (__VSX__) + c/quant.h:594 "scalar" (else) + + "neon" is marked RAN: this development host is arm64 Darwin, and + `echo | cc -dM -E -` (no -mcpu, matching this project's own Makefile + comment "ARCH unset -> no -mcpu, default build byte-identical") + defines __ARM_NEON but NOT __ARM_FEATURE_MATMUL_INT8 on this + machine, so its own default (unaccelerated) build takes the "neon" + branch of the ladder above -- confirmed by running that exact + compiler invocation, not by reading the source alone. This is a + compiler-preprocessor probe of the flags the default build actually + uses, not a captured engine stdout banner: no engine binary was + built or run for this (no model runs, per the dispatch's hard + bound). The other six entries are taken from the C source only + (INFERRED from c/quant.h's literals, not independently reproduced on + real hardware for each ISA). + + VACUOUS-GATE NOTE: test_every_roster_entry_parses_from_real_banner_text + below iterates IDOT_KERNELS and checks each entry parses to itself -- + but _BANNER_RE is ITSELF built from IDOT_KERNELS + ("|".join(IDOT_KERNELS)), so a renamed entry (same count, different + spelling) produces a banner the correspondingly-mutated regex still + matches: the assertion compares the mutation against itself and can + pass having checked nothing about the entry's real spelling. Only + test_roster_matches_frozen_expectation below, which compares against + an independent literal transcribed from c/quant.h rather than against + IDOT_KERNELS itself, actually defends the CONTENT of each entry; the + per-entry parse loop is kept because it still defends something real + (that parse_engine_banner's kernel-dispatch group and IDOT_KERNELS + stay in sync with each other), just not entry spelling on its own. + """ + + def test_every_roster_entry_parses_from_real_banner_text(self): + for kernel in IDOT_KERNELS: + with self.subTest(kernel=kernel): + fields = parse_engine_banner( + _banner(**{"idot: avx2": f"idot: {kernel}"})) + self.assertEqual(fields["kernel"], kernel) + + def test_roster_matches_frozen_expectation(self): + # Frozen from c/quant.h:582-594's #if/#elif ladder. If the + # engine's ladder changes, this literal and the roster both + # change, deliberately and together. Unlike the per-entry parse + # loop above, this does NOT derive its expectation from + # IDOT_KERNELS or from _BANNER_RE (which is itself built from + # IDOT_KERNELS) -- it is the independent source transcription + # that makes a same-length rename of any entry (which the parse + # loop and the count/uniqueness check below both miss) fail. + EXPECTED_IDOT_KERNELS = ("avx512-vnni", "avx-vnni", "avx2", + "neon-i8mm", "neon", "vsx", "scalar") + self.assertEqual(IDOT_KERNELS, EXPECTED_IDOT_KERNELS) + + def test_roster_is_not_accidentally_empty_or_singleton(self): + # A cheap sanity backstop for the roster itself, independent of + # any one entry's own test above. Redundant with the frozen- + # literal test for a length change, but kept because it is a + # different, cheaper check that would survive even if the + # literal above ever needed updating for a real ladder change. + self.assertEqual(len(IDOT_KERNELS), 7) + self.assertEqual(len(set(IDOT_KERNELS)), 7) + + +class ParseEngineLoadedTest(unittest.TestCase): + def test_exact_loaded_returns_typed_fields(self): + fields = parse_engine_loaded(_LOADED) + self.assertEqual(fields, { + "kind": "LOADED", "load_s": 12.34, "resident_mb": 5678.90, + "layers": 32, "experts": 128, "mtp": "ACTIVE", "draft": 4, + }) + + def test_non_string_raises(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(1234) + + def test_unrecognized_text_raises(self): + with self.assertRaises(PreambleError): + parse_engine_loaded("not a load record") + + def test_trailing_newline_rejected(self): + # See ParseEngineBannerTest.test_trailing_newline_rejected: same + # fullmatch-vs-$ subtlety, same real-caller shape (stdout lines + # iterated with their newline still attached). + with self.assertRaises(PreambleError): + parse_engine_loaded(_LOADED + "\n") + + # -- layers: [1, 128] -- + + def test_layers_lower_bound_accepted(self): + fields = parse_engine_loaded(_loaded(**{"layers=32": "layers=1"})) + self.assertEqual(fields["layers"], 1) + + def test_layers_lower_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"layers=32": "layers=0"})) + + def test_layers_upper_bound_accepted(self): + fields = parse_engine_loaded(_loaded(**{"layers=32": "layers=128"})) + self.assertEqual(fields["layers"], 128) + + def test_layers_upper_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"layers=32": "layers=129"})) + + # -- experts: [1, 4096] -- + + def test_experts_lower_bound_accepted(self): + fields = parse_engine_loaded(_loaded(**{"experts=128": "experts=1"})) + self.assertEqual(fields["experts"], 1) + + def test_experts_lower_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"experts=128": "experts=0"})) + + def test_experts_upper_bound_accepted(self): + fields = parse_engine_loaded(_loaded(**{"experts=128": "experts=4096"})) + self.assertEqual(fields["experts"], 4096) + + def test_experts_upper_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"experts=128": "experts=4097"})) + + # -- exactly two decimal digits on load_s / resident_mb -- + + def test_two_decimal_places_accepted(self): + fields = parse_engine_loaded(_LOADED) + self.assertEqual(fields["load_s"], 12.34) + + def test_one_decimal_place_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"12.34s": "12.3s"})) + + def test_three_decimal_places_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"12.34s": "12.345s"})) + + # -- MTP / draft interaction -- + + def test_absent_mtp_allows_nonzero_draft(self): + fields = parse_engine_loaded( + _loaded(**{"MTP ACTIVE (draft=4)": "MTP absent (draft=5)"})) + self.assertEqual(fields["mtp"], "absent") + self.assertEqual(fields["draft"], 5) + + def test_active_mtp_allows_nonzero_draft(self): + fields = parse_engine_loaded(_LOADED) + self.assertEqual(fields["mtp"], "ACTIVE") + self.assertEqual(fields["draft"], 4) + + def test_draft_upper_bound_accepted(self): + fields = parse_engine_loaded(_loaded(**{"draft=4)": "draft=63)"})) + self.assertEqual(fields["draft"], 63) + + def test_draft_upper_bound_rejected(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"draft=4)": "draft=64)"})) + + def test_disabled_multiplexed_requires_zero_draft(self): + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded( + **{"MTP ACTIVE (draft=4)": + "MTP DISABLED (multiplexed serve) (draft=4)"})) + + def test_disabled_multiplexed_with_zero_draft_parses(self): + fields = parse_engine_loaded(_loaded( + **{"MTP ACTIVE (draft=4)": + "MTP DISABLED (multiplexed serve) (draft=0)"})) + self.assertEqual(fields["mtp"], "DISABLED (multiplexed serve)") + self.assertEqual(fields["draft"], 0) + + def test_oversized_field_raises_preamble_error_not_bare_value_error(self): + # See ParseEngineBannerTest's identical test: same digit-count + # cap gap, same fix, this function's own layers field. + huge = "1" + "0" * 4300 + with self.assertRaises(PreambleError): + parse_engine_loaded(_loaded(**{"layers=32": f"layers={huge}"})) + + +class ParseEnginePreambleTest(unittest.TestCase): + def test_dispatches_to_banner(self): + self.assertEqual( + parse_engine_preamble(_BANNER), parse_engine_banner(_BANNER)) + + def test_dispatches_to_loaded(self): + self.assertEqual( + parse_engine_preamble(_LOADED), parse_engine_loaded(_LOADED)) + + def test_unowned_line_returns_none(self): + self.assertIsNone(parse_engine_preamble("some ordinary log line")) + + def test_banner_prefixed_but_malformed_still_raises(self): + with self.assertRaises(PreambleError): + parse_engine_preamble("== GLM C engine but garbled ==") + + def test_loaded_prefixed_but_malformed_still_raises(self): + with self.assertRaises(PreambleError): + parse_engine_preamble("loaded in not a valid record") + + def test_non_string_raises(self): + with self.assertRaises(PreambleError): + parse_engine_preamble(3.14) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_eval_glm.py b/c/tests/test_eval_glm.py new file mode 100644 index 000000000..49f327a08 --- /dev/null +++ b/c/tests/test_eval_glm.py @@ -0,0 +1,1082 @@ +"""tools/eval_glm.py must accept only the engine's real SCORE stdout +records and refuse everything else with a named error: an unparsed or +duplicated banner/load preamble, a SCORE line whose numeric token is not +the canonical finite ``%.6f``/``%.17g`` spelling the engine actually +emits, and a foreign stdout line that is neither a preamble nor a SCORE +record. It must also write result rows incrementally (one flush per +request, never buffered until completion) and mark incomplete runs, +including a pre-launch refusal, before the engine is ever started. + +Checks enumerated from the source (`tools/eval_glm.py`, read in full +before writing this module) and covered below, grouped by the function +that performs them: + +- `parse_c17g`: both exact finite spellings the engine actually emits -- + the shipped ``%.6f`` form and the newer ``%.17g`` evidence form -- and + nothing else; this module imports nothing from and shares no code with + `check_data_logprob_gaps.py`, which parses its own independent grammar. +- `parse_score_result` / `_SCORE_RE`: the shipped ``%.6f`` form AND the + ``%.17g`` evidence form (both numeric forms), non-finite/malformed + rejection, and the `` `` metadata bounds. +- `classify_score_stdout` / `ScoreStdoutClassifier`: the banner/load + preamble lifecycle (missing, duplicated, out-of-order, or a SCORE + record before the load record all refuse); every line that is not an + exact banner, an exact load record, or an exact SCORE record refuses + by name -- a foreign line is never silently treated as a score, which + is exactly the defect dev's plain ``line[0] in "-0123456789"`` filter + does not catch (differential bite, below). +- `score_request_wire`: strict ASCII/LF request grammar, the per-record + SHA-256 digest, and the inclusive 256 MiB engine text limit. +- `completion_error`: the exact zero-exit/complete-count/positive-token + denominator that alone passes. +- `main`: incremental durability (one written+flushed row per completed + request, never buffered until the run ends) and pre-launch INCOMPLETE + marking (no benchmark tasks selected; zero SCORE requests produced; + every choice's context/continuation split is empty) -- the engine is + never launched for any of these; a mid-run crash or interrupt still + leaves the INCOMPLETE marker and terminates the child process; a + partial run still prints the accuracy table over whatever rows landed. + +Deferred (need a live binary this module does not have access to): +- `test_c_emitted_c17g_corpus_is_canonical`, which drives + ``test_logprob_wire --score-c17g-fixture`` (a binary produced by a + different part of this project's build, not present here). + +No model is run by the committed tests -- every case here drives +`eval_glm.py` against an injected stand-in for the direct engine launch +(a fake ``subprocess.Popen`` returning canned stdout/stderr), never a +real ``./glm`` process. +""" + +import contextlib +import hashlib +import importlib.util +import io +import json +import os +import pathlib +import signal +import sys +import tempfile +import types +import unittest +from unittest import mock + + +HERE = pathlib.Path(__file__).resolve().parent +TOOLS = HERE.parent / "tools" + +_spec = importlib.util.spec_from_file_location( + "eval_glm_under_test", TOOLS / "eval_glm.py") +EVAL = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(EVAL) + + +class EvalGlmEvidenceTests(unittest.TestCase): + BANNER = ( + "== GLM C engine (glm_moe_dsa), cache=64 experts/layer | " + "compute experts@4-bit dense@8-bit | idot: neon-i8mm ==") + + @staticmethod + def loaded(state="ACTIVE", draft=1, layers=78, experts=256, + load="1.00", resident="1.00"): + return (f"loaded in {load}s | resident dense: {resident} MB | " + f"layers={layers} experts={experts} | MTP {state} " + f"(draft={draft})") + + def run_eval_main(self, stdout_records): + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + process = types.SimpleNamespace( + returncode=0, stderr=(), stdout=tuple(stdout_records), + wait=lambda: 0, poll=lambda: 0, terminate=lambda: None) + with tempfile.TemporaryDirectory() as tmp: + output = pathlib.Path(tmp) / "results.csv" + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size":3}\n') + argv = [ + "eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--glm", "/fake/glm", "--out", + str(output), + ] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + stderr_buf = io.StringIO() + stdout_buf = io.StringIO() + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object(EVAL.subprocess, "Popen", + return_value=process) as popen, \ + contextlib.redirect_stderr(stderr_buf), \ + contextlib.redirect_stdout(stdout_buf): + rc = EVAL.main() + self.last_popen_kwargs = popen.call_args.kwargs if popen.called else {} + self.last_stderr = stderr_buf.getvalue() + self.last_stdout = stdout_buf.getvalue() + return rc, output.read_text(), popen.call_count + + def test_exact_score_text_survives_csv(self): + text = "-8.5534581234567888" + exact, value, contlen, greedy = EVAL.parse_score_result( + f"{text} 4096 1") + self.assertEqual(exact, text) + self.assertEqual(contlen, 4096) + self.assertEqual(greedy, 1) + self.assertTrue(value < 0) + + out = io.StringIO() + meta = ("task", 3, 2, 4096, 17, 2) + EVAL.write_result_row(out, 9, meta, exact, greedy) + self.assertEqual( + out.getvalue(), + "9,task,3,2,4096,17,2,-8.5534581234567888,1\n") + + def test_shipped_dot6f_form_is_still_accepted(self): + # The tool must accept BOTH the shipped %.6f form the engine + # actually prints today AND the %.17g evidence form. + exact, value, contlen, greedy = EVAL.parse_score_result( + "-8.553458 4096 1") + self.assertEqual(exact, "-8.553458") + self.assertEqual(value, -8.553458) + self.assertEqual((contlen, greedy), (4096, 1)) + + def test_direct_call_rejects_trailing_newline_record(self): + # See parse_score_result's own docstring: _SCORE_RE's "$" anchor + # matches just before a trailing "\n" as well as at true + # end-of-string, so fullmatch -> match/search would accept a + # newline-terminated record fullmatch correctly rejects. The + # classifier's own callers strip the newline first and are + # provably out of reach of this (see the docstring), but this + # function is also called directly -- pin the direct-call + # contract so a future weakening doesn't slip in unnoticed there. + with self.assertRaises(EVAL.EvidenceError): + EVAL.parse_score_result("-8.553458 4096 1\n") + + def test_parse_c17g_rejects_trailing_garbage(self): + # parse_c17g is only ever reached, elsewhere in this module, + # through _SCORE_RE's own "^...$"-anchored capture group, which + # already excludes trailing garbage before parse_c17g ever sees + # the text -- so nothing else in this module's test suite drives + # parse_c17g directly. Direct coverage: verified (RAN, against a + # scratch mutant copy, not committed) that weakening the + # fullmatch to match here is actually an EQUIVALENT mutant for + # every input tried -- the round-trip format comparison a few + # lines below the match call independently rejects any text with + # a non-canonical tail, since format(value, ...) can never + # reproduce trailing garbage. This test pins parse_c17g's own + # contract directly rather than only through callers; it does + # not by itself prove the fullmatch call is load-bearing. + with self.assertRaises(EVAL.EvidenceError): + EVAL.parse_c17g("1.5garbage") + with self.assertRaises(EVAL.EvidenceError): + EVAL.parse_c17g("-8.553458 ") # trailing space after a valid token + self.assertEqual(EVAL.parse_c17g("1.5"), 1.5) + + def test_parse_c17g_rejects_bare_trailing_decimal_point(self): + # "1." is valid Python float() syntax (float("1.") == 1.0), so + # only the round-trip format comparison -- not the decimal + # group's "one-or-more" digit requirement -- is what actually + # rejects it: format(1.0, ".17g") == "1", never "1.". Verified + # (RAN, scratch mutant, not committed) that widening that group + # to "zero-or-more" does not change parse_c17g's accept/reject + # outcome for any of these inputs; kept as direct-coverage + # regression pins, not as proof the digit requirement is + # independently load-bearing. + for bad in ("1.", "-0.", "12."): + with self.subTest(bad=bad): + with self.assertRaises(EVAL.EvidenceError): + EVAL.parse_c17g(bad) + + def test_parse_c17g_rejects_single_digit_exponent(self): + # C's %e/%g exponent is zero-padded to at least two digits + # ("e-05", never "e-5"). Verified (RAN, scratch mutant, not + # committed) that widening the exponent's trailing-digit count + # from {1,2} to {0,2} does NOT change parse_c17g's outcome here + # either, for the same reason as the two tests above: no + # canonical %.6f/%.17g spelling ever produces a single-digit + # exponent, so the round-trip check rejects it independently of + # this group's width. Use a magnitude small enough that %.17g + # actually chooses exponential notation (a mid-size value like + # 1e5 round-trips through plain "100000" instead and would not + # even reach the exponent group). + text = format(-1.0000000000000001e-05, ".17g") + self.assertEqual(text, "-1.0000000000000001e-05") + self.assertEqual(EVAL.parse_c17g(text), -1.0000000000000001e-05) + with self.assertRaises(EVAL.EvidenceError): + EVAL.parse_c17g(text.replace("e-05", "e-5")) + + def test_parse_engine_loaded_rejects_overflowing_metrics(self): + # load_s/resident_mb have no magnitude cap in _FIXED2_TEXT (only + # a fixed 2-decimal-digit suffix), so a long enough digit run + # converts via float() to +inf; unlike parse_c17g there is no + # round-trip format comparison to catch that independently here, + # so the isfinite check is this function's only defense against + # it -- a removed/weakened isfinite check would silently accept + # an infinite load time or resident size. + huge = "1" + "0" * 350 + ".00" + for field in ("load_s", "resident_mb"): + with self.subTest(field=field): + kwargs = {"load": huge} if field == "load_s" else {"resident": huge} + line = self.loaded(**kwargs) + with self.assertRaisesRegex( + EVAL.PreambleError, "must be finite"): + EVAL.parse_engine_loaded(line) + + def test_score_request_digest_binds_strict_ascii_bytes_including_lf(self): + requests = ("1 1 1 2", "2 1 0 1 2") + lines, payload, digests = EVAL.score_request_wire(requests, 3) + self.assertEqual(lines, ( + b"1 1 1 2\n", b"2 1 0 1 2\n")) + self.assertEqual(payload, b"".join(lines)) + self.assertEqual( + digests, + tuple(hashlib.sha256(line).hexdigest() for line in lines)) + self.assertNotEqual( + digests[1], + hashlib.sha256(b"2 1 0 1 2").hexdigest()) + bad_requests = ( + "", "two\nlines", "cr\rline", "1 1", "1 1 0", + "1 1 0 1 2", "1 1 0 1 junk", "1 1 0 3", + "1 1 -1 1", "1 1 00 1", "2147483648 1 0 1", + "1 2147483647 0 1", "evidence-μ", + ) + for bad in bad_requests: + with self.subTest(bad=bad): + with self.assertRaises(EVAL.EvidenceError): + EVAL.score_request_wire((bad,), 3) + with self.assertRaises(EVAL.EvidenceError): + EVAL.score_request_wire((), 3) + + def test_nonfinite_and_malformed_scores_refuse(self): + bad = [ + "nan 1 1", "inf 1 1", "-inf 1 1", "-1 1", "-1 1 1 extra", + "not-a-number 1 1", "-1 -1 1", "-1 0 1", "-1 1 2", "1 1 1", + "-1 1 1", "-1\t1 1", "-1 01 1", "-1_0 1 1", + "-1 2147483648 1", "+0 1 0", "-01 1 0", + "-0_125 1 0", "-١ 1 0", "-1e-9999 1 0", + "-1.00000000000000000 1 0", "-1e-9 1 0", + "-1e--09 1 0", "-1e+009 1 0", "-1.25 1 0 junk", + ] + for line in bad: + with self.subTest(line=line): + with self.assertRaises(EVAL.EvidenceError): + EVAL.parse_score_result(line) + + def test_stdout_grammar_refuses_unknown_records(self): + self.assertIsNone(EVAL.classify_score_stdout(self.BANNER + "\n")) + for state, draft in (("ACTIVE", 0), ("ACTIVE", 1), + ("absent", 0), ("absent", 2), + ("DISABLED (multiplexed serve)", 0)): + with self.subTest(state=state, draft=draft): + self.assertIsNone(EVAL.classify_score_stdout( + self.loaded(state, draft) + "\n")) + exact, _, _, _ = EVAL.classify_score_stdout("-1.25 1 0\n") + self.assertEqual(exact, "-1.25") + bad = ( + "\n", "unexpected banner\n", "nan 1 0\n", "inf 1 0\n", + " -1.25 1 0\n", "-1.25 1 0 \n", "-1.25 1 0\t\n", + "-1.25 1 0", "-1.25 1 0\r\n", + "PROF 0.001 1 1 0.000 0.000 0.000 0.000 0.000 1\n", + "DONE 7 STAT 1 1.00 0.0 1.00 1 0\n", + "== GLM C engine fabricated ==\n", + "== GLM C engine (glm_moe_dsa), cache=0 experts/layer | " + "compute experts@4-bit dense@8-bit | idot: neon ==\n", + "== GLM C engine (glm_moe_dsa), cache=064 experts/layer | " + "compute experts@4-bit dense@8-bit | idot: neon ==\n", + "loaded in 1.0s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP ACTIVE (draft=1)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP DISABLED (multiplexed serve) (draft=1)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP unknown (draft=0)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP ACTIVE (draft=64)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=0 experts=256 | " + "MTP ACTIVE (draft=1)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=129 experts=256 | " + "MTP ACTIVE (draft=1)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=78 experts=0 | " + "MTP ACTIVE (draft=1)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=78 experts=4097 | " + "MTP ACTIVE (draft=1)\n", + "loaded in -0.01s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP ACTIVE (draft=1)\n", + "loaded in nan s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP ACTIVE (draft=1)\n", + "loaded in 1.0s | resident dense: 1.00 MB | layers=78 experts=256 | " + "MTP ACTIVE (draft=1)\n", + "loaded in 1.00s | resident dense: 1.00 MB | layers=2147483648 experts=256 | " + "MTP ACTIVE (draft=1)\n", + " == GLM C engine (glm_moe_dsa), cache=64 experts/layer | " + "compute experts@4-bit dense@8-bit | idot: neon ==\n", + ) + for line in bad: + with self.subTest(line=line): + with self.assertRaises(EVAL.EvidenceError): + EVAL.classify_score_stdout(line) + + def test_loaded_prefix_collision_is_refused_not_passed_through(self): + # engine_evidence.parse_engine_preamble dispatches by literal prefix + # (line.startswith("loaded in")), not by a semantic match against + # the load record's shape. "loaded index ..." shares that 9-char + # prefix purely because "index" itself starts with "in", so it is + # routed into parse_engine_loaded and refused there, rather than + # returned as an ordinary unowned log line. Confirmed at primary + # source (c/colibri.c on origin/dev): no engine anywhere in this + # tree ever prints a line starting with "loaded index" -- the only + # stdout "loaded" record any engine emits is the exact "loaded in + # ...s | resident dense: ..." line this module already pins above. + # This is deliberately fail-loud dispatch behavior, not a bug -- + # see engine_evidence.parse_engine_preamble's own docstring + # (fixed in this cycle to document the prefix mechanism and this + # exact collision explicitly, since no engine anywhere in this + # tree emits "loaded index" today). This test pins eval_glm's + # own consumer-visible behavior for the same collision. + with self.assertRaises(EVAL.EvidenceError): + EVAL.classify_score_stdout("loaded index 5 whatever\n") + + def test_oversized_numeric_field_refuses_as_evidence_error(self): + # engine_evidence's own int()/float() calls now catch this at + # the root (a numeric preamble field beyond the interpreter's + # 4300-digit int-string conversion limit raises engine_evidence's + # own PreambleError, not a bare ValueError) -- see + # test_engine_evidence.py's identical-purpose tests on + # parse_engine_banner/parse_engine_loaded directly. This + # module's three call sites still catch ValueError alongside + # PreambleError too, as defense in depth for any other + # ValueError the shared module might someday raise; this test + # pins the end-to-end result through eval_glm's own consumers. + big = "1" + "0" * 4300 + banner = ( + f"== GLM C engine (glm_moe_dsa), cache={big} experts/layer | " + "compute experts@4-bit dense@8-bit | idot: avx2 ==\n") + with self.assertRaises(EVAL.EvidenceError): + EVAL.classify_score_stdout(banner) + parser = EVAL.ScoreStdoutClassifier() + with self.assertRaises(EVAL.EvidenceError): + parser.classify(banner) + # is_score_preamble is a query function (bool return); it must + # swallow the malformed-field case as "not a preamble" rather + # than raise at all. + self.assertFalse(EVAL.is_score_preamble(banner.rstrip("\n"))) + + def test_oversized_vocab_json_refuses_as_evidence_error(self): + # Same error-contract class as F3, in a different function: + # json.loads() raises a bare ValueError (not its + # json.JSONDecodeError subclass) for an integer literal beyond + # Python's 4300-digit int-string conversion limit, which + # score_snapshot_vocab's except clause did not catch -- the + # bare ValueError escaped main()'s only try/except around this + # call (which catches EvidenceError to write the INCOMPLETE + # marker and return before ever launching the engine), so + # main() crashed outright with NO INCOMPLETE marker written at + # all, worse than the ordinary prelaunch-refusal contract every + # other pre-engine-launch failure gets. + huge_vocab = "1" + "0" * 4300 + with tempfile.TemporaryDirectory() as tmp: + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size": ' + huge_vocab + '}\n') + with self.assertRaises(EVAL.EvidenceError): + EVAL.score_snapshot_vocab(tmp) + + def test_oversized_vocab_json_marks_incomplete_end_to_end(self): + # The end-to-end contract test_oversized_vocab_json_refuses_ + # as_evidence_error's unit-level pin implies: main() must reach + # prelaunch_incomplete() (INCOMPLETE marker written, engine + # never launched, exit 1) rather than crash uncaught. + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + huge_vocab = "1" + "0" * 4300 + with tempfile.TemporaryDirectory() as tmp: + output = pathlib.Path(tmp) / "results.csv" + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size": ' + huge_vocab + '}\n') + argv = [ + "eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--glm", "/fake/glm", "--out", + str(output), + ] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object( + EVAL.subprocess, "Popen", + side_effect=AssertionError("engine launched")): + rc = EVAL.main() + self.assertEqual(rc, 1) + self.assertIn("# INCOMPLETE:", output.read_text()) + + def test_banner_kernels_and_load_boundaries_are_exact(self): + self.assertEqual( + EVAL.parse_engine_banner(self.BANNER)["kernel"], "neon-i8mm") + for kernel in ("avx512-vnni", "avx-vnni", "avx2", "neon-i8mm", + "neon", "vsx", "scalar"): + line = self.BANNER.replace("neon-i8mm", kernel) + self.assertEqual(EVAL.parse_engine_banner(line)["kernel"], kernel) + with self.assertRaises(EVAL.PreambleError): + EVAL.parse_engine_banner(self.BANNER.replace("neon-i8mm", "fabricated")) + + for layers, experts in ((1, 1), (128, 4096)): + parsed = EVAL.parse_engine_loaded(self.loaded( + layers=layers, experts=experts)) + self.assertEqual((parsed["layers"], parsed["experts"]), + (layers, experts)) + + def test_score_stream_owns_banner_load_lifecycle(self): + parser = EVAL.ScoreStdoutClassifier() + self.assertIsNone(parser.classify(self.BANNER + "\n")) + self.assertIsNone(parser.classify(self.loaded("absent", 2) + "\n")) + exact, _, _, _ = parser.classify("-1.25 1 0\n") + self.assertEqual(exact, "-1.25") + parser.finish() + + cases = ( + [self.loaded() + "\n", self.BANNER + "\n"], + [self.BANNER + "\n", "-1.25 1 0\n"], + [self.BANNER + "\n", self.BANNER + "\n"], + [self.BANNER + "\n", self.loaded() + "\n", + self.loaded() + "\n"], + [self.BANNER + "\n"], + ) + for records in cases: + with self.subTest(records=records): + parser = EVAL.ScoreStdoutClassifier() + with self.assertRaises(EVAL.EvidenceError): + for record in records: + parser.classify(record) + parser.finish() + + def test_eval_main_uses_stateful_score_stream(self): + rc, output, launches = self.run_eval_main(( + self.BANNER + "\n", + self.loaded("absent", 2) + "\n", + "-1 2 1\n", "-2 2 0\n", "-3 2 0\n", + )) + self.assertEqual(rc, 0) + self.assertEqual(launches, 1) + self.assertIn("# finished: 3/3", output) + + def test_result_rows_are_written_and_flushed_incrementally(self): + # A run interrupted mid-task must leave a valid partial file -- + # rows are written and flushed per-request, never buffered until + # the run completes. Simulate a mid-run crash by having the fake + # engine's stdout iterator raise after the first scored record; + # the CSV must already contain that row. + + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + class CrashingStdout: + def __init__(self, lines): + self._lines = list(lines) + + def __iter__(self): + for index, line in enumerate(self._lines): + yield line + if index == 2: # after banner+load+one SCORE row + raise OSError("engine died mid-run") + + stdout_lines = [ + self.BANNER + "\n", self.loaded("absent", 2) + "\n", + "-1 2 1\n", + "-2 2 0\n", + ] + process = types.SimpleNamespace( + returncode=1, stderr=(), stdout=CrashingStdout(stdout_lines), + wait=lambda: 1, poll=lambda: 1, terminate=lambda: None) + with tempfile.TemporaryDirectory() as tmp: + output = pathlib.Path(tmp) / "results.csv" + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size":3}\n') + argv = [ + "eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--glm", "/fake/glm", "--out", + str(output), + ] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object(EVAL.subprocess, "Popen", + return_value=process), \ + self.assertRaises(OSError): + EVAL.main() + text = output.read_text() + self.assertIn(",-1,1\n", text.replace(".000000", "")) + self.assertNotIn("# finished:", text) + # The crash-path marker itself must be present, not + # just the absence of "# finished:" -- a downstream consumer + # scans for this exact line to know the run never reached a + # complete denominator. + self.assertIn( + "# INCOMPLETE: evaluator terminated before a complete " + "denominator", text) + + def test_row_is_durable_on_disk_before_the_run_finishes(self): + # test_result_rows_are_written_and_flushed_incrementally (above) + # is named for the per-row flush but cannot actually detect + # losing it: that test reads the output file only AFTER + # EVAL.main() has unwound through its `finally` block, which + # closes out_f (and Python flushes a file on close) regardless + # of whether any PER-ROW flush ever ran. It defends the ARTIFACT + # after a clean-ish exit, not the PROPERTY of durability DURING + # the run. + # + # This test instead opens a SECOND, independent file handle on + # the same path while main() is still running (has not + # returned) and reads through it -- a second handle can only + # see bytes actually flushed to the OS by the writer, never + # bytes still sitting only in the writer's own buffer. The + # probe runs from inside the fake stdout generator, resumed + # only once the consumer (main()'s per-line loop body, including + # write_result_row + out_f.flush() for that line) has fully + # finished with the PREVIOUS line and is asking for the next one + # -- so a snapshot taken there reflects exactly what has been + # made durable so far, mid-run. + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + class ProbingStdout: + def __init__(self, lines, probe_path): + self._lines = list(lines) + self._probe_path = probe_path + self.snapshots = [] + + def __iter__(self): + for line in self._lines: + yield line + try: + with open(self._probe_path) as f: + self.snapshots.append(f.read()) + except FileNotFoundError: + self.snapshots.append("") + + with tempfile.TemporaryDirectory() as tmp: + output = pathlib.Path(tmp) / "results.csv" + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size":3}\n') + stdout = ProbingStdout(( + self.BANNER + "\n", + self.loaded("absent", 2) + "\n", + "-1 2 1\n", + "-2 2 0\n", + ), str(output)) + process = types.SimpleNamespace( + returncode=0, stderr=(), stdout=stdout, + wait=lambda: 0, poll=lambda: 0, terminate=lambda: None) + argv = [ + "eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--glm", "/fake/glm", "--out", + str(output), + ] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object(EVAL.subprocess, "Popen", + return_value=process): + rc = EVAL.main() + self.assertEqual(rc, 0) + # snapshots[0]/[1]: after the banner/load preamble lines, before + # any SCORE result -- no row yet. + self.assertNotIn(",-1,1\n", stdout.snapshots[0].replace(".000000", "")) + self.assertNotIn(",-1,1\n", stdout.snapshots[1].replace(".000000", "")) + # snapshots[2]: taken right after the FIRST SCORE line was fully + # processed but BEFORE the second one is even requested -- + # main() has not returned. The first row must already be + # readable through the independent handle; the second must not. + first_row_snapshot = stdout.snapshots[2].replace(".000000", "") + self.assertIn(",-1,1\n", first_row_snapshot, + "first row not durable before the run finished -- " + "the per-row flush is not doing its job") + self.assertNotIn(",-2,0\n", first_row_snapshot) + + def test_child_is_terminated_on_mid_run_interrupt_or_exception(self): + # SIGINT/SIGTERM/any exception mid-run must not leave the + # engine child running. The fake engine here never exits on its + # own -- .poll() keeps returning None (as a real child that + # ignores its stdin being closed would) until .terminate() is + # actually called -- so a passing test proves main() called + # terminate() itself rather than relying on the child to die. + request_raw = b"2 2 1 2 1 2\n" + request_digest = hashlib.sha256(request_raw).hexdigest() + + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + banner = self.BANNER + "\n" + loaded = self.loaded("absent", 2) + "\n" + + def interrupting_stdout(): + yield banner + yield loaded + yield f"SCORE 0 {request_digest} -1 2 1\n" + raise KeyboardInterrupt("operator pressed Ctrl+C") + + class NeverExitingProcess: + def __init__(self): + self.returncode = None + self.stderr = () + self.stdout = interrupting_stdout() + self.terminated = False + self.terminate_calls = 0 + self.wait_calls = 0 + + def poll(self): + return 0 if self.terminated else None + + def terminate(self): + self.terminated = True + self.terminate_calls += 1 + + def wait(self): + self.wait_calls += 1 + return 0 + + process = NeverExitingProcess() + with tempfile.TemporaryDirectory() as tmp: + output = pathlib.Path(tmp) / "results.csv" + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size":3}\n') + argv = [ + "eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--glm", "/fake/glm", "--out", + str(output), + ] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object(EVAL.subprocess, "Popen", + return_value=process), \ + self.assertRaises(KeyboardInterrupt): + EVAL.main() + self.assertEqual(process.terminate_calls, 1) + self.assertGreaterEqual(process.wait_calls, 1) + + @unittest.skipIf( + sys.platform == "win32", + "os.kill(pid, SIGTERM) on Windows is TerminateProcess with exit code 15: " + "no handler runs, so the POSIX mechanism under test does not exist there " + "(the SIGINT/exception half above covers child cleanup on Windows)") + def test_sigterm_mid_run_terminates_the_child_and_propagates(self): + # The SIGTERM half: a real SIGTERM (not just an ordinary + # Python exception) delivered while the child is running must + # also be converted into child cleanup, not left to Python's + # default SIGTERM handling (which does not run this module's + # `finally` cleanup at all). + request_raw = b"2 2 1 2 1 2\n" + request_digest = hashlib.sha256(request_raw).hexdigest() + + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + banner = self.BANNER + "\n" + loaded = self.loaded("absent", 2) + "\n" + + def stdout_then_sigterm(): + yield banner + yield loaded + yield f"SCORE 0 {request_digest} -1 2 1\n" + os.kill(os.getpid(), signal.SIGTERM) + # Not reached if the handler fires promptly, as it must. + yield f"SCORE 1 {request_digest} -2 2 0\n" + + class NeverExitingProcess: + def __init__(self): + self.returncode = None + self.stderr = () + self.stdout = stdout_then_sigterm() + self.terminated = False + self.terminate_calls = 0 + self.wait_calls = 0 + + def poll(self): + return 0 if self.terminated else None + + def terminate(self): + self.terminated = True + self.terminate_calls += 1 + + def wait(self): + self.wait_calls += 1 + return 0 + + process = NeverExitingProcess() + previous_handler = signal.getsignal(signal.SIGTERM) + try: + with tempfile.TemporaryDirectory() as tmp: + output = pathlib.Path(tmp) / "results.csv" + (pathlib.Path(tmp) / "config.json").write_text( + '{"vocab_size":3}\n') + argv = [ + "eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--glm", "/fake/glm", "--out", + str(output), + ] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object(EVAL.subprocess, "Popen", + return_value=process), \ + self.assertRaises(EVAL.ChildTerminateRequested): + EVAL.main() + finally: + # Defensive: main()'s own finally already restores the prior + # handler, but never trust a test to leave process-global + # signal state behind if the assertion above ever fails. + signal.signal(signal.SIGTERM, previous_handler) + self.assertEqual(process.terminate_calls, 1) + self.assertGreaterEqual(process.wait_calls, 1) + self.assertEqual(signal.getsignal(signal.SIGTERM), previous_handler) + + def test_engine_run_completes_and_reports_finished(self): + # The engine only ever emits the byte-compatible legacy + # three-field form (" "); a clean run + # over that form completes and writes every row. + rc, output, launches = self.run_eval_main(( + self.BANNER + "\n", + self.loaded("absent", 2) + "\n", + "-1 2 1\n", "-2 2 0\n", "-3 2 0\n", + )) + self.assertEqual(rc, 0) + self.assertEqual(launches, 1) + self.assertIn("# finished: 3/3", output) + + def test_classifier_refuses_unknown_lines(self): + # ScoreStdoutClassifier is the class main() actually constructs + # -- classify_score_stdout (the standalone function, covered by + # test_stdout_grammar_refuses_unknown_records) is a separate + # code path main() never calls. A foreign line must be refused + # by the classifier itself, not merely by the standalone + # function. + foreign_lines = ( + "PROF 0.001 1 1 0.000 0.000 0.000 0.000 0.000 1\n", + "not a score line at all\n", + "nan 1 0\n", + "SCORE 0 " + "a" * 64 + " -1 1 1\n", # no reader emits this shape + ) + for line in foreign_lines: + with self.subTest(line=line): + parser = EVAL.ScoreStdoutClassifier() + parser.classify(self.BANNER + "\n") + parser.classify(self.loaded("absent", 2) + "\n") + with self.assertRaises(EVAL.EvidenceError): + parser.classify(line) + + def test_eval_main_rejects_lifecycle_and_blank_records(self): + banner = self.BANNER + "\n" + loaded = self.loaded("absent", 2) + "\n" + scores = ("-1 2 1\n", "-2 2 0\n", "-3 2 0\n") + cases = { + "load_before_banner": (loaded, banner) + scores, + "score_before_load": (banner, scores[0], loaded) + scores[1:], + "duplicate_load": (banner, loaded, loaded) + scores, + "missing_load_at_eof": (banner,), + "blank_before_banner": ("\n", banner, loaded) + scores, + "blank_between_preambles": (banner, "\n", loaded) + scores, + "blank_after_scores": (banner, loaded) + scores + ("\n",), + } + for name, records in cases.items(): + with self.subTest(name=name): + rc, output, launches = self.run_eval_main(records) + self.assertEqual(rc, 1) + self.assertEqual(launches, 1) + self.assertIn("# INCOMPLETE:", output) + self.assertNotIn("# finished:", output) + + def test_only_complete_zero_exit_denominator_passes(self): + # NOTE: completion_error's contract changed from the original + # ported oracle -- it now matches dev's own exit-code contract + # exactly (a partial or nonzero-exit-but-nonempty run is no + # longer fatal here; see its docstring), so several of the + # original oracle's assertions below are inverted rather than + # reused verbatim. + self.assertIsNone(EVAL.completion_error(0, 3, 3, 9)) + # A clean exit with zero requests scored is no longer flagged by + # completion_error itself (dev's own contract only fires on a + # NONZERO exit with zero scored; `expected`/`continuation_tokens` + # are not otherwise consulted). + self.assertIsNone(EVAL.completion_error(0, 0, 0, 0)) + self.assertIsNone(EVAL.completion_error(0, 3, 3, 0)) + # A nonzero exit that still scored at least one request is a + # partial run, not fatal. + self.assertIsNone(EVAL.completion_error(2, 7, 7, 7)) + self.assertIsNone(EVAL.completion_error(0, 6, 7, 6)) + # Fatal only when the engine exits nonzero with NOTHING scored... + self.assertIsNotNone(EVAL.completion_error(2, 0, 7, 0)) + self.assertIn("zero requests scored", EVAL.completion_error(2, 0, 7, 0)) + # ...or a stream_error is present regardless of completion count. + self.assertIn("broken", EVAL.completion_error(0, 7, 7, 7, "broken")) + self.assertIn("broken", EVAL.completion_error(0, 0, 7, 0, "broken")) + + def test_empty_selection_refuses_before_engine_launch(self): + with tempfile.TemporaryDirectory() as tmp: + out = pathlib.Path(tmp) / "results.csv" + argv = ["eval_glm.py", "--snap", tmp, "--tasks", "", "--out", str(out)] + with mock.patch.object(sys, "argv", argv), \ + mock.patch.object(EVAL.subprocess, "Popen") as popen: + rc = EVAL.main() + self.assertEqual(rc, 1) + popen.assert_not_called() + text = out.read_text() + self.assertIn("# INCOMPLETE: 0/0; error=no benchmark tasks selected", text) + self.assertNotIn("finished: 0/0", text) + + def test_zero_request_task_refuses_before_engine_launch(self): + class FakeTokenizer: + @staticmethod + def from_file(path): + return object() + + with tempfile.TemporaryDirectory() as tmp: + root = pathlib.Path(tmp) + (root / "empty.jsonl").write_text("") + out = root / "results.csv" + argv = ["eval_glm.py", "--snap", tmp, "--data", tmp, + "--tasks", "empty", "--out", str(out)] + fake = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": fake}), \ + mock.patch.object(EVAL.subprocess, "Popen", + side_effect=AssertionError("engine launched")) as popen: + rc = EVAL.main() + self.assertEqual(rc, 1) + popen.assert_not_called() + text = out.read_text() + self.assertIn( + "# INCOMPLETE: 0/0; error=selected tasks produced zero SCORE requests", + text) + self.assertNotIn("finished: 0/0", text) + + def test_zero_continuation_choices_refuse_before_engine_launch(self): + class Encoded: + def __init__(self, ids): + self.ids = ids + + class BoundaryTokenizer: + @staticmethod + def from_file(path): + return BoundaryTokenizer() + + @staticmethod + def encode(text): + return Encoded({ + "ctx": [1], "ctxgood": [1, 2], "good": [2], + "ctxvanish": [1], "vanish": [], "": [], + }.get(text, [1])) + + cases = { + "one_empty": [{"ctx": "ctx", "choices": [""], "gold": 0}], + "all_empty": [{"ctx": "ctx", "choices": ["", ""], "gold": 0}], + "boundary_still_empty": [ + {"ctx": "ctx", "choices": ["vanish"], "gold": 0}], + "mixed_positive_zero": [ + {"ctx": "ctx", "choices": ["good", ""], "gold": 0}], + } + tokenizer = BoundaryTokenizer() + for name, docs in cases.items(): + with self.subTest(name=name), \ + self.assertRaisesRegex(EVAL.EvidenceError, + "no positive context/continuation"): + EVAL.build_requests(tokenizer, {"task": docs}) + + tokenizers = types.SimpleNamespace(Tokenizer=BoundaryTokenizer) + for name, docs in cases.items(): + with self.subTest(prelaunch=name), \ + tempfile.TemporaryDirectory() as tmp: + root = pathlib.Path(tmp) + data = root / f"{name}.jsonl" + data.write_text(json.dumps(docs[0]) + "\n") + out = root / "results.csv" + argv = ["eval_glm.py", "--snap", str(root), + "--data", str(root), "--tasks", name, + "--out", str(out)] + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict( + sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object( + EVAL.subprocess, "Popen", + side_effect=AssertionError( + "engine launched")) as popen: + rc = EVAL.main() + self.assertEqual(rc, 1) + popen.assert_not_called() + text = out.read_text() + self.assertIn("# INCOMPLETE: 0/0; error=", text) + self.assertNotIn("# finished:", text) + + def test_dry_run_does_not_require_a_vocabulary(self): + # dev's own --dry never looked up config.json's + # vocab_size at all -- it only builds and tokenizes requests, + # then stops. This module's vocabulary/digest binding is a + # per-request-wire step for the real engine launch, not a + # plumbing check, so --dry must not depend on it. + class Encoded: + ids = [1, 2] + + class FakeTokenizer: + @staticmethod + def from_file(path): + return FakeTokenizer() + + @staticmethod + def encode(text): + return Encoded() + + with tempfile.TemporaryDirectory() as tmp: + # No config.json at all in the snapshot directory -- a real + # vocabulary lookup would raise EvidenceError immediately. + argv = ["eval_glm.py", "--snap", tmp, "--tasks", "smoke", + "--limit", "1", "--dry"] + tokenizers = types.SimpleNamespace(Tokenizer=FakeTokenizer) + with mock.patch.object(sys, "argv", argv), \ + mock.patch.dict(sys.modules, {"tokenizers": tokenizers}), \ + mock.patch.object( + EVAL, "score_snapshot_vocab", + side_effect=AssertionError( + "vocabulary looked up during --dry")) as vocab, \ + mock.patch.object( + EVAL.subprocess, "Popen", + side_effect=AssertionError("engine launched")) as popen: + rc = EVAL.main() + self.assertIsNone(rc) + vocab.assert_not_called() + popen.assert_not_called() + + def test_partial_run_still_reports_the_table_and_exits_zero(self): + # dev's own exit-code contract (coli bench does + # `sys.exit(subprocess.call(cmd, ...))`; diag_harness.py parses + # this tool's stdout table from a subprocess call) exits nonzero + # ONLY when the engine produced nothing at all. A partial run -- + # some but not all requests scored, clean stream -- must still + # print the accuracy table and exit 0; the INCOMPLETE marker is + # additive (alongside "# finished", not instead of it). + rc, output, launches = self.run_eval_main(( + self.BANNER + "\n", + self.loaded("absent", 2) + "\n", + "-1 2 1\n", # only 1 of 3 requests scored + )) + self.assertEqual(rc, 0) + self.assertEqual(launches, 1) + self.assertIn("# finished: 1/3", output) + self.assertIn("# INCOMPLETE: 1/3 requests scored", output) + self.assertIn( + "WARNING: only 1/3 requests scored", self.last_stderr) + self.assertIn("MEAN acc_norm", self.last_stdout) + + def test_python_engine_byte_limits_are_inclusive_and_preallocation(self): + engine_limit = 256 << 20 + self.assertEqual(EVAL._ENGINE_TEXT_MAX_BYTES, engine_limit) + self.assertEqual( + EVAL._checked_engine_text_size(engine_limit, "SCORE"), + engine_limit) + with self.assertRaises(EVAL.EvidenceError): + EVAL._checked_engine_text_size(engine_limit + 1, "SCORE") + + with tempfile.TemporaryDirectory() as tmp: + root = pathlib.Path(tmp) + score_config = root / "config.json" + score_raw = b'{"vocab_size":4}\n' + score_config.write_bytes(score_raw) + with mock.patch.object( + EVAL, "_ENGINE_TEXT_MAX_BYTES", len(score_raw)): + self.assertEqual(EVAL.score_snapshot_vocab(root), 4) + score_config.write_bytes(score_raw + b" ") + with self.assertRaisesRegex(EVAL.EvidenceError, "256 MiB"): + EVAL.score_snapshot_vocab(root) + + request = "1 1 0 1" + request_bytes = len((request + "\n").encode("ascii")) + with mock.patch.object( + EVAL, "_ENGINE_TEXT_MAX_BYTES", request_bytes): + _, payload, _ = EVAL.score_request_wire((request,), 4) + self.assertEqual(len(payload), request_bytes) + with mock.patch.object( + EVAL, "_ENGINE_TEXT_MAX_BYTES", request_bytes - 1): + with self.assertRaisesRegex(EVAL.EvidenceError, "256 MiB"): + EVAL.score_request_wire((request,), 4) + + +class DifferentialBiteTests(unittest.TestCase): + """dev's classifier silently accepts a foreign stdout line the new + copy refuses. dev's side is not itself invoked here (no ported dev + module exists in this tree); it is asserted against the same fixture + line via dev's documented filter logic, ported verbatim inline. Only + the new copy's refusal is exercised by calling real code.""" + + FOREIGN_LINE = "1 1 1\n" # shaped like dev's own accepted grammar + + def test_dev_copy_silently_accepts_a_foreign_numeric_line(self): + # dev's inline stdout filter (ported verbatim as the oracle): any + # line starting with a digit or '-' is treated as a SCORE record, + # with no further validation at all. + line = self.FOREIGN_LINE.strip() + self.assertTrue(line and line[0] in "-0123456789") + parts = line.split() + logprob = float(parts[0]) # dev: "try: logprob = float(parts[0])" + # dev accepts this as a real SCORE result -- a false positive: a + # log-likelihood can never be positive, but dev's filter never + # checks the sign (or finiteness, or field count) at all. + self.assertEqual(logprob, 1.0) + + def test_new_copy_refuses_the_same_foreign_line_by_name(self): + with self.assertRaisesRegex( + EVAL.EvidenceError, "SCORE logprob is not finite/non-positive"): + EVAL.classify_score_stdout(self.FOREIGN_LINE) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_exact_dot.c b/c/tests/test_exact_dot.c new file mode 100644 index 000000000..df2881610 --- /dev/null +++ b/c/tests/test_exact_dot.c @@ -0,0 +1,60 @@ +/* test_exact_dot.c — properties of exact_dot.h (the opt-in exact verify mode's kernel): + * 1. order independence: 200 random permutations of a cancellation-heavy f32 dot are bit-identical; + * 2. exactness: 1e30 + 1 - 1e30 style sums come out exactly, where float accumulation loses them; + * 3. agreement: when no rounding is possible (small integers), exact == naive; + * 4. (w*s)*x path: order independent and equal to the f32 path on exactly-representable data; + * 5. specials: NaN / inf classification matches float; + * 6. cost: ns per element vs a plain float loop (printed, not asserted). */ +#include +#include +#include +#include +#include +#include "../exact_dot.h" + +static uint64_t rs = 0x9E3779B97F4A7C15ull; +static uint32_t rnd(void){ rs ^= rs << 13; rs ^= rs >> 7; rs ^= rs << 17; return (uint32_t)(rs >> 11); } +static float rndf(void){ uint32_t u = rnd(); int e = 100 + (int)(u % 55); return ((u & 1) ? -1.f : 1.f) * ldexpf((float)((u >> 8) & 0xFFFF) + 1, e - 127 - 16); } +static uint32_t fbits(float f){ uint32_t u; memcpy(&u, &f, 4); return u; } +static float naive(const float *x, const float *y, int n){ float a = 0; for(int i = 0; i < n; i++) a += x[i] * y[i]; return a; } +static int fails = 0; +#define CHECK(c, msg) do{ if(!(c)){ printf(" [FAIL] %s\n", msg); fails++; } else printf(" [PASS] %s\n", msg); }while(0) + +int main(void){ + enum { N = 4096 }; + static float x[N], y[N]; static int idx[N]; + for(int i = 0; i < N; i++){ x[i] = rndf(); y[i] = rndf(); } + for(int i = 0; i < N; i += 2){ x[i + 1] = -x[i]; y[i + 1] = y[i] * 1.0000001f; } /* near-cancelling pairs */ + float ref = exd_dot_ff(x, y, N); int same = 1; + for(int p = 0; p < 200 && same; p++){ + for(int i = 0; i < N; i++) idx[i] = i; + for(int i = N - 1; i > 0; i--){ int j = (int)(rnd() % (uint32_t)(i + 1)); int t = idx[i]; idx[i] = idx[j]; idx[j] = t; } + exd_acc a; exd_init(&a); for(int i = 0; i < N; i++) exd_add_ff(&a, x[idx[i]], y[idx[i]]); + if(fbits(exd_finish(&a)) != fbits(ref)) same = 0; + } + CHECK(same, "order independence: 200 permutations bit-identical (cancellation-heavy)"); + { float a[3] = {1e30f, 1.f, -1e30f}, b[3] = {1.f, 1.f, 1.f}; + CHECK(exd_dot_ff(a, b, 3) == 1.0f && naive(a, b, 3) == 0.0f, "exactness: 1e30 + 1 - 1e30 == 1 (float loop gives 0)"); } + { static float xi[N], yi[N]; for(int i = 0; i < N; i++){ xi[i] = (float)((int)(rnd() % 9) - 4); yi[i] = (float)((int)(rnd() % 9) - 4); } + CHECK(fbits(exd_dot_ff(xi, yi, N)) == fbits(naive(xi, yi, N)), "agreement with naive when no rounding can occur"); } + { static int w[N]; static float s[N], xv[N]; float refw; exd_acc a; exd_init(&a); + for(int i = 0; i < N; i++){ w[i] = (int)(rnd() % 16) - 8; s[i] = ldexpf(1.f + (float)(rnd() % 7) / 8.f, -(int)(rnd() % 6)); xv[i] = rndf(); exd_add_wsx(&a, w[i], s[i], xv[i]); } + refw = exd_finish(&a); int ok = 1; + for(int p = 0; p < 50 && ok; p++){ for(int i = 0; i < N; i++) idx[i] = i; + for(int i = N - 1; i > 0; i--){ int j = (int)(rnd() % (uint32_t)(i + 1)); int t = idx[i]; idx[i] = idx[j]; idx[j] = t; } + exd_acc b; exd_init(&b); for(int i = 0; i < N; i++) exd_add_wsx(&b, w[idx[i]], s[idx[i]], xv[idx[i]]); if(fbits(exd_finish(&b)) != fbits(refw)) ok = 0; } + CHECK(ok, "(w*s)*x path: 50 permutations bit-identical"); + exd_acc c2; exd_init(&c2); for(int i = 0; i < N; i++) exd_add_ff(&c2, (float)w[i] * s[i], xv[i]); /* w*s exact for these s */ + CHECK(fbits(exd_finish(&c2)) == fbits(refw), "(w*s)*x == exact f32 path when w*s is representable"); } + { float a[2] = {1.f, INFINITY}, b[2] = {1.f, 2.f}; CHECK(isinf(exd_dot_ff(a, b, 2)) && exd_dot_ff(a, b, 2) > 0, "inf propagates with sign"); + float c[2] = {INFINITY, 0.f}, d[2] = {0.f, 1.f}; CHECK(isnan(exd_dot_ff(c, d, 2)), "inf*0 -> nan"); + float e[2] = {NAN, 1.f}; CHECK(isnan(exd_dot_ff(e, b, 2)), "nan propagates"); } + { struct timespec t0, t1; volatile float sink = 0; int reps = 200; + clock_gettime(CLOCK_MONOTONIC, &t0); for(int r = 0; r < reps; r++) sink += naive(x, y, N); clock_gettime(CLOCK_MONOTONIC, &t1); + double tn = ((t1.tv_sec - t0.tv_sec) * 1e9 + (t1.tv_nsec - t0.tv_nsec)) / (double)reps / N; + clock_gettime(CLOCK_MONOTONIC, &t0); for(int r = 0; r < reps; r++) sink += exd_dot_ff(x, y, N); clock_gettime(CLOCK_MONOTONIC, &t1); + double te = ((t1.tv_sec - t0.tv_sec) * 1e9 + (t1.tv_nsec - t0.tv_nsec)) / (double)reps / N; + printf(" [COST] naive float loop %.2f ns/elem, exact %.2f ns/elem (%.1fx), n=%d\n", tn, te, te / tn, N); (void)sink; } + printf("%s (%d failures)\n", fails ? "FAIL" : "ALL PASS", fails); + return fails ? 1 : 0; +} diff --git a/c/tests/test_fp8_cuda.cu b/c/tests/test_fp8_cuda.cu index 6c98c51e6..fe374fd06 100644 --- a/c/tests/test_fp8_cuda.cu +++ b/c/tests/test_fp8_cuda.cu @@ -27,7 +27,9 @@ #include #include #include -#include +#if !defined(__HIPCC__) +#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */ +#endif #include "../backend_cuda.cu" @@ -333,5 +335,55 @@ int main(void){ } coli_cuda_shutdown(); } + + /* ---- Phase 3: LUT-gate lifecycle across shutdown/re-init --------------- */ + { /* g_fp8_lut_ready is process-wide while the e4m3 table is per-device. + * SHUTDOWN is the only site that clears it; coli_cuda_init never writes + * it. That is sufficient because init refuses to rebuild contexts while + * a set is live: a re-init naming the SAME set returns 1 and leaves the + * contexts -- and therefore the published table -- untouched, and one + * naming a DIFFERENT set is refused outright. So the device set cannot + * widen past what the last publish covered without passing through + * shutdown, which clears the flag. This block pins all three edges. */ + int devs[1]={0}; + enum { LO=4, LI=128 }; + uint8_t lw[LO*LI]; float ls[1]={1.f}; + for(size_t i=0;i1 ? 1 : ndev}; /* {0,1} on a multi-GPU box, else out of range */ + if(coli_cuda_init(other,2)!=0){ printf("FAIL different-set re-init was not refused\n"); return 1; } + if(coli_cuda_device_count()!=1){ printf("FAIL refused re-init still changed the context set\n"); return 1; } + if(!coli_cuda_tensor_upload(<,lw,ls,8,LI,LO,0)){ printf("FAIL refused re-init disturbed the LUT flag\n"); return 1; } + coli_cuda_tensor_free(lt); lt=nullptr; + } + + coli_cuda_shutdown(); + printf("lut-gate lifecycle: shutdown clears; same-set re-init keeps; different-set re-init refused\n"); + } printf("OK\n"); return 0; } diff --git a/c/tests/test_fp8_refusal_canary.py b/c/tests/test_fp8_refusal_canary.py new file mode 100644 index 000000000..4d6dda306 --- /dev/null +++ b/c/tests/test_fp8_refusal_canary.py @@ -0,0 +1,120 @@ +"""Drift canary for the absorb-path fmt-refusal message shape. + +tests/test_fp8_serve_batch_e2e.py machine-matches the engine's death on the +batched fmt=8 serve path with one regex (its REFUSAL constant, the family +`(qt_addrow|qt_matvec_rows): unsupported fmt=`), in two load-bearing +places: the pass lane's engine-death scan, and the expect-bite mode the +fleet's old-binary half PASSES ONLY THROUGH. The product strings live in +c/colibri.c (qt_addrow's refusal and qt_matvec_rows'); an innocent reword +of either -- "unsupported" to "unhandled", dropping the function-name +prefix -- would not break any build or unit suite, but it would silently +degrade the bite proof's machine-match into a generic "engine died", and +the e2e test only runs where a real fmt=8 container exists (never in CI). +This canary closes that gap in CI: it fails, by name, the moment the +matcher and the source strings drift apart, in either direction. + +What is load-bearing is exactly what this file asserts, no more: + * each function still has at least one refusal-class stderr message + (its runtime firing is pinned separately by tests/test_qt_addrow.c's + fork+waitpid refusal cases, which also require the "refus" word); + * at least ONE refusal-class message from each function matches the e2e + REFUSAL regex -- imported from the e2e module, never re-typed, so the + two files cannot agree by coincidence (existential, not universal: a + future second, differently-worded guard in the same function is a new + refusal, not drift, and must not fail this canary); + * the regex is still SELECTIVE -- it must not match an arbitrary death + line, or expect-bite would file any crash as the bite. +Everything else about the messages (wording after the matched prefix, +fmt lists, line breaks) is deliberately unpinned: message edits that keep +the matchable shape must stay free. +""" +import re +import sys +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent +sys.path.insert(0, str(HERE)) # sibling import under any invocation style +from test_fp8_serve_batch_e2e import REFUSAL + +COLIBRI_C = HERE.parent / "colibri.c" +FUNCTIONS = ("qt_addrow", "qt_matvec_rows") + +# One C string literal (escapes included), and an fprintf-to-stderr whose +# message is one or more adjacent literals (the refusals wrap across lines). +_STRING = r'"(?:[^"\\\n]|\\.)*"' +_FPRINTF = re.compile(r'fprintf\s*\(\s*stderr\s*,\s*(' + _STRING + r'(?:\s*' + _STRING + r')*)') + + +def stderr_messages(source): + """Every fprintf(stderr, ...) format string in `source`, with adjacent + literals concatenated (still escaped -- the matched prefix contains no + escapes, so matching against the raw literal is faithful to what the + runtime prints before the %d substitution).""" + messages = [] + for match in _FPRINTF.finditer(source): + parts = re.findall(_STRING, match.group(1)) + messages.append("".join(part[1:-1] for part in parts)) + return messages + + +class RefusalShapeCanaryTest(unittest.TestCase): + maxDiff = None + + @classmethod + def setUpClass(cls): + cls.messages = stderr_messages(COLIBRI_C.read_text(encoding="utf-8")) + + def messages_of(self, function): + return [m for m in self.messages if m.startswith(function + ":")] + + def test_each_function_still_names_a_refusal_the_matcher_greps(self): + for function in FUNCTIONS: + named = self.messages_of(function) + self.assertTrue( + named, + f"colibri.c has no stderr message starting '{function}:' -- the " + f"refusal site was renamed or removed, and the D-I2 bite matcher " + f"(test_fp8_serve_batch_e2e.REFUSAL) can no longer identify this " + f"death; update the matcher and the fleet bite lane together") + refusals = [m for m in named if "refus" in m] + self.assertTrue( + refusals, + f"{function}: no refusal-class stderr message left (the 'refus' " + f"discipline tests/test_qt_addrow.c pins at runtime) -- if the " + f"refusal moved, this canary and the e2e matcher must follow it") + self.assertTrue( + any(REFUSAL.search(m) for m in refusals), + f"none of {function}'s refusal messages matches the e2e " + f"REFUSAL regex {REFUSAL.pattern!r} -- an innocent reword " + f"silently turns the fleet bite proof's machine-match into a " + f"generic 'engine died'; keep one matchable refusal per " + f"function or change test_fp8_serve_batch_e2e.REFUSAL in the " + f"same commit. (Existential on purpose: an ADDITIONAL, " + f"differently-worded guard in this function is new coverage, " + f"not drift.)\nmessages: {refusals}") + + def test_matcher_stays_selective(self): + """The other drift direction: a loosened REFUSAL regex would make + expect-bite accept ANY death as the bite. Pin that it rejects a + representative non-refusal death line and the empty string, and + that it keys on the FUNCTION FAMILY, not the bare `unsupported + fmt=` words -- a matcher loosened to the words alone would accept + any future non-absorb `unsupported fmt=` message as the bite.""" + self.assertIsNone(REFUSAL.search("malloc: out of memory allocating 42 GB")) + self.assertIsNone(REFUSAL.search("colibri engine exited unexpectedly")) + self.assertIsNone(REFUSAL.search("")) + self.assertIsNone(REFUSAL.search("qt_resolve_fmt: unsupported fmt=9 refusing")) + # The REAL in-tree near-miss (not synthetic): layer_cuda_shard_kvb's + # refusal shares the `unsupported ... fmt=` words but is a different + # guard family (shard-layout, pinned by tests/test_shard_kvb_refuse.c) + # -- a loosened matcher that swallowed it would misfile a shard + # refusal as the absorb bite in a fleet lane. + self.assertIsNone(REFUSAL.search( + "layer_cuda_shard_kvb: unsupported kv_b fmt=5 for the head-shard " + "upload (only fmt 1/2/3/4 match the per-row byte/scale strides " + "computed here) -- refusing the shard")) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_fp8_serve_batch_e2e.py b/c/tests/test_fp8_serve_batch_e2e.py new file mode 100644 index 000000000..52bb35b62 --- /dev/null +++ b/c/tests/test_fp8_serve_batch_e2e.py @@ -0,0 +1,668 @@ +"""Defect-closure pin for the batched-serve fmt=8 abort (D-I2): two-slot +SERVE_BATCH mux against a REAL fp8 (fmt=8) `kv_b_proj` container. + +The defect: batched serve mode (openai_server's Engine always launches the +engine with SERVE_BATCH=1; --kv-slots > 1 gives the mux real slots) decodes +kv_b through the MLA absorb path -- qt_addrow / qt_matvec_rows in colibri.c. +Before the fmt=8 absorb branches landed, those functions had no fp8 case: +the first generation request against a container whose kv_b_proj resolves to +fmt=8 (fp8-e4m3-b128 -- both production fp8 container classes, f8_full and +f8x4g64, are in this class) killed the engine with the named refusal + + qt_addrow: unsupported fmt=8 for the per-row-scale absorb path ... + -- refusing rather than misread t->s[row]/t->q4 + +and the client saw a 500 `engine_error`. Single-stream chat never hit it, +so nothing in the FakeEngine-backed server suite could: the abort lives +below the engine wire protocol, on a code path only a real fmt=8 container +reaches. This test pins the closure end to end: serve the real container +batched, occupy BOTH KV slots concurrently, and require 200 + text on both +requests with the engine still alive afterwards. + +A real fmt=8 container is hundreds of GB and exists only on the model +hosts, never in CI, so discovery is by environment variable: + + COLI_FP8_CONTAINER root of an engine-loadable container whose + kv_b_proj is fp8 (fmt=8). Unset => named SKIP. + COLI_ENGINE engine binary override (same convention as coli); + default: the built `colibri` next to the server. + SET but wrong => loud FAIL, never a skip. + COLI_FP8_EXPECT_BITE=1 proof-of-bite mode for the fleet's old-binary + half: the run PASSES only if the invocation FAILS + to serve AND the captured server stderr carries + the named `unsupported fmt=` refusal. A death + without the refusal, or a clean serve, FAILS. + COLI_FP8_READY_TIMEOUT seconds to wait for the server to come up + (default 1800 -- container loads are large). + COLI_FP8_GENERATE_TIMEOUT seconds per generation request (default + 1800; two concurrent cold prefills on a + storage-bound host can be slow). + +Set-but-wrong is a FAILURE, not a skip: if COLI_FP8_CONTAINER names a +missing path, a container whose kv_b_proj is NOT fmt=8 (e.g. an int8 +spine), or a shard set with duplicate tensor names, this test would +otherwise pass (or fail for the wrong reason) without ever entering the +absorb branch it exists to pin -- the vacuous-gate hazard. The safetensors +headers are checked first (pure header reads, no tensor data) and the +mismatch is a loud failure naming what was found. The same doctrine covers +COLI_ENGINE: a set-but-nonexistent engine path fails, it does not skip. +(Known unreachable corner: a shape with O == ceil(O/128)*ceil(I/128) makes +per-row and per-block scale counts collide and the unstamped loader would +resolve fmt=1 where this check says fmt=8 -- real GLM kv_b O is in the +thousands, so no planned container can reach it.) + +The engine/server child environment is constructed fresh: ambient COLI_* +and engine-knob variables (KV8, KV_TQ, ...) are stripped except an +explicit backend/location allowlist plus two value-restricted knobs -- +exactly ABSORB=0 (the ratified non-absorb fleet arm; any other ABSORB +value is stripped), and exactly COLI_CUDA_ATTN=1 WHEN the lane opts in +with COLI_FP8_E2E_CUDA_ABSORB=1 (default: stripped, so a leaked ambient +COLI_CUDA_ATTN can never swap the code path under test) -- and a leaked +COLI_API_KEY cannot 401 the ready poll; the kept/dropped set is printed +and attached to every failure so the run artifact shows the env the +invocation actually saw. + +Lane semantics, stated exactly: by DEFAULT every lane (CPU or CUDA host) +exercises the CPU absorb arms -- the CUDA fmt=8 absorb decode is gated +on COLI_CUDA_ATTN (path selection) AND CUDA_DENSE (kv_b's +cuda_eligible), both of which this test strips unless the lane +explicitly opts in. A CUDA lane that sets COLI_FP8_E2E_CUDA_ABSORB=1 +(plus its COLI_CUDA/COLI_GPU bindings and ambient COLI_CUDA_ATTN=1 and +CUDA_DENSE=1) pins the CUDA absorb decode end-to-end, and the test then +REQUIRES the engine's GPU-dense boot line as a witness -- an opt-in run +whose engine reports resident-dense-on-CPU fails rather than banking a +vacuous green. Without the opt-in, the CUDA absorb arm is pinned only +by tests/test_backend_cuda.cu. +""" +import json +import os +import re +import signal +import socket +import struct +import subprocess +import sys +import threading +import time +import unittest +from pathlib import Path +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen + +HERE = Path(__file__).resolve().parent +C_DIR = HERE.parent +CONTAINER = os.environ.get("COLI_FP8_CONTAINER") +ENGINE_OVERRIDE = os.environ.get("COLI_ENGINE") +EXPECT_BITE = os.environ.get("COLI_FP8_EXPECT_BITE") == "1" +READY_TIMEOUT = float(os.environ.get("COLI_FP8_READY_TIMEOUT", "1800")) +GENERATE_TIMEOUT = float(os.environ.get("COLI_FP8_GENERATE_TIMEOUT", "1800")) +PROMPT = "The primary colors are" # the DR-10 fixed prompt: short, deterministic continuation +MAX_TOKENS = 24 +# The refusal family this test exists to keep dead. Both absorb entry points +# carry the same named message shape; match the family, not one function, so +# a regression through either surfaces by name in the failure output. +REFUSAL = re.compile(r"(qt_addrow|qt_matvec_rows): unsupported fmt=") +ENGINE_DEATH = "colibri engine exited unexpectedly" + +# Child-environment policy (the invocation of record must not ride ambient +# state): backend selection and read-only model LOCATION are the only knob +# namespaces a lane may pass through -- CUDA lanes need COLI_CUDA/ +# COLI_GPU(S) per the DR-10 bindings, and a split/mirrored container needs +# the engine to be TOLD where its shards live (location config says where +# the weights ARE; a behavior knob changes what the engine DOES with them +# -- only the former class passes): +# COLI_MODEL_DIRS -- extra shard directories (st_init_multi's SPLIT +# layout); stripping it hides the operator's shards +# and fails as a bogus "not an fmt=8 container". +# COLI_MODEL_MIRROR -- read-only replica dirs (multi-SSD read fan-out). +# COLI_MMAP -- how weights are mapped from disk; load +# placement/feasibility, not decode semantics. +# COLI_DISK_WEIGHTS -- disk-resident weight policy; feasibility for +# hundreds-of-GB loads, not decode semantics. +# Every other COLI_* (COLI_API_KEY -> 401'd ready poll) and the bare +# engine knobs below (KV8/KV_TQ change the KV format of record; +# SNAP/SERVE/SERVE_BATCH/NGEN/KV_SLOTS belong to the server, which sets +# its own) are stripped, and the strip is recorded. Documented legacy +# aliases are stripped ALONGSIDE their primaries so a scrubbed primary +# cannot resurface under its old name: SNAP_MIRROR (consulted only when +# COLI_MODEL_MIRROR is unset/empty -- i.e. precisely after this scrub) is +# name-stripped, and TEMP is stripped only when FULLY NUMERIC (matching +# temp_from_env's strtod whole-string test: a numeric TEMP is the +# deprecated sampling alias, a path TEMP is the Windows/ROCm temp +# directory and must survive). Value-restricted exceptions: the ratified +# non-absorb fleet arm selects itself with ABSORB=0, so exactly that +# value passes (any other ambient ABSORB is stripped -- both decode paths +# at ABSORB=0/default are ratified-equivalent, so a leaked "0" can shift +# which path is pinned but never fake a pass). +# +# The CUDA-absorb opt-in (COLI_FP8_E2E_CUDA_ABSORB=1) is handled by +# INJECTION, not passthrough -- see CUDA_ABSORB_INJECT below. A first +# attempt value-allowed COLI_CUDA_ATTN=1 and CUDA_DENSE=1 through the scrub +# and relied on the LANE to set both ambient; the fresh-env allowlist only +# ever KEEPS a variable already in the environment, so when the lane set +# COLI_CUDA_ATTN=1 but not the bare (non-COLI) CUDA_DENSE, the eligibility +# knob never reached the child and the run booted resident-dense-on-CPU +# (the boot-line witness caught it as a hard failure). The opt-in is the +# single source of truth for the bundle the path needs, so the test now +# SETS it directly and reports it, instead of hoping the lane assembled +# the same set correctly. +ENV_KNOBS_ALLOWED = ("COLI_CUDA", "COLI_GPU", "COLI_GPUS", "COLI_NO_OMP_TUNE", + "COLI_MODEL_DIRS", "COLI_MODEL_MIRROR", "COLI_MMAP", + "COLI_DISK_WEIGHTS") +ENV_KNOBS_VALUE_ALLOWED = {"ABSORB": ("0",)} +ENV_KNOBS_STRIPPED = ("KV8", "KV_TQ", "ABSORB", "CAP_RAISE", "CUDA_DENSE", + "COLI_CUDA_ATTN", "SNAP", "SNAP_MIRROR", "SERVE", + "SERVE_BATCH", "NGEN", "KV_SLOTS") +CUDA_ABSORB_OPT_IN = "COLI_FP8_E2E_CUDA_ABSORB" +# The exact engine knobs the CUDA fmt=8 absorb decode needs, INJECTED into +# the child (with the lane's own COLI_CUDA=1 backend binding) when the lane +# opts in. COLI_CUDA_ATTN=1 selects the CUDA absorb dispatch; CUDA_DENSE=1 +# makes kv_b/o cuda_eligible -- qt_load grants eligibility only under +# g_cuda_dense (colibri.c:2123) and there is NO VRAM/budget gate on it +# (colibri.c:10966), so CUDA_DENSE=1 reaching the child is sufficient on a +# device that fits the dense tensors. CUDA_DENSE is a bare name (no COLI_ +# prefix), which is exactly why passthrough could not carry it and explicit +# injection is required. Both are stripped from ambient (above) so an +# unrequested value can never leak in; under the opt-in the test's own "1" +# is what the child sees, recorded in the kept-knobs line as injected. +CUDA_ABSORB_INJECT = {"COLI_CUDA_ATTN": "1", "CUDA_DENSE": "1"} + +# Boot-line witness for the opt-in arm: requesting a path is not proof it +# ran. The engine announces its residency decision at boot +# (colibri.c:11047-11049); under the opt-in this test asserts the GPU-dense +# line and REFUSES the CPU-dense line, so a misconfigured lane (e.g. the +# eligibility knob lost again) can never bank a vacuous green. +CUDA_DENSE_BOOT = "[CUDA] mode: routed experts + resident dense tensors" +CUDA_CPU_DENSE_BOOT = "[CUDA] mode: routed experts only (resident dense on CPU)" + + +def observed_cuda_boot_mode(stderr_text): + """The engine's `[CUDA] mode: ...` boot line as it appears in stderr, or + None if it never printed. Echoed by the witness so a PASS artifact + SHOWS the observed residency, not just an assertion that succeeded.""" + marker = "[CUDA] mode:" + start = stderr_text.find(marker) + if start == -1: + return None + end = stderr_text.find("\n", start) + return stderr_text[start:end if end != -1 else None] + + +def cuda_absorb_witness_failure(stderr_text): + """None when the boot-line witness holds for the CUDA-absorb opt-in; + otherwise the failure message. + + It checks two things against the engine's stderr: that the + resident-dense-on-CPU boot line is absent, and that the + routed+resident-dense line is present. Module-level so it can be called + without constructing the test case, and so it can be exercised directly + once there is a caller outside the container-gated test below -- there is + none today, so on a machine without a container this function has no + coverage at all.""" + if CUDA_CPU_DENSE_BOOT in stderr_text: + return ("CUDA-absorb opt-in was requested but the engine booted with " + f"{CUDA_CPU_DENSE_BOOT!r} -- resident dense stayed on the CPU, " + "kv_b never became cuda_eligible, and the CUDA absorb path " + "cannot have run (vacuous green refused; check CUDA_DENSE=1 " + "reached the child, see the kept-knobs line)") + if CUDA_DENSE_BOOT not in stderr_text: + return ("CUDA-absorb opt-in was requested but the engine's boot " + f"witness {CUDA_DENSE_BOOT!r} never appeared on stderr -- " + "cannot certify the CUDA absorb path ran") + return None + + +def _numeric_temp(value): + """temp_from_env's whole-string strtod test: TEMP acts as the deprecated + sampling alias only when fully numeric; a path value is the system temp + directory and is not a knob.""" + if not value: + return False + try: + float(value) + return True + except ValueError: + return False + + +def _child_env(): + """Fresh server/engine environment per the policy above. Returns + (env, report): the report names kept knobs with values and dropped + knobs by NAME only (a dropped COLI_API_KEY must not leak its value + into a run artifact).""" + opt_in = os.environ.get(CUDA_ABSORB_OPT_IN) == "1" + env, kept, dropped = {}, [], [] + for name, value in os.environ.items(): + # The CUDA-absorb bundle is INJECTED below when opted in, never taken + # from ambient -- so the child sees the test's "1", not a leaked + # value, and an unrequested ambient copy is always stripped here. + if name in CUDA_ABSORB_INJECT: + dropped.append(name) + elif name in ENV_KNOBS_ALLOWED or value in ENV_KNOBS_VALUE_ALLOWED.get(name, ()): + env[name] = value + kept.append(f"{name}={value}") + elif name.startswith("COLI_") or name in ENV_KNOBS_STRIPPED: + dropped.append(name) + elif name == "TEMP" and _numeric_temp(value): + dropped.append(name) # numeric TEMP = deprecated COLI_TEMP alias + else: + env[name] = value # non-knob system env (PATH, HOME, LD_LIBRARY_PATH, ...) + if opt_in: + for name, value in CUDA_ABSORB_INJECT.items(): + env[name] = value + kept.append(f"{name}={value} (injected)") + dropped = [d for d in dropped if d not in CUDA_ABSORB_INJECT] + return env, ("kept knobs: " + (", ".join(sorted(kept)) or "(none)") + + "; dropped knobs: " + (", ".join(sorted(dropped)) or "(none)")) + + +def _default_engine(): + """The built glm engine next to openai_server.py -- same candidates, + same order, as the server's own default_engine().""" + for name in ("colibri", "glm"): + for suffix in ("", ".exe"): + candidate = C_DIR / (name + suffix) + if candidate.exists(): + return candidate + return C_DIR / "colibri" + + +ENGINE = Path(ENGINE_OVERRIDE) if ENGINE_OVERRIDE else _default_engine() + + +def _shard_headers(container): + """Merged {tensor name: (shape, nbytes)} across every *.safetensors shard + header in the container root, plus the duplicate-name collisions found on + the way. Header-only reads (u64 length + JSON): no tensor data is + touched, so scanning a 140-shard container is cheap. Duplicates are + returned rather than silently last-shard-wins: the engine refuses a + duplicate tensor name across shards (st_init's D-2 guard), so a merged + map that picked either copy could disagree with the load in either + direction.""" + headers, owner, duplicates = {}, {}, [] + shards = sorted(Path(container).glob("*.safetensors")) + for shard in shards: + with open(shard, "rb") as f: + (hlen,) = struct.unpack(" 0, + f"COLI_FP8_CONTAINER={CONTAINER} holds no *.safetensors shards") + self.assertFalse( + duplicates, + f"COLI_FP8_CONTAINER={CONTAINER} carries duplicate tensor names " + f"across shards (the engine's st_init refuses exactly this, and " + f"this check could otherwise judge a shard the engine never " + f"loads): {'; '.join(duplicates[:5])}") + is_fmt8, detail = _kv_b_fmt8_check(headers) + self.assertTrue(is_fmt8, + f"COLI_FP8_CONTAINER={CONTAINER} is not an fmt=8 kv_b " + f"container; this test would be vacuous against it: {detail}") + + server = _Server(_free_port()) + self.addCleanup(server.close) + + # Startup: poll /v1/models (touches only the server, not the engine + # generate path) until it answers, and read the served model id from + # it rather than hardcoding one. Sleep EVERY iteration -- a fast + # non-200 answer must not turn the poll into a hot spin -- and fail + # fast on auth/host-guard rejections, which no amount of waiting + # will turn into a 200. + deadline = time.time() + READY_TIMEOUT + model_id = None + while time.time() < deadline: + if server.proc.poll() is not None: + self._fail_with_server_evidence( + server, f"server exited rc={server.proc.returncode} before READY") + status = None + try: + status, body = _get(server.url("/v1/models"), timeout=10) + except (URLError, OSError): + pass # not accepting yet -- keep waiting + if status == 200: + model_id = json.loads(body)["data"][0]["id"] + break + if status in (401, 403): + self._fail_with_server_evidence( + server, f"ready poll got HTTP {status} from /v1/models -- " + f"an auth/host-guard rejection, not a slow load; " + f"waiting longer cannot fix it") + time.sleep(2) + if model_id is None: + self._fail_with_server_evidence( + server, f"server not ready within {READY_TIMEOUT:.0f}s") + + # The D-I2 invocation class: two generation requests IN FLIGHT + # TOGETHER, pinned to distinct KV slots so the 2-slot mux really + # multiplexes (conversation hashing would put one identical prompt in + # one slot). temperature 0 / fixed prompt per the invocation of + # record. On a pre-fix engine the FIRST decode through the absorb + # path kills the engine and one or both of these come back 500. + results = [None, None] + + def ask(slot): + results[slot] = _post(server.url("/v1/completions"), { + "model": model_id, "prompt": PROMPT, "max_tokens": MAX_TOKENS, + "temperature": 0, "cache_slot": slot, + }, timeout=GENERATE_TIMEOUT) + + threads = [threading.Thread(target=ask, args=(slot,)) for slot in (0, 1)] + for t in threads: + t.start() + for t in threads: + t.join(GENERATE_TIMEOUT + 60) + + if EXPECT_BITE: + self._assert_bite(server, results) + return + + for slot, outcome in enumerate(results): + self.assertIsNotNone(outcome, f"slot {slot}: request never completed " + f"(asking thread still blocked)") + kind = outcome[0] + if kind == "timeout": + self._fail_with_server_evidence( + server, f"slot {slot}: generation request timed out after " + f"{outcome[1]:.0f}s (COLI_FP8_GENERATE_TIMEOUT to raise; " + f"a timeout is not a hang and not a protocol failure)") + if kind == "neterr": + self._fail_with_server_evidence( + server, f"slot {slot}: connection failed mid-request " + f"({outcome[1]}) -- engine/server dropped the stream") + _, status, body, resp_headers = outcome + if status != 200: + self._fail_with_server_evidence( + server, + f"slot {slot}: HTTP {status} (the D-I2 defect signature is a " + f"500 engine_error here)\nresponse body: " + f"{body.decode(errors='replace')[:2000]}") + payload = json.loads(body) + text = payload["choices"][0]["text"] + # "Coherent content", pinned mechanically: nonempty text with at + # least one real word, and the engine accounted for generated + # tokens. (Semantic quality belongs to the numeric battery, not + # this defect pin.) + self.assertTrue(text.strip(), + f"slot {slot}: 200 with empty text -- content-free " + f"success is not closure") + self.assertRegex(text, r"[A-Za-z]{2}", + f"slot {slot}: no word-like content in {text!r}") + self.assertGreater(payload["usage"]["completion_tokens"], 0, + f"slot {slot}: usage reports zero generated tokens") + # Concurrency witness: with two requests pinned to two distinct + # free slots, neither should queue. A silently serialized mux + # would park the second request for the first one's whole + # generation -- minutes, not milliseconds -- while every other + # assertion here stayed green. + queue_wait = resp_headers.get("x-colibri-queue-wait-ms") + self.assertIsNotNone(queue_wait, + f"slot {slot}: no x-colibri-queue-wait-ms response " + f"header -- the admission witness is gone") + self.assertLess(float(queue_wait), 1000.0, + f"slot {slot}: queued {queue_wait} ms behind the other " + f"slot -- the 2-slot mux is serializing, not batching") + # Echo the observed admission latency: the concurrency witness is + # a positive coverage claim, so its measured value belongs in the + # PASS artifact, not only in the failure path. + print(f"[fp8-e2e] witness: slot {slot} admission queue-wait = " + f"{queue_wait} ms", flush=True) + + # No engine death, three ways: the named refusal family never fired, + # the server never recorded an engine exit, and the server still + # answers after both generations. + stderr = server.stderr() + refusal = REFUSAL.search(stderr) + self.assertIsNone( + refusal, "the engine printed the absorb-path refusal this test " + f"exists to keep dead:\n{stderr[stderr.rfind('qt_'):][:500]}") + self.assertNotIn(ENGINE_DEATH, stderr, + "the engine died during the batched run") + self.assertIsNone(server.proc.poll(), "server process exited mid-test") + status, _ = _get(server.url("/v1/models"), timeout=30) + self.assertEqual(status, 200, "server stopped answering after the batched pair") + + # Boot-line witness (opt-in arm only): requesting the CUDA absorb + # path is not proof it ran -- certify it from the engine's own boot + # report, and refuse the CPU-dense line outright. The observed line + # is ECHOED on both pass and fail (positive coverage must be a + # captured artifact, not an inference from a green assertion), so a + # CI log carries the routed+dense observation verbatim. + if os.environ.get(CUDA_ABSORB_OPT_IN) == "1": + observed = observed_cuda_boot_mode(stderr) + print(f"[fp8-e2e] witness: observed boot mode = {observed!r}", + flush=True) + witness_failure = cuda_absorb_witness_failure(stderr) + if witness_failure: + self._fail_with_server_evidence(server, witness_failure) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_gguf_dequant.py b/c/tests/test_gguf_dequant.py new file mode 100644 index 000000000..3efd21881 --- /dev/null +++ b/c/tests/test_gguf_dequant.py @@ -0,0 +1,133 @@ +#!/usr/bin/env python3 +"""tools/gguf_dequant.py: ggml block dequantization, pinned to ggml's math. + +Two independent checks: + * committed golden vectors (tests/fixtures/gguf_dequant_golden.npz), generated + with llama.cpp's gguf-py reference on deterministic random blocks -- run in + CI with only numpy; + * if the gguf package is importable, a live comparison against gguf.quants. + dequantize on fresh random blocks, so drift from the reference is caught. + +Also covers the scalar types (F32/F16/BF16) and the refusal paths (unsupported +type, numel/byte-count mismatch). +""" + +import sys +import unittest +from pathlib import Path + +try: + import numpy as np +except ImportError as exc: + raise unittest.SkipTest("numpy not installed: %s" % exc) + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "tools")) + +import gguf_reader +from gguf_reader import GGUFError +from gguf_dequant import ( + dequantize, + dequantize_q2_K, + dequantize_q3_K, + dequantize_q4_K, + dequantize_q5_K, + dequantize_q6_K, + dequantize_q8_0, +) + +GOLDEN = Path(__file__).resolve().parent / "fixtures" / "gguf_dequant_golden.npz" + +BLOCK_CASES = [ + ("Q2_K", gguf_reader.GGML_TYPE_Q2_K, 84, dequantize_q2_K), + ("Q3_K", gguf_reader.GGML_TYPE_Q3_K, 110, dequantize_q3_K), + ("Q4_K", gguf_reader.GGML_TYPE_Q4_K, 144, dequantize_q4_K), + ("Q5_K", gguf_reader.GGML_TYPE_Q5_K, 176, dequantize_q5_K), + ("Q6_K", gguf_reader.GGML_TYPE_Q6_K, 210, dequantize_q6_K), + ("Q8_0", gguf_reader.GGML_TYPE_Q8_0, 34, dequantize_q8_0), +] + + +class GoldenDequantTest(unittest.TestCase): + def setUp(self): + if not GOLDEN.is_file(): + self.skipTest("golden fixture missing: %s" % GOLDEN) + self.golden = np.load(str(GOLDEN)) + + def test_block_types_match_golden(self): + for name, ggml_type, block_bytes, _func in BLOCK_CASES: + raw = self.golden["%s_raw" % name] + expected = self.golden["%s_expected" % name].reshape(-1) + got = dequantize(raw.tobytes(), ggml_type, expected.size) + self.assertEqual(got.shape, expected.shape) + np.testing.assert_allclose(got, expected, rtol=0, atol=0, equal_nan=True, + err_msg="%s dequant mismatch" % name) + + def test_dispatcher_uses_same_functions(self): + for name, ggml_type, _bb, func in BLOCK_CASES: + raw = self.golden["%s_raw" % name] + direct = func(raw).reshape(-1) + dispatched = dequantize(raw.tobytes(), ggml_type, direct.size) + np.testing.assert_array_equal(dispatched, direct) + + +class ScalarDequantTest(unittest.TestCase): + def test_f32(self): + values = np.array([0.0, 1.5, -2.25, 3.5e10], dtype=np.float32) + got = dequantize(values.tobytes(), gguf_reader.GGML_TYPE_F32, values.size) + np.testing.assert_array_equal(got, values) + + def test_f16(self): + values = np.array([0.0, 1.5, -2.25, 0.125], dtype=np.float16) + got = dequantize(values.tobytes(), gguf_reader.GGML_TYPE_F16, values.size) + np.testing.assert_array_equal(got, values.astype(np.float32)) + + def test_bf16(self): + u16 = np.array([0x0000, 0x3FC0, 0xC040, 0x3E00], dtype=np.uint16) + got = dequantize(u16.tobytes(), gguf_reader.GGML_TYPE_BF16, u16.size) + expected = (u16.astype(np.uint32) << 16).view(np.float32) + np.testing.assert_array_equal(got, expected) + + +class DequantRefusalTest(unittest.TestCase): + def test_unsupported_type(self): + with self.assertRaisesRegex(GGUFError, "unsupported ggml type"): + dequantize(b"\x00" * 16, 100, 16) + + def test_numel_not_multiple_of_block(self): + with self.assertRaisesRegex(GGUFError, "not a multiple of block"): + dequantize(b"\x00" * 84, gguf_reader.GGML_TYPE_Q2_K, 255) + + def test_byte_count_mismatch(self): + with self.assertRaisesRegex(GGUFError, "expected"): + dequantize(b"\x00" * 84, gguf_reader.GGML_TYPE_Q2_K, 512) + + +class LiveReferenceTest(unittest.TestCase): + def test_matches_gguf_py_on_random_blocks(self): + try: + import gguf + from gguf.quants import dequantize as reference + except ImportError as exc: + self.skipTest("gguf package not installed: %s" % exc) + + rng = np.random.default_rng(20260825) + mapping = { + "Q2_K": (gguf.GGMLQuantizationType.Q2_K, gguf_reader.GGML_TYPE_Q2_K, 84, 256), + "Q3_K": (gguf.GGMLQuantizationType.Q3_K, gguf_reader.GGML_TYPE_Q3_K, 110, 256), + "Q4_K": (gguf.GGMLQuantizationType.Q4_K, gguf_reader.GGML_TYPE_Q4_K, 144, 256), + "Q5_K": (gguf.GGMLQuantizationType.Q5_K, gguf_reader.GGML_TYPE_Q5_K, 176, 256), + "Q6_K": (gguf.GGMLQuantizationType.Q6_K, gguf_reader.GGML_TYPE_Q6_K, 210, 256), + "Q8_0": (gguf.GGMLQuantizationType.Q8_0, gguf_reader.GGML_TYPE_Q8_0, 34, 32), + } + for name, (ref_type, our_type, block_bytes, block_elems) in mapping.items(): + for _ in range(25): + nb = int(rng.integers(1, 5)) + raw = rng.integers(0, 256, size=nb * block_bytes, dtype=np.uint8) + expected = reference(raw.reshape(nb, block_bytes).copy(), ref_type).reshape(-1) + got = dequantize(raw.tobytes(), our_type, nb * block_elems) + np.testing.assert_allclose(got, expected, rtol=0, atol=0, equal_nan=True, + err_msg="%s live mismatch" % name) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_gguf_olmoe_profile.py b/c/tests/test_gguf_olmoe_profile.py new file mode 100644 index 000000000..42dd64043 --- /dev/null +++ b/c/tests/test_gguf_olmoe_profile.py @@ -0,0 +1,287 @@ +#!/usr/bin/env python3 +"""tools/gguf_olmoe_profile.py: OLMoE name mapping, layout and int8 quantization. + +The strongest check here is the quantizer comparison against +tools/convert_olmoe_merged.py, the torch code the existing OLMoE container was +built with. It runs in two forms: test_matches_reference_quantizer_golden uses +the committed fixture (tests/fixtures/olmoe_quantize_row_golden.npz), generated +once with that torch reference, so CI needs only numpy; the live variant +test_matches_reference_quantizer_live re-runs the torch reference when it is +installed. If the two disagree, the new GGUF converter would produce a container +the engine reads differently from the known-good one. +""" + +import sys +import unittest +from pathlib import Path + +try: + import numpy as np +except ImportError as exc: + raise unittest.SkipTest("numpy not installed: %s" % exc) + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "tools")) + +import gguf_olmoe_profile as profile + +GOLDEN_QUANT = (Path(__file__).resolve().parent / "fixtures" + / "olmoe_quantize_row_golden.npz") + + +class MappingTest(unittest.TestCase): + def test_parse_expert(self): + self.assertEqual(profile.parse_expert("blk.3.ffn_gate_exps.weight"), (3, "gate")) + self.assertEqual(profile.parse_expert("blk.0.ffn_down_exps.weight"), (0, "down")) + self.assertIsNone(profile.parse_expert("blk.0.attn_q.weight")) + self.assertIsNone(profile.parse_expert("token_embd.weight")) + + def test_dense_target(self): + self.assertEqual(profile.dense_target("token_embd.weight"), + "model.embed_tokens.weight") + self.assertEqual(profile.dense_target("output.weight"), "lm_head.weight") + self.assertEqual(profile.dense_target("output_norm.weight"), "model.norm.weight") + self.assertEqual(profile.dense_target("blk.5.attn_q.weight"), + "model.layers.5.self_attn.q_proj.weight") + self.assertEqual(profile.dense_target("blk.11.ffn_gate_inp.weight"), + "model.layers.11.mlp.gate.weight") + self.assertEqual(profile.dense_target("blk.0.ffn_norm.weight"), + "model.layers.0.post_attention_layernorm.weight") + self.assertIsNone(profile.dense_target("blk.0.some_unknown.weight")) + + +class LayoutTest(unittest.TestCase): + def test_to_logical_reverses_dims(self): + flat = np.arange(6, dtype=np.float32) + logical = profile.to_logical(flat, [3, 2]) + self.assertEqual(logical.shape, (2, 3)) + np.testing.assert_array_equal(logical, flat.reshape(2, 3)) + + def test_to_logical_1d_is_identity(self): + flat = np.arange(5, dtype=np.float32) + np.testing.assert_array_equal(profile.to_logical(flat, [5]), flat) + + def test_expert_matrix_shape(self): + self.assertEqual(profile.expert_matrix_shape([2048, 1024, 64]), (1024, 2048)) + self.assertEqual(profile.expert_matrix_shape([1024, 2048, 64]), (2048, 1024)) + + def test_restore_rope_layout_inverts_llama_permutation(self): + def forward(w, n_head, n_kv=None): + n = n_kv if (n_kv and n_head != n_kv) else n_head + return np.ascontiguousarray( + w.reshape((n, 2, w.shape[0] // n // 2) + w.shape[1:]).swapaxes(1, 2).reshape(w.shape)) + + rng = np.random.default_rng(11) + for (out, n_head, n_kv) in ((16, 4, 4), (24, 4, 2), (12, 2, None), (64, 4, None)): + hf = rng.normal(size=(out, 3)).astype(np.float32) + permuted = forward(hf, n_head, n_kv) + if out // (n_kv if (n_kv and n_head != n_kv) else n_head) // 2 != 2: + self.assertFalse(np.array_equal(permuted, hf)) + restored = profile.restore_rope_layout(permuted, n_head, n_kv) + np.testing.assert_array_equal(restored, hf) + + def test_restore_rope_layout_matches_torch_reference(self): + try: + import torch + except ImportError as exc: + self.skipTest("torch unavailable: %s" % exc) + for (out, n_head, n_kv) in ((16, 4, 4), (24, 4, 2), (12, 2, None)): + rng = np.random.default_rng(out) + hf = torch.from_numpy(rng.normal(size=(out, 7)).astype(np.float32)) + n = n_kv if (n_kv and n_head != n_kv) else n_head + permuted = (hf.reshape(n, 2, out // n // 2, *hf.shape[1:]) + .swapaxes(1, 2).reshape(hf.shape)) + restored = profile.restore_rope_layout(permuted.numpy(), n_head, n_kv) + np.testing.assert_array_equal(restored, hf.numpy()) + + +class QuantizationTest(unittest.TestCase): + def test_quantize_row_math(self): + w = np.array([[1.0, -2.0, 0.5], [0.0, 0.0, 0.0]], dtype=np.float32) + q, s = profile.quantize_row(w) + self.assertEqual(q.dtype, np.int8) + self.assertEqual(q.shape, (2, 3)) + self.assertEqual(s.shape, (2,)) + # row 0: absmax 2 -> scale 2/127, q = round(w/scale) + self.assertAlmostEqual(float(s[0]), 2.0 / 127.0, places=7) + expected = np.clip(np.rint(w[0] / s[0]), -128, 127).astype(np.int8) + np.testing.assert_array_equal(q[0], expected) + # reconstruction W ~= q*scale + np.testing.assert_allclose(q.astype(np.float32) * s[:, None], w, atol=2.0 / 127.0) + + def test_quantize_row_rejects_non_2d(self): + with self.assertRaises(ValueError): + profile.quantize_row(np.zeros(8, dtype=np.float32)) + + def test_matches_reference_quantizer_golden(self): + """quantize_row pinned to the torch reference via a committed fixture. + + The .npz holds the deterministic inputs plus convert_olmoe_merged.py's + int8 weights and scales (seed 20260825, 20 variable-shape cases), so CI + runs this check with only numpy installed. + """ + if not GOLDEN_QUANT.is_file(): + self.skipTest("golden fixture missing: %s" % GOLDEN_QUANT) + golden = np.load(str(GOLDEN_QUANT)) + for i in range(int(golden["count"])): + w = golden["case_%02d_w" % i] + q_got, s_got = profile.quantize_row(w) + np.testing.assert_array_equal( + q_got, golden["case_%02d_q" % i], err_msg="case %d int8" % i) + np.testing.assert_allclose( + s_got, golden["case_%02d_s" % i], rtol=0, atol=0, + err_msg="case %d scales" % i) + + def test_matches_reference_quantizer_live(self): + try: + import torch + import convert_olmoe_merged as reference + except ImportError as exc: + self.skipTest("torch/convert_olmoe_merged unavailable: %s" % exc) + + golden = np.load(str(GOLDEN_QUANT)) if GOLDEN_QUANT.is_file() else None + rng = np.random.default_rng(20260825) + for case in range(20): + rows = int(rng.integers(1, 8)) + cols = int(rng.integers(1, 40)) + w = rng.normal(0.0, 1.0, size=(rows, cols)).astype(np.float32) + if golden is not None: + # the committed fixture and the live reference must agree + np.testing.assert_array_equal(w, golden["case_%02d_w" % case]) + q_ref, s_ref = reference.quantize_row(torch.from_numpy(w)) + q_got, s_got = profile.quantize_row(w) + np.testing.assert_array_equal(q_got, q_ref.cpu().numpy()) + np.testing.assert_allclose(s_got, s_ref.cpu().numpy(), rtol=0, atol=0) + + def test_merge_expert_layout(self): + rng = np.random.default_rng(7) + hidden, inter = 4, 6 + gate = rng.normal(0, 1, (inter, hidden)).astype(np.float32) + up = rng.normal(0, 1, (inter, hidden)).astype(np.float32) + down = rng.normal(0, 1, (hidden, inter)).astype(np.float32) + merged, scales = profile.merge_expert(gate, up, down) + + qg, sg = profile.quantize_row(gate) + qu, su = profile.quantize_row(up) + qd, sd = profile.quantize_row(down) + np.testing.assert_array_equal( + merged, np.concatenate((qg.reshape(-1), qu.reshape(-1), qd.reshape(-1)))) + np.testing.assert_array_equal( + scales, np.concatenate((sg, su, sd))) + self.assertEqual(merged.dtype, np.int8) + self.assertEqual(scales.dtype, np.float32) + self.assertEqual(merged.size, inter * hidden * 2 + hidden * inter) + self.assertEqual(scales.size, inter + inter + hidden) + + +class ConfigTest(unittest.TestCase): + def _meta(self): + return { + "olmoe.embedding_length": 2048, + "olmoe.block_count": 16, + "olmoe.attention.head_count": 16, + "olmoe.attention.head_count_kv": 16, + "olmoe.expert_count": 64, + "olmoe.expert_used_count": 8, + "olmoe.feed_forward_length": 1024, + "olmoe.rope.freq_base": 10000.0, + "olmoe.attention.layer_norm_rms_epsilon": 1e-5, + "tokenizer.ggml.tokens": ["a"] * 50304, + } + + def test_build_config(self): + config = profile.build_config(self._meta()) + self.assertEqual(config["hidden_size"], 2048) + self.assertEqual(config["num_hidden_layers"], 16) + self.assertEqual(config["num_attention_heads"], 16) + self.assertEqual(config["num_key_value_heads"], 16) + self.assertEqual(config["num_experts"], 64) + self.assertEqual(config["num_experts_per_tok"], 8) + self.assertEqual(config["intermediate_size"], 1024) + self.assertEqual(config["vocab_size"], 50304) + self.assertFalse(config["norm_topk_prob"]) + self.assertEqual(config["model_type"], "olmoe") + self.assertEqual(config["architectures"], ["OlmoeForCausalLM"]) + + def test_build_config_norm_topk_prob_override(self): + config = profile.build_config(self._meta(), norm_topk_prob=True) + self.assertTrue(config["norm_topk_prob"]) + + def test_build_config_missing_key(self): + meta = self._meta() + del meta["olmoe.expert_count"] + with self.assertRaises(ValueError): + profile.build_config(meta) + + def test_build_config_missing_tokens(self): + meta = self._meta() + del meta["tokenizer.ggml.tokens"] + with self.assertRaises(ValueError): + profile.build_config(meta) + + +class TokenizerTest(unittest.TestCase): + def _meta(self): + return { + "tokenizer.ggml.tokens": ["hello", " world", "", "Ġt", "a"], + "tokenizer.ggml.merges": ["Ġ t", "Ġ Ġ"], + "tokenizer.ggml.token_type": [1, 1, 3, 4, 6], + "tokenizer.ggml.bos_token_id": 2, + "tokenizer.ggml.eos_token_id": 2, + "tokenizer.ggml.padding_token_id": 0, + "tokenizer.ggml.add_bos_token": False, + "tokenizer.ggml.add_eos_token": False, + "tokenizer.chat_template": "{{ bos_token }}...", + } + + def test_vocab_and_merges(self): + tokenizer = profile.build_tokenizer(self._meta()) + self.assertEqual(tokenizer["model"]["type"], "BPE") + self.assertEqual(tokenizer["model"]["vocab"], { + "hello": 0, " world": 1, "": 2, "Ġt": 3, "a": 4}) + self.assertEqual(tokenizer["model"]["merges"], ["Ġ t", "Ġ Ġ"]) + + def test_added_tokens_only_special_and_user_defined(self): + tokenizer = profile.build_tokenizer(self._meta()) + self.assertEqual(tokenizer["added_tokens"], [ + {"id": 2, "content": "", "special": True}, + {"id": 3, "content": "Ġt", "special": False}, + ]) + + def test_special_ids(self): + tokenizer = profile.build_tokenizer(self._meta()) + self.assertEqual(tokenizer["bos_token"], "") + self.assertEqual(tokenizer["eos_token"], "") + self.assertEqual(tokenizer["pad_token"], "hello") + self.assertFalse(tokenizer["add_bos_token"]) + self.assertFalse(tokenizer["add_eos_token"]) + self.assertEqual(tokenizer["chat_template"], "{{ bos_token }}...") + + def test_missing_tokens_raises(self): + meta = self._meta() + del meta["tokenizer.ggml.tokens"] + with self.assertRaises(ValueError): + profile.build_tokenizer(meta) + + def test_byte_token_not_added(self): + meta = self._meta() + meta["tokenizer.ggml.token_type"] = [6, 6, 6, 6, 6] + tokenizer = profile.build_tokenizer(meta) + self.assertEqual(tokenizer["added_tokens"], []) + + def test_missing_token_type_defaults_to_normal(self): + meta = self._meta() + del meta["tokenizer.ggml.token_type"] + tokenizer = profile.build_tokenizer(meta) + self.assertEqual(tokenizer["added_tokens"], []) + + def test_partial_token_type_defaults_to_normal(self): + meta = self._meta() + meta["tokenizer.ggml.token_type"] = [3] # only the first token is typed + tokenizer = profile.build_tokenizer(meta) + self.assertEqual(tokenizer["added_tokens"], [ + {"id": 0, "content": "hello", "special": True}, + ]) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_gguf_reader.py b/c/tests/test_gguf_reader.py new file mode 100644 index 000000000..c4e6c463c --- /dev/null +++ b/c/tests/test_gguf_reader.py @@ -0,0 +1,322 @@ +#!/usr/bin/env python3 +"""tools/gguf_reader.py + tests/gguf_fixture.py: pure-python GGUF round trip. + +Builds a deterministic GGUF fixture in a temp directory and asserts the reader +recovers every metadata value type (scalars, bool, string, flat arrays, nested +array), every tensor header (name/dims/type) and every tensor payload +byte-for-byte. Then exercises the refusal paths an untrusted-container reader +must hit loudly: bad magic, unsupported version, truncated file, duplicate +tensor names, unaligned tensor offset, tensor range beyond EOF, unsupported +ggml type, rank above cap, missing/invalid/honored alignment, malformed +metadata values and capped counts. No third-party dependency. +""" + +import os +import struct +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "tools")) +sys.path.insert(0, str(Path(__file__).resolve().parent)) + +import gguf_reader +from gguf_reader import GGUFError, GGUFReader +from gguf_fixture import ( + METADATA_ARRAY, + METADATA_BOOL, + METADATA_FLOAT32, + METADATA_STRING, + METADATA_UINT32, + _encode_string, + _encode_value, + write_gguf_fixture, +) + + +def _raw_header(magic=b"GGUF", version=3, tensor_count=0, kv_count=0): + return magic + struct.pack(" int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--template", type=Path, required=True) + arguments = parser.parse_args() + + if not arguments.template.exists(): + print(f"SKIP: manca {arguments.template}; il riferimento non c'e' e " + f"questo test non ha verificato nulla") + return 2 # salto != successo (vedi docstring) + try: + import jinja2 # noqa: F401 + except ImportError: + print("SKIP: jinja2 non installato; senza non c'e' riferimento") + return 2 # salto != successo (vedi docstring) + + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + import openai_server + + openai_server.ARCH = "glm" + template_text = arguments.template.read_text(encoding="utf-8") + + failures = 0 + for label, case in CASES.items(): + theirs = reference(template_text, messages=case["messages"]) + ours = openai_server.render_chat(case["messages"], enable_thinking=False) + if ours == theirs: + print(f"ok {label}") + else: + show(label, ours, theirs) + failures += 1 + + # Prosecuzione: l'ultimo turno assistant e' da CONTINUARE. GLM non ha un terminatore di + # turno (e' il token di ruolo seguente a chiudere), quindi la forma aperta e' il template + # con add_generation_prompt=False cosi' com'e' -- niente da togliere, a differenza di + # ChatML. Il prompt finisce su <|assistant|>{contenuto}. + aperto = [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + produced = openai_server.render_chat_for_arch(aperto, enable_thinking=False, + add_generation_prompt=False) + expected = reference(template_text, messages=aperto, add_generation_prompt=False) + if produced == expected: + print("ok prosecuzione: turno aperto = template(add_generation_prompt=False)") + else: + show("prosecuzione", produced, expected) + failures += 1 + if not produced.endswith("La capitale e'"): + print(f"FAIL prosecuzione: il prompt non finisce sull'apertura del client: " + f"{produced[-60:]!r}") + failures += 1 + # Controllo negativo: col ramo normale lo stesso scambio DEVE finire sulla cue. + if not openai_server.render_chat_for_arch( + aperto, enable_thinking=False).endswith("<|assistant|>"): + print("FAIL prosecuzione: il ramo normale non emette piu' il prompt di generazione") + failures += 1 + + print() + if failures: + print(f"TEST FAIL ({failures} casi)") + return 1 + print("template GLM-5.2: il gateway e' identico al riferimento (ragionamento spento)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/c/tests/test_glm53_oracles.py b/c/tests/test_glm53_oracles.py new file mode 100644 index 000000000..13226029b --- /dev/null +++ b/c/tests/test_glm53_oracles.py @@ -0,0 +1,82 @@ +"""The GLM-5.3 harnesses, seen by unittest. + +tests/glm53_*_harness.py are argparse programs, not unittest modules. Under +their old test_*.py names `make test-python` imported them, found no TestCase +and counted nothing, so the suite stayed green while the chat template, serve, +streaming, vision and Vulkan paths were never exercised (#1700). + +Two things are checked here. Everywhere, with the standard library only: a +harness that cannot find what it needs exits 2, never 0, so a skip cannot pass +for a verification. And the two stdlib-only oracles run against their fixtures +when GLM53_TINY (tools/make_glm53_tiny.py) and GLM53_MM_TINY +(tools/make_glm53_multimodal_tiny.py) point at them, as the GLM-5.3 CI job +does; without them those two tests are skipped with the reason on the record. +""" +import os +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent +BINARY = next((HERE.parent / name for name in ("glm53", "glm53.exe") + if (HERE.parent / name).exists()), None) + + +def harness(name, *args): + return subprocess.run( + [sys.executable, str(HERE / f"glm53_{name}_harness.py"), *args], + capture_output=True, text=True, timeout=600) + + +class Glm53HarnessSkipTest(unittest.TestCase): + def test_missing_input_exits_2(self): + """Every harness, pointed at a fixture that is not there, says SKIP + and exits 2. The binary is never reached, so none is needed.""" + with tempfile.TemporaryDirectory() as empty: + missing = str(Path(empty, "missing")) + cases = { + "chat_template": ["--template", missing], + "tiny": ["--binary", "glm53", "--fixture", missing], + "multimodal_tiny": ["--binary", "glm53", "--fixture", missing], + "serve": ["--binary", "glm53", "--fixture", missing], + "streaming": ["--binary", "glm53", "--quantized", missing, + "--dequantized", missing], + "vision_serve": ["--binary", "glm53", "--fixture", missing], + "vulkan": ["--binary", "glm53", "--fixture", missing], + } + self.assertEqual(sorted(cases), sorted( + path.name[len("glm53_"):-len("_harness.py")] + for path in HERE.glob("glm53_*_harness.py")), + "a harness was added or renamed: give it a case here") + for name, args in cases.items(): + with self.subTest(harness=name): + result = harness(name, *args) + self.assertEqual(result.returncode, 2, result.stdout + result.stderr) + self.assertIn("SKIP", result.stdout) + + +class Glm53TinyOracleTest(unittest.TestCase): + def run_oracle(self, name, variable): + fixture = os.environ.get(variable) + if not fixture: + self.skipTest(f"{variable} not set to a fixture") + # Set but unusable is a failure, not a skip: whoever set it expected a run. + self.assertIsNotNone(BINARY, f"{variable} is set but glm53 is not built") + result = harness(name, "--binary", str(BINARY), "--fixture", fixture) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertIn("PASS", result.stdout) + + def test_text_oracle(self): + """KDA, DSA, mHC, dense FFN and routed MoE, token-exact against the + transformers reference in f32.""" + self.run_oracle("tiny", "GLM53_TINY") + + def test_multimodal_oracle(self): + """The vision tower and the image tokens in the prompt, token-exact.""" + self.run_oracle("multimodal_tiny", "GLM53_MM_TINY") + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_glm_oracle.py b/c/tests/test_glm_oracle.py new file mode 100644 index 000000000..90c8325fd --- /dev/null +++ b/c/tests/test_glm_oracle.py @@ -0,0 +1,287 @@ +"""Real-process GLM oracle exit-status regression; stdlib only. + +Discovery skips when the generated fixture is absent. Direct execution (as in +CI) requires it, so a missing engine/fixture cannot silently pass the gate. +""" +import argparse +import copy +import json +import os +from pathlib import Path +import re +import shutil +import struct +import subprocess +import tempfile +import unittest + +C_DIR = Path(__file__).resolve().parents[1] +ENGINE = C_DIR / ("colibri.exe" if os.name == "nt" else "colibri") +SNAP = C_DIR / "glm_tiny" +REFERENCE = C_DIR / "ref_glm.json" + + +class GlmOracleTest(unittest.TestCase): + @classmethod + def setUpClass(cls): + if not (ENGINE.is_file() and (SNAP / "model.safetensors").is_file() + and REFERENCE.is_file()): + raise unittest.SkipTest("build colibri and generate the GLM tiny oracle first") + cls.reference = json.loads(REFERENCE.read_text(encoding="utf-8")) + cls.vocab = json.loads((SNAP / "config.json").read_text(encoding="utf-8"))["vocab_size"] + # Keep the independent fixture for the numerical gate. Use observed TF + # predictions only to place exit-status tests at exact mismatch counts, + # even on a host with one or two floating-point near ties. + cls.tf_reference = cls.observed_tf_reference(cls.reference) + + @classmethod + def run_oracle(cls, reference=None, *, tf=True, strict="1", raw=None, + snapshot=None, extra_env=None): + # Do not inherit serving/sampling/offload knobs from an interactive shell. + keep = {"PATH", "SYSTEMROOT", "WINDIR", "TEMP", "TMP", "HOME", "USERPROFILE", + "LD_LIBRARY_PATH", "DYLD_LIBRARY_PATH", "ASAN_OPTIONS", "UBSAN_OPTIONS"} + env = {k: v for k, v in os.environ.items() if k in keep} + env.update(SNAP=str(SNAP if snapshot is None else snapshot), COLI_TEMP="0", OMP_NUM_THREADS="4") + if tf: + env["TF"] = "1" + if strict is not None: + env["ORACLE_STRICT"] = strict + env.update(extra_env or {}) + with tempfile.TemporaryDirectory(prefix="colibri-oracle-") as tmp: + path = Path(tmp) / "ref.json" + path.write_text(raw if raw is not None else json.dumps( + cls.reference if reference is None else reference), encoding="utf-8") + env["REF"] = str(path) + return subprocess.run([str(ENGINE), "64", "16", "16"], cwd=C_DIR, + env=env, capture_output=True, text=True, + encoding="utf-8", errors="replace", timeout=60) + + @classmethod + def observed_tf_reference(cls, reference): + result = cls.run_oracle(reference, strict="0") + if result.returncode: + raise AssertionError(result.stdout + result.stderr) + score = re.search(r"C vs oracle: (\d+)/(\d+) positions", result.stdout) + if score is None or int(score[2]) != len(reference["tf_pred"]): + raise AssertionError("missing TF score: " + result.stdout + result.stderr) + ref = copy.deepcopy(reference) + mismatches = re.findall(r"\[ORACLE\] mismatch pos=(\d+) expected=(\d+) got=(-?\d+)", result.stderr) + if len(mismatches) != int(score[2]) - int(score[1]): + raise AssertionError("incomplete TF diagnostics: " + result.stderr) + for pos, expected, got in mismatches: + pos, expected, got = int(pos), int(expected), int(got) + if not (0 <= pos < len(ref["tf_pred"]) and 0 <= got < cls.vocab + and ref["tf_pred"][pos] == expected): + raise AssertionError("invalid TF prediction: " + result.stderr) + ref["tf_pred"][pos] = got + return ref + + def assert_failed(self, result): + self.assertEqual(result.returncode, 1, result.stdout + result.stderr) + + def corrupted(self, tf, count=1): + ref = copy.deepcopy(self.tf_reference if tf else self.reference) + if tf: + for i in range(count): + ref["tf_pred"][i] = (ref["tf_pred"][i] + 1) % self.vocab + else: + ref["full_ids"][-1] = (ref["full_ids"][-1] + 1) % self.vocab + return ref + + def test_valid_reference_passes_both_modes(self): + for tf in (True, False): + with self.subTest(tf=tf): + result = self.run_oracle(tf=tf, extra_env={"ORACLE_TF_MAX_MISMATCHES": "2"}) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + total = len(self.reference["full_ids"]) + new = total - len(self.reference["prompt_ids"]) + if tf: + score = re.search(r"C vs oracle: (\d+)/(\d+) positions", result.stdout) + self.assertIsNotNone(score, result.stdout) + self.assertEqual(int(score[2]), total) + self.assertGreaterEqual(int(score[1]), total - 2) + else: + self.assertIn(f"Matching tokens: {new}/{new}", result.stdout) + + def test_exact_mode_rejects_one_wrong_prediction(self): + for tf in (True, False): + with self.subTest(tf=tf): + self.assert_failed(self.run_oracle(self.corrupted(tf), tf=tf)) + + def test_teacher_forcing_allowance_boundary(self): + total = len(self.reference["tf_pred"]) + for wrong in (0, 1, 2, 3): + with self.subTest(wrong=wrong): + result = self.run_oracle(self.corrupted(True, wrong), + extra_env={"ORACLE_TF_MAX_MISMATCHES": "2"}) + self.assertEqual(result.returncode, int(wrong > 2), result.stdout + result.stderr) + self.assertIn(f"{total - wrong}/{total} positions", result.stdout) + + def test_explicit_zero_allowance_is_exact(self): + for wrong in (0, 1): + with self.subTest(wrong=wrong): + result = self.run_oracle(self.corrupted(True, wrong), + extra_env={"ORACLE_TF_MAX_MISMATCHES": "0"}) + self.assertEqual(result.returncode, wrong, result.stdout + result.stderr) + + def test_teacher_forcing_allowance_does_not_relax_greedy(self): + self.assert_failed(self.run_oracle(self.corrupted(False), tf=False, + extra_env={"ORACLE_TF_MAX_MISMATCHES": "2"})) + + def test_invalid_allowance_fails_before_comparison(self): + total = len(self.reference["tf_pred"]) + for value in ("", "-1", "+2", " 2", "2 ", "2x", "2.0", str(total), str(total + 1), "9" * 40): + with self.subTest(value=value): + result = self.run_oracle(extra_env={"ORACLE_TF_MAX_MISMATCHES": value}) + self.assert_failed(result) + self.assertIn("ORACLE_TF_MAX_MISMATCHES", result.stderr) + self.assertNotIn("C vs oracle:", result.stdout) + + def test_diagnostic_mode_keeps_report_only_exit_status(self): + for tf in (True, False): + for strict in (None, "0"): + with self.subTest(tf=tf, strict=strict): + result = self.run_oracle(self.corrupted(tf, 3), tf=tf, strict=strict, + extra_env={"ORACLE_TF_MAX_MISMATCHES": "2"}) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + + def test_missing_short_and_long_tf_predictions_fail_cleanly(self): + for value in (None, [], self.reference["tf_pred"][:-1], + self.reference["tf_pred"] + [0], "not an array"): + with self.subTest(value=value): + ref = copy.deepcopy(self.reference) + if value is None: + del ref["tf_pred"] + else: + ref["tf_pred"] = value + self.assert_failed(self.run_oracle(ref, extra_env={"ORACLE_TF_MAX_MISMATCHES": "2"})) + + def test_invalid_token_ids_fail_before_inference(self): + for field in ("prompt_ids", "full_ids", "tf_pred"): + for value in (-1, self.vocab, 1.5, True, "1", None, 1e100): + with self.subTest(field=field, value=value): + ref = copy.deepcopy(self.reference) + ref[field][0] = value + result = self.run_oracle(ref) + self.assert_failed(result) + self.assertIn("[ORACLE]", result.stderr) + + def test_truncated_json_and_mismatched_prompt_fail(self): + raw = json.dumps(self.reference) + for invalid in (raw[:-1], raw[:-2], raw + " garbage", raw + "\0garbage"): + with self.subTest(raw=invalid[-30:]): + self.assert_failed(self.run_oracle(raw=invalid)) + ref = copy.deepcopy(self.reference) + ref["prompt_ids"][0] = (ref["prompt_ids"][0] + 1) % self.vocab + self.assert_failed(self.run_oracle(ref, tf=False)) + + def test_empty_continuation_fails(self): + ref = copy.deepcopy(self.reference) + ref["full_ids"] = ref["prompt_ids"][:] + self.assert_failed(self.run_oracle(ref, tf=False)) + + def test_escaped_nul_cannot_alias_reference_keys(self): + for field in ("prompt_ids", "full_ids", "tf_pred"): + ref = copy.deepcopy(self.reference) + ref[field + "\0ignored"] = ref.pop(field) + for tf in (True, False): + for strict in ("1", None): + with self.subTest(field=field, tf=tf, strict=strict): + self.assert_failed(self.run_oracle(ref, tf=tf, strict=strict)) + + def test_non_json_whitespace_fails(self): + raw = json.dumps(self.reference) + for whitespace in ("\f", "\v"): + for invalid in (whitespace + raw, raw.replace(":", ":" + whitespace, 1), + raw.replace("[", "[" + whitespace, 1), raw + whitespace): + for tf in (True, False): + for strict in ("1", None): + with self.subTest(raw=invalid, tf=tf, strict=strict): + self.assert_failed(self.run_oracle(raw=invalid, tf=tf, strict=strict)) + + def test_other_modes_cannot_bypass_strict_comparison(self): + for mode in ("REPLAY", "CONSIST", "SERVE", "SCORE", "ABLATE_SCORE", "EXPERT_WORKER", + "I4_ACC512_TEST", "I3_AVX512_TEST", "COLI_ANS_PACK", "COLI_PROMPT"): + with self.subTest(mode=mode): + result = self.run_oracle(extra_env={mode: "1"}) + self.assert_failed(result) + self.assertIn("ORACLE_STRICT", result.stderr) + + def test_reference_diagnostics_remain_available_without_strict(self): + # REPLAY and CONSIST use the reference token sequence, not predictions, + # and retain their upstream precedence over TF when both are selected. + ref = copy.deepcopy(self.reference) + del ref["tf_pred"] + for mode, marker in (("REPLAY", "REPLAY decode:"), ("CONSIST", "CONSIST OK")): + for strict in (None, "0"): + for tf in (False, True): + with self.subTest(mode=mode, strict=strict, tf=tf): + result = self.run_oracle(ref, tf=tf, strict=strict, + extra_env={mode: "1"}) + self.assertEqual(result.returncode, 0, result.stdout + result.stderr) + self.assertIn(marker, result.stdout) + self.assertNotIn("C vs oracle:", result.stdout) + self.assertNotIn("Matching tokens:", result.stdout) + + def test_nonfinite_model_logits_fail_both_modes(self): + # The tiny generator stores an F32 lm_head in one safetensors shard. + # Poison a single output row, including when it would not win argmax. + with tempfile.TemporaryDirectory(prefix="colibri-oracle-weights-") as tmp: + snapshot = Path(tmp) + shutil.copy2(SNAP / "config.json", snapshot / "config.json") + weights = (SNAP / "model.safetensors").read_bytes() + header_size = struct.unpack_from(" #include #include -#include +#if !defined(__HIPCC__) +#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */ +#endif #include "../backend_cuda.cu" diff --git a/c/tests/test_gsgemv.c b/c/tests/test_gsgemv.c new file mode 100644 index 000000000..f5c8ef7bd --- /dev/null +++ b/c/tests/test_gsgemv.c @@ -0,0 +1,206 @@ +/* ============================================================================ + * test_gsgemv.c — standalone unit tests for the group-scaled int8 GEMV + * (gsgemv.h: matmul_q_gs) that qwen36 runs for every expert matmul. + * No model, no weights: links the SAME gsgemv.h the engine links. + * + * gsgemv.h has three tiers: AVX2/FMA (row-interleaved), SSE4.1 (row- + * interleaved, routed through sse41_kernels.h) and scalar. The AVX2 and + * scalar tiers must keep their exact sequence of float operations -- float + * addition is not associative, and this engine's token stream is required + * to stay byte-identical to the reference. reference_gs_gemv below is a + * VERBATIM copy of the kernel as it shipped before either the AVX2 + * row-interleave or the SSE4.1 tier existed, and the AVX2/scalar properties + * compare the live kernel against it with memcmp on raw float bits. The + * SSE4.1 tier has no pre-existing byte-identical output to match -- its + * reduction tree is a genuinely different shape -- so it alone is checked + * by max relative+absolute error instead. + * + * P1 EXPERT SHAPES — the two shapes the engine actually runs (I=2048,O=512 + * for gate/up and I=512,O=2048 for down, gs=64) match. + * P2 ROW TAIL — O not a multiple of the unroll width still matches, so the + * tail rows cannot quietly accumulate in a different order. + * P3 SCALAR PATH — gs not a multiple of 32 (AVX2) / 16 (SSE4.1) takes the + * scalar fallback. The dispatch is a runtime property of gs, so that + * path needs the same gate. + * P4 PARTIAL GROUP — I not a multiple of gs leaves a short final group. + * P5 SHORT FINAL GROUP — a final group of fewer than 16 (AVX2) / 8 (SSE4.1) + * elements. Both vector paths step through a group in fixed-width chunks + * and DROP such a remainder. That is pre-existing, dormant-in-this-model + * behaviour (I is 2048 or 512 and gs is 64, so the remainder is always + * zero), but it is pinned here so any future restructure reproduces it + * rather than silently correcting it -- a correction would change output, + * which is exactly what byte-identical tiers must not do. The SSE4.1 + * tier's own tail-drop width (<8, half the AVX2 tier's <16) means + * reference_gs_gemv needs its OWN SSE4.1-shaped branch below: on an + * SSE4.1-only build, without that branch the reference would fall through + * to the undropped scalar path and P5 could never pass. + * Exit 0 = all pass. + * ==========================================================================*/ +#include +#include +#include +#include +#include +#if defined(__AVX2__) && defined(__FMA__) +#include +#endif + +#include "../gsgemv.h" + +static int fails = 0; +#define CHECK(cond, msg) do { \ + if (cond) { printf(" ok %s\n", msg); } \ + else { printf(" FAIL %s\n", msg); fails++; } \ +} while (0) + +/* Deterministic across libc implementations, unlike rand(). */ +static uint32_t rng_state = 0x9E3779B9u; +static uint32_t rng_next(void) { rng_state = rng_state * 1664525u + 1013904223u; return rng_state; } +static float rng_f32(void) { return (float)((int32_t)(rng_next() >> 8) - 8388608) / 8388608.0f; } + +/* --------------------------------------------------------------------------- + * VERBATIM copy of matmul_q_gs as it shipped before the SSE4.1 tier was + * added. Do not tidy, reformat or "improve" this: its value is that it is + * the old arithmetic, character for character. If this drifts, the test + * proves nothing. + * ------------------------------------------------------------------------- */ +static void reference_gs_gemv(float *y, const float *x, const int8_t *q, const float *scale, + int I, int O, int gs) { + int ng = (I + gs - 1) / gs; +#if defined(__AVX2__) && defined(__FMA__) + if ((gs & 31) == 0) { + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + const float *sc = scale + (int64_t)o * ng; + float acc = 0.f; + for (int gi = 0; gi < ng; gi++) { + __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps(); + int base = gi * gs, end = base + gs; if (end > I) end = I; + for (int i = base; i + 16 <= end; i += 16) { + __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); + a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0); + a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1); + } + a0 = _mm256_add_ps(a0, a1); + __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1)); + s = _mm_add_ps(s, _mm_movehl_ps(s,s)); + s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1)); + /* fmaf, not `acc += x * sc[gi]`: gsgemv.h's real AVX2 tier + * (gs_group_sum + fmaf) fuses this step explicitly, so an + * implicit multiply-add here only matches it when the compiler + * happens to auto-contract -- true at -O2/-O3 with the default + * -ffp-contract=fast, false at -O1 or with contraction off + * (confirmed: -O3 -ffp-contract=off reproduces the mismatch). + * Matching the explicit fmaf makes the comparison exact + * regardless of optimisation flags, instead of accidentally so. */ + acc = fmaf(_mm_cvtss_f32(s), sc[gi], acc); + } + y[o] = acc; + } + return; + } +#elif defined(__SSE4_1__) + /* Mirrors the SSE4.1 kernel's intentional <8 tail-drop so P5 pins that + * behaviour instead of comparing against a reference that always sums + * the full remainder. */ + if ((gs & 15) == 0) { + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + const float *sc = scale + (int64_t)o * ng; + float acc = 0.f; + for (int gi = 0; gi < ng; gi++) { + int base = gi * gs, end = base + gs; if (end > I) end = I; + float part = 0.f; + for (int i = base; i + 8 <= end; i += 8) + for (int k = 0; k < 8; k++) part += x[i+k] * (float)w[i+k]; + acc += part * sc[gi]; + } + y[o] = acc; + } + return; + } +#endif + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + const float *sc = scale + (int64_t)o * ng; + float acc = 0.f; + for (int gi = 0; gi < ng; gi++) { + int base = gi * gs, end = base + gs; if (end > I) end = I; + float part = 0.f; + for (int i = base; i < end; i++) part += x[i] * (float)w[i]; + acc += part * sc[gi]; + } + y[o] = acc; + } +} + +/* Fills one random case, runs both kernels, compares the results. + * The AVX2 and scalar tiers stay exact: memcmp on raw float bits, per the + * engine's byte-identical requirement. The SSE4.1 tier's tree reduction is + * not bit-reproducible against the scalar reference -- float addition is + * not associative -- so that tier alone is checked by max relative error. */ +static int rows_match(int I, int O, int gs) { + int ng = (I + gs - 1) / gs; + float *x = malloc((size_t)I * sizeof *x); + int8_t *q = malloc((size_t)O * I); + float *sc = malloc((size_t)O * ng * sizeof *sc); + float *y = malloc((size_t)O * sizeof *y); + float *ref = malloc((size_t)O * sizeof *ref); + if (!x || !q || !sc || !y || !ref) { fprintf(stderr, "out of memory\n"); exit(2); } + + for (int i = 0; i < I; i++) x[i] = rng_f32(); + for (int64_t i = 0; i < (int64_t)O * I; i++) q[i] = (int8_t)(rng_next() >> 24); + for (int i = 0; i < O * ng; i++) sc[i] = rng_f32() * 0.01f; + + /* Poison both outputs differently: a kernel that skips a row must not pass + * by leaving stale bytes that happen to agree. */ + memset(y, 0x5A, (size_t)O * sizeof *y); + memset(ref, 0xA5, (size_t)O * sizeof *ref); + + reference_gs_gemv(ref, x, q, sc, I, O, gs); + matmul_q_gs(y, x, q, sc, I, O, gs); + +#if defined(__SSE4_1__) && !(defined(__AVX2__) && defined(__FMA__)) + /* No FMA and a different tree reduction than the scalar reference: pure + * relative error blows up on near-zero-ref rows (cancellation), so use a + * combined absolute+relative bound instead. */ + int equal = 1; + for (int o = 0; o < O; o++) { + float diff = fabsf(y[o] - ref[o]); + if (diff > 1e-5f + 1e-4f * fabsf(ref[o])) { equal = 0; break; } + } +#else + int equal = memcmp(y, ref, (size_t)O * sizeof *y) == 0; +#endif + free(x); free(q); free(sc); free(y); free(ref); + return equal; +} + +int main(void) { + printf("test_gsgemv: group-scaled int8 GEMV matches its reference\n"); + + /* ---- P1: the shapes the engine actually runs ------------------------- */ + CHECK(rows_match(2048, 512, 64), "P1 gate/up shape (I=2048, O=512, gs=64) matches"); + CHECK(rows_match(512, 2048, 64), "P1 down shape (I=512, O=2048, gs=64) matches"); + + /* ---- P2: O not a multiple of the unroll width ------------------------ */ + CHECK(rows_match(2048, 515, 64), "P2 O=515 (tail rows) matches"); + CHECK(rows_match(2048, 6, 64), "P2 O=6 (fewer rows than one unrolled block) matches"); + CHECK(rows_match(2048, 1, 64), "P2 O=1 (single row) matches"); + + /* ---- P3: gs not a multiple of 32/16 takes the scalar fallback -------- */ + CHECK(rows_match(2048, 512, 48), "P3 gs=48 (scalar fallback path) matches"); + CHECK(rows_match(2000, 517, 48), "P3 scalar fallback with ragged I and O matches"); + + /* ---- P4: I not a multiple of gs leaves a short final group ----------- */ + CHECK(rows_match(2000, 512, 64), "P4 I=2000 (final group of 16) matches"); + + /* ---- P5: final group under the vector width, dropped by AVX2/SSE4.1 -- */ + CHECK(rows_match(1990, 512, 64), "P5 I=1990 (final group of 6, dropped) reproduces the old result"); + + printf("\n%s (%d failure%s)\n", fails ? "TEST FAIL" : "ALL PASS", fails, fails == 1 ? "" : "s"); + return fails ? 1 : 0; +} diff --git a/c/tests/test_i4_grouped.c b/c/tests/test_i4_grouped.c index 04e8d8877..f76c8918b 100644 --- a/c/tests/test_i4_grouped.c +++ b/c/tests/test_i4_grouped.c @@ -12,11 +12,9 @@ * nibble edges 0 and 15 (which decode to -8 and +7 — an offset encoding, NOT * two's complement; getting this backwards is silent and looks like noise). * - * FP note: the kernel sums each group in f32 (AVX2 accumulator + scalar tail) - * while the reference sums in double, so we compare against a relative epsilon - * rather than bit-exactly. The tolerance is tight enough that a wrong scale - * index, a wrong group boundary or a swapped nibble cannot hide under it — - * those are O(1) relative errors, not O(1e-6). */ + * FP note: the kernel sums in f32 while the format reference sums in double, so + * we compare those two against a relative epsilon. Forced-SSE builds separately + * require byte identity with the scalar f32 operation tree. */ #define main coli_glm_main_unused #include "../colibri.c" #undef main @@ -53,15 +51,54 @@ static void ref_grouped(double *y, double *mag, const float *x, const uint8_t *q } } +/* Scalar contract: this is the scalar grouped kernel's operation tree, not a + * higher-precision format reference. The forced-SSE builds compare its output + * byte for byte with the four-output-row SIMD path. */ +#ifdef COLI_I4_GROUPED_SCALAR_EXACT +static void ref_grouped_scalar(float *y,const float *x,const uint8_t *q4, + const float *scale,int S,int I,int O,int gs){ + int rb=(I+1)/2,ng=(I+gs-1)/gs; + for(int o=0;oI) glen=I-base; + float sc=scl[g]; + for(int i=base;i>1]; + a+=(xs[i]*(float)((int)(byte&0xF)-8) + +xs[i+1]*(float)((int)(byte>>4)-8))*sc; + }else{ + uint8_t byte=w[i>>1]; + a+=xs[i]*(float)((int)(byte&0xF)-8)*sc; + } + } + } + y[(int64_t)s*O+o]=a; + } + } +} +#endif + static int check(const char *name, int S, int I, int O, int gs, int fill_edges){ int rb=(I+1)/2, ng=(I+gs-1)/gs; uint8_t *q4=malloc((size_t)O*rb); float *scale=malloc((size_t)O*ng*sizeof(float)); float *x=malloc((size_t)S*I*sizeof(float)); float *y=malloc((size_t)S*O*sizeof(float)); +#ifdef COLI_I4_GROUPED_SCALAR_EXACT + float *ys=malloc((size_t)S*O*sizeof(float)); +#endif double *yr=malloc((size_t)S*O*sizeof(double)); double *ym=malloc((size_t)S*O*sizeof(double)); - if(!q4||!scale||!x||!y||!yr||!ym){ fprintf(stderr,"%s: OOM\n",name); return 1; } + if(!q4||!scale||!x||!y||!yr||!ym +#ifdef COLI_I4_GROUPED_SCALAR_EXACT + ||!ys +#endif + ){ fprintf(stderr,"%s: OOM\n",name); return 1; } for(size_t i=0;i<(size_t)O*rb;i++) q4[i]=(uint8_t)(xr()&0xFF); if(fill_edges){ @@ -75,6 +112,9 @@ static int check(const char *name, int S, int I, int O, int gs, int fill_edges){ matmul_i4_grouped(y,x,q4,scale,S,I,O,gs); ref_grouped(yr,ym,x,q4,scale,S,I,O,gs); +#ifdef COLI_I4_GROUPED_SCALAR_EXACT + ref_grouped_scalar(ys,x,q4,scale,S,I,O,gs); +#endif int bad=0; double worst=0; for(int i=0;i>4)-8)){ + fprintf(stderr,"rows4 unpack: rb=%d o=%d byte=%d row=%d value=%u\n", + rb,o,byte,row,packed); + free(q4); return 1; + } + } + } + } + free(q4); + } + printf(" rows4 unpack: all 256 byte values per lane, tight/strided rows ok\n"); return 0; } #endif @@ -162,6 +254,9 @@ static int check_pair(const char *name, int S, int I, int O, int gs){ int main(void){ int fail=0; printf("test_i4_grouped: matmul_i4_grouped vs plain-C dequant reference\n"); +#if defined(__SSE4_1__) && !defined(__AVX2__) + fail|=check_rows4_unpack(); +#endif /* the shape the g64 checkpoints actually use */ fail|=check("gs=64, I multiple of gs", 2, 512, 8, 64, 0); @@ -185,6 +280,7 @@ int main(void){ /* batch: S>1 exercises the per-s inner loop against a shared scale row */ fail|=check("gs=64, batch S=8", 8, 320, 6, 64, 0); + fail|=check("one-byte rows, odd I=1", 1, 1, 4, 2, 0); #ifdef COLI_HAVE_GROUPED_PAIR printf("test_i4_grouped: matmul_i4_grouped_pair (fused gate+up) vs two separate calls\n"); @@ -193,6 +289,7 @@ int main(void){ fail|=check_pair("pair: gs=64, odd I (I=201)", 2, 201, 4, 64); fail|=check_pair("pair: gs=64, decode S=1", 1, 320, 6, 64); fail|=check_pair("pair: gs=128, I=512", 2, 512, 4, 128); + fail|=check_pair("pair: one-byte rows, odd I=1", 1, 1, 4, 2); #endif if(fail){ printf("test_i4_grouped: FAIL\n"); return 1; } diff --git a/c/tests/test_inkling_cache_index.c b/c/tests/test_inkling_cache_index.c index 797f6af4e..72d5571e3 100644 --- a/c/tests/test_inkling_cache_index.c +++ b/c/tests/test_inkling_cache_index.c @@ -82,6 +82,14 @@ int main(void){ free_cache(&m); check_lookup_scaling(44); check_lookup_scaling(219); + { + double avail = mem_avail_bytes(); + double probe = compat_mem_available_gb(); + CHECK(avail > 0.0, "mem_avail_bytes returned 0; auto cap would be 16 experts/layer"); + double gb = avail / 1e9; + double gap = gb > probe ? gb - probe : probe - gb; + CHECK(gap < 0.25, "mem_avail_bytes diverged from the shared probe (%.3f vs %.3f GB)", gb, probe); + } if(failures){ fprintf(stderr,"inkling cache index: %d failure(s)\n",failures); return 1; } puts("inkling cache index: ok"); return 0; diff --git a/c/tests/test_inkling_mem_probe.py b/c/tests/test_inkling_mem_probe.py new file mode 100644 index 000000000..c25f6b6e1 --- /dev/null +++ b/c/tests/test_inkling_mem_probe.py @@ -0,0 +1,79 @@ +"""Inkling auto-sized its expert cache from mem_avail_bytes(). + +On Windows that function took the #else return 0 branch, so every +`coli serve --model ` with no --cap used 16 experts/layer +regardless of installed RAM, and /health reported ram_total_gb 0.0. +""" +import ctypes +import re +import sys +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent +ENGINE_C = HERE.parent / "inkling.c" + + +def _c_function(source, signature): + match = re.search(re.escape(signature) + r"\s*\{", source) + if not match: + raise ValueError("no definition for " + signature) + start = match.start() + brace = match.end() - 1 + depth = 0 + for index, char in enumerate(source[brace:], start=brace): + if char == "{": + depth += 1 + elif char == "}": + depth -= 1 + if depth == 0: + return source[start:index + 1] + raise ValueError("unbalanced braces in " + signature) + + +def _windows_avail_phys(): + class MEMORYSTATUSEX(ctypes.Structure): + _fields_ = [ + ("dwLength", ctypes.c_ulong), + ("dwMemoryLoad", ctypes.c_ulong), + ("ullTotalPhys", ctypes.c_ulonglong), + ("ullAvailPhys", ctypes.c_ulonglong), + ("ullTotalPageFile", ctypes.c_ulonglong), + ("ullAvailPageFile", ctypes.c_ulonglong), + ("ullTotalVirtual", ctypes.c_ulonglong), + ("ullAvailVirtual", ctypes.c_ulonglong), + ("ullAvailExtendedVirtual", ctypes.c_ulonglong), + ] + stat = MEMORYSTATUSEX() + stat.dwLength = ctypes.sizeof(stat) + ok = ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(stat)) + return stat.ullAvailPhys if ok else 0 + + +class InklingMemProbeTest(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.source = ENGINE_C.read_text(encoding="utf-8") + cls.avail = _c_function(cls.source, "static double mem_avail_bytes(void)") + cls.hwinfo = _c_function(cls.source, "static void serve_hwinfo(Model *m)") + + def test_mem_avail_bytes_uses_the_shared_probe(self): + self.assertIn("compat_mem_available_gb", self.avail) + self.assertNotRegex( + self.avail, r"#else\s*\n\s*return 0;", + "Windows still takes a return-0 branch instead of measuring RAM") + + def test_serve_hwinfo_fills_ram_when_proc_is_missing(self): + self.assertIn("compat_meminfo_gb", self.hwinfo) + + def test_windows_host_has_ram_the_old_probe_would_ignore(self): + if sys.platform != "win32": + return + free = _windows_avail_phys() + self.assertGreater(free, 1_000_000_000, + "this host reports no reclaimable RAM") + if re.search(r"#else\s*\n\s*return 0;", self.avail): + self.fail( + "mem_avail_bytes() returns 0 on Windows; this host has " + f"{free / 1e9:.1f} GB free, so auto cap would stay at " + "16 experts/layer") diff --git a/c/tests/test_inkling_serve_framing.c b/c/tests/test_inkling_serve_framing.c index 9bd1e7cce..d48eabe06 100644 --- a/c/tests/test_inkling_serve_framing.c +++ b/c/tests/test_inkling_serve_framing.c @@ -101,14 +101,18 @@ static void test_malformed_submit_cannot_become_control(void) } /* The gateway answers `ERROR CONTEXT_EXCEEDED ...` with a 400 - * context_length_exceeded; any other refusal text reaches the client as a 500. */ + * context_length_exceeded; any other refusal text reaches the client as a 500. + * max_tokens is a ceiling: coli chat's 16384 used to 400 a prompt that fits. */ static void test_over_long_prompt_is_refused_with_the_context_frame(void) { assert(setenv("CTX_MAX","16",1)==0); assert(prompt_reject(12,4)==NULL); + assert(prompt_reject(2,16384)==NULL); + assert(coli_serve_budget(2,16384,16,0)==14); const char *refusal=prompt_reject(30,4); assert(refusal); assert(strcmp(refusal,"CONTEXT_EXCEEDED prompt_tokens=30 requested=4 capacity=16")==0); + assert(prompt_reject(16,4)!=NULL); } int main(void) diff --git a/c/tests/test_install_closure.py b/c/tests/test_install_closure.py new file mode 100644 index 000000000..23501da74 --- /dev/null +++ b/c/tests/test_install_closure.py @@ -0,0 +1,72 @@ +"""What `make install` puts on a machine has to be what the launcher needs. + +Two invariants from the field report in #1689: + +- the launcher derives its libexec directory from its own path, so it must + resolve symlinks first: a merged-/usr system exposes /bin/coli as a link to + /usr/bin/coli, and abspath() kept the alias, deriving /libexec/colibri; +- every root-level Python module the launcher reaches (tools/pack_python.py + computes that closure for the release archive) must be in the handwritten + `make install` list, or installed `coli serve` fails on an import the + source tree satisfied (v41_dsml.py in 1.12.0). +""" +import os +import re +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + +HERE = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(HERE / "tools")) +import pack_python # noqa: E402 + + +def installed_root_modules(makefile): + """The .py files the install recipe copies into $(LIBEXECDIR)/ itself.""" + text = makefile.read_text(encoding="utf-8") + body = text[text.index("\ninstall:"):] + body = body[:body.index("\nuninstall:")] + body = body.replace("\\\n", " ") + names = set() + for line in body.splitlines(): + if "$(INSTALL) -m 644" in line and line.rstrip().endswith("$(DESTDIR)$(LIBEXECDIR)/"): + names |= set(re.findall(r"\b([A-Za-z0-9_]+\.py)\b", line)) + return names + + +class InstallClosure(unittest.TestCase): + def test_every_reached_root_module_is_installed(self): + reached = {p.name for p in pack_python.needed(HERE) + if p.parent == HERE and p.suffix == ".py"} + installed = installed_root_modules(HERE / "Makefile") + self.assertTrue(reached, "pack_python reached no root module; the parser is broken") + self.assertEqual(reached - installed, set(), + "reached by the launcher but missing from `make install`") + + @unittest.skipIf(sys.platform == "win32", "symlinks need privileges on Windows") + def test_launcher_resolves_a_merged_usr_alias(self): + with tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + (root / "usr" / "bin").mkdir(parents=True) + (root / "usr" / "libexec").mkdir() + launcher = root / "usr" / "bin" / "coli" + launcher.write_bytes((HERE / "coli").read_bytes()) + launcher.chmod(0o755) + # the support modules of this checkout stand in for the installed ones + os.symlink(HERE, root / "usr" / "libexec" / "colibri") + os.symlink(root / "usr" / "bin", root / "bin") + env = {k: v for k, v in os.environ.items() if k not in ("COLI_ENGINE", "PYTHONPATH")} + for alias in (root / "bin" / "coli", root / "usr" / "bin" / "coli"): + with self.subTest(alias=str(alias.relative_to(root))): + result = subprocess.run([sys.executable, str(alias), "--version"], + capture_output=True, text=True, env=env, + cwd=tmp, timeout=60) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertIn("colibri", result.stdout) + self.assertNotIn("ModuleNotFoundError", result.stderr) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_int_kernel_exact.c b/c/tests/test_int_kernel_exact.c index aa6201738..25318a61a 100644 --- a/c/tests/test_int_kernel_exact.c +++ b/c/tests/test_int_kernel_exact.c @@ -117,11 +117,21 @@ int main(void){ * float (per-gruppo: int esatto -> fmaf con la scala; poi * sx). Il kernel * e' opt-in e non-bit-identico al f32 a gruppi; contro il SUO riferimento * deve invece essere esatto al bit su ogni ISA. Copre gs=64 e gs=128 - * (bpg=2) e una coda I%gs!=0. */ + * (bpg=2), una coda I%gs!=0, e le S che accendono le forme multi-riga: + * S=3 (solo per-riga), S=6/9 (tile 1x4 + resto), S=18 (due s-tile AMX, + * 16+2, sui build AMX — il gate S>=AMX_S_MIN e' il default 8). O=136 + * lascia un resto O%16 cosi' il ramo AMX consegna anche la coda di + * output al path vettoriale. */ { - int cases[][2]={{64,2048},{128,2048},{64,2000}}; - for(int t=0;t<3;t++){ - int gs=cases[t][0], I=cases[t][1], O=128, S=3; +#ifdef COLI_HAVE_AMX_I4P + /* Emulator runs (Intel SDE) cannot take the OS tile-arming path; the + * arming is engine POLICY, the tile math is what this gate proves, so + * the test may force the state to reach the AMX kernel under SDE. */ + if(getenv("COLI_TEST_FORCE_AMX")){ coli_amx_state=1; } +#endif + int cases[][3]={{64,2048,3},{64,2048,9},{128,2048,18},{64,2000,6},{64,128,18}}; + for(int t=0;t<5;t++){ + int gs=cases[t][0], I=cases[t][1], O=136, S=cases[t][2]; int rb=(I+1)/2, ng=(I+gs-1)/gs; uint8_t *qp=malloc((size_t)O*rb); for(size_t i=0;i<(size_t)O*rb;i++) qp[i]=(uint8_t)rnd(); uint8_t *ql=malloc((size_t)O*rb); memcpy(ql,qp,(size_t)O*rb); planarize_i4(ql,O,I); diff --git a/c/tests/test_json.c b/c/tests/test_json.c index 56941a907..da0a4d19f 100644 --- a/c/tests/test_json.c +++ b/c/tests/test_json.c @@ -30,6 +30,42 @@ int main(void) { CHECK(values->kids[1]->num == -2.5); CHECK(values->kids[2]->num == 300.0); CHECK(strcmp(json_get(root, "unicode")->str, "λ 🚀") == 0); + json_free(root); + + /* Checked key lookup must not alias a name through a decoded NUL. The + * legacy API still returns its original permissive representation. */ + const char *nul_key="{\"x\\u0000ignored\":1}"; + CHECK(json_parse_checked(nul_key) == NULL); + root=json_parse(nul_key,NULL); + CHECK(root && json_get(root,"x") && json_get(root,"x")->num == 1.0); + json_free(root); + + /* Legal escapes, including NUL in a string value, remain valid. */ + root=json_parse_checked("{\"\\u0078\":\"a\\u0000b\\f\\u000b\"}"); + CHECK(root && json_get(root,"x")); + CHECK(memcmp(json_get(root,"x")->str,"a\0b\f\v",5) == 0); + json_free(root); + + const char whitespace[]=" \t\r\n\f\v"; + const char *positions[]={ + "%c{\"x\":[1]}", "{%c\"x\":[1]}", "{\"x\"%c:[1]}", + "{\"x\":%c[1]}", "{\"x\":[%c1]}", "{\"x\":[1%c]}", + "{\"x\":[1]%c}", "{\"x\":[1]}%c", + }; + for (size_t i=0; it == J_ARR && values->len == 1); + CHECK(values->kids[0]->num == 1.0); + json_free(root); + } + } puts("json tests: ok"); return 0; diff --git a/c/tests/test_k3_chat_tools.c b/c/tests/test_k3_chat_tools.c index b7cc74b6d..de0c7df2c 100644 --- a/c/tests/test_k3_chat_tools.c +++ b/c/tests/test_k3_chat_tools.c @@ -116,6 +116,22 @@ int main(void){ "G 0\n",want); } + /* C: continuation -- the FINAL assistant turn is left OPEN. The prior turns render as + * usual, but the last turn has no <|close|>/<|end_of_msg|> and NO fresh generation cue + * is appended (contrast tail_plain/tail_think in every case above). */ + expect(&T,"C leaves the final assistant turn open, no cue appended", + "K3CHAT1\nM user 6\nhello?C 0 2\nhiG 0\n", + "<|open|>message role=\"user\"<|sep|>hello?<|close|>message<|sep|><|end_of_msg|>" + "<|open|>message role=\"assistant\"<|sep|><|open|>response<|sep|>hi"); + + /* C with reasoning: the think block is closed, the response is left open -- the model + * resumes writing its answer, its reasoning already behind it. */ + expect(&T,"C with reasoning closes think and leaves response open", + "K3CHAT1\nM user 6\nhello?C 3 2\nwhyhiG 0\n", + "<|open|>message role=\"user\"<|sep|>hello?<|close|>message<|sep|><|end_of_msg|>" + "<|open|>message role=\"assistant\"<|sep|><|open|>think<|sep|>why<|close|>think<|sep|>" + "<|open|>response<|sep|>hi"); + /* Malformed records are refused, not misread. */ { int ids[256], sp[4], thinking=0; @@ -123,6 +139,8 @@ int main(void){ "K3CHAT1\nY 200 4\nshortG 0\n", /* lengths past the payload */ "K3CHAT1\nO 0 3 2\nfooxxG 0\n", /* index < 1 */ "K3CHAT1\nB 0 0 0 1\nX 3 0\nfooG 0\n", /* unknown call record */ + "K3CHAT1\nC 0 2\nhiM user 1\nxG 0\n", /* a turn after the open final turn */ + "K3CHAT1\nC 0 2\nhiC 0 1\nyG 0\n", /* two open final turns */ }; for(size_t i=0;i avail ? a2 - avail : avail - a2; + check(gap < 0.25, "compat_meminfo_gb e compat_mem_available_gb non concordano"); } printf(" disponibile: %.2f GB\n", avail); check(avail > 0.0, "la misura vale 0: la piattaforma non e' coperta (era il bug di Windows)"); double total = total_gb(); diff --git a/c/tests/test_mxfp4_cuda.cu b/c/tests/test_mxfp4_cuda.cu index 48c1dddbf..8f567e4ee 100644 --- a/c/tests/test_mxfp4_cuda.cu +++ b/c/tests/test_mxfp4_cuda.cu @@ -25,7 +25,12 @@ #include #include #include +#if defined(__HIPCC__) +#include "../backend_gpu_compat.h" /* this TU links against a separately compiled backend_cuda.cu, + so it needs the CUDA->HIP mapping itself */ +#else #include +#endif /* quant.h is C (it uses _Thread_local, which nvcc's C++ front end rejects), so * the reference is compiled separately as C and reached through this one @@ -68,8 +73,8 @@ static void compare_case(const char *what, const float *y_cpu, const float *y_gp if (std::isnan(a) != std::isnan(b)) bad++; continue; } - if (isinf(a) || isinf(b)) { /* exponent 255: both must agree it is inf */ - if (isinf(a) != isinf(b) || (isinf(a) && ((a > 0) != (b > 0)))) bad++; + if (std::isinf(a) || std::isinf(b)) { /* exponent 255: both must agree it is inf */ + if (std::isinf(a) != std::isinf(b) || (std::isinf(a) && ((a > 0) != (b > 0)))) bad++; continue; } double den = fabs(a) > 1e-6 ? fabs(a) : 1e-6; @@ -111,6 +116,13 @@ static void one_case(const char *what, int S, int I, int O, int fixed_exp) { } compare_case(what, y_cpu, y_gpu, S, I, O); + /* The engine recycles host slots: identical addresses must upload fresh + * bytes and scales, even when device scratch already has enough capacity. */ + memset(q4, 0x22, (size_t)O * rb); + memset(e8, 127, (size_t)O * ng); + mxfp4_ref(y_cpu, x, q4, e8, S, I, O); + if (!coli_cuda_matmul_mxfp4(y_gpu, x, q4, e8, S, I, O)) fails++; + else compare_case("recycled host slot", y_cpu, y_gpu, S, I, O); done: free(q4); free(e8); free(x); free(y_cpu); free(y_gpu); } @@ -199,6 +211,41 @@ static void all_codes(void) { else printf(" ok all 16 e2m1 codes decode exactly (cpu == gpu == spec)\n"); } +/* Resident upload/update must use O*ceil(I/32) BYTES, including tails. + * Positive inputs avoid cancellation so every missing group affects the result. */ +static void resident_case(int I, int O) { + const int S = 2, rb = (I + 1) / 2, ng = (I + 31) / 32; + uint8_t *q = (uint8_t *)malloc((size_t)O * rb); + uint8_t *sc = (uint8_t *)malloc((size_t)O * ng); + float *x = (float *)malloc((size_t)S * I * sizeof(float)); + float *want = (float *)malloc((size_t)S * O * sizeof(float)); + float *got = (float *)malloc((size_t)S * O * sizeof(float)); + memset(q, 0x22, (size_t)O * rb); + for (int i = 0; i < S * I; i++) x[i] = 0.25f * (1 + i % 3); + size_t count0, bytes0, count1, bytes1; + coli_cuda_stats(0, &count0, &bytes0); + ColiCudaTensor *t = nullptr; + for (int pass = 0; pass < 2; pass++) { + for (int i = 0; i < O * ng; i++) sc[i] = (uint8_t)(125 + (i + pass) % 5); + const float *scales = reinterpret_cast(sc); + int ok = pass ? coli_cuda_tensor_update(t, q, scales) + : coli_cuda_tensor_upload(&t, q, scales, 7, I, O, 0); + if (!ok) { printf(" FAIL resident upload/update\n"); fails++; break; } + size_t expected = (size_t)O * (rb + ng); + coli_cuda_stats(0, &count1, &bytes1); + if (coli_cuda_tensor_bytes(t) != expected || count1 != count0 + 1 || bytes1 != bytes0 + expected) { + printf(" FAIL resident byte accounting I=%d O=%d\n", I, O); fails++; + } + mxfp4_ref(want, x, q, sc, S, I, O); + if (!coli_cuda_matmul(&t, got, x, nullptr, nullptr, 7, S, I, O, 0, 0)) fails++; + else compare_case(pass ? "resident refresh" : "resident upload", want, got, S, I, O); + } + coli_cuda_tensor_free(t); + coli_cuda_stats(0, &count1, &bytes1); + if (count0 != count1 || bytes0 != bytes1) { printf(" FAIL resident free accounting\n"); fails++; } + free(q); free(sc); free(x); free(want); free(got); +} + int main(void) { int ndev = 0; if (cudaGetDeviceCount(&ndev) != cudaSuccess || ndev < 1) { @@ -211,6 +258,8 @@ int main(void) { return 0; } + resident_case(33, 3); /* 6 exponent bytes, not 12 */ + resident_case(257, 5); /* 45 exponent bytes, not 20 */ all_codes(); one_case("decode + matmul", 1, 64, 32, -1); one_case("multi-row batch", 4, 128, 64, -1); @@ -226,6 +275,13 @@ int main(void) { mixed_254_255(); one_case("exponent 127 -> unit scale", 1, 64, 16, 127); + coli_cuda_shutdown(); + if (!coli_cuda_init(&dev0, 1)) fails++; + else { + one_case("after shutdown/reinit", 2, 128, 32, 127); + coli_cuda_shutdown(); + } + printf(fails ? "test_mxfp4_cuda: %d failure(s)\n" : "test_mxfp4_cuda: ok\n", fails); return fails != 0; } diff --git a/c/tests/test_mxfp4_expert_cuda.cu b/c/tests/test_mxfp4_expert_cuda.cu new file mode 100644 index 000000000..10b219df6 --- /dev/null +++ b/c/tests/test_mxfp4_expert_cuda.cu @@ -0,0 +1,58 @@ +/* Streaming SiTU-GLU expert vs the CPU MXFP4 decoder, including odd widths. */ +#include +#include +#include +#if defined(__HIPCC__) +#include "../backend_gpu_compat.h" +#else +#include +#endif +#include "../backend_cuda.h" +extern "C" void mxfp4_ref(float *, const float *, const unsigned char *, const unsigned char *, int, int, int); + +static int check_case(int S, int D, int I, float b1, float b2) { + std::vector gw((size_t)I * ((D + 1) / 2)), uw(gw.size()), dw((size_t)D * ((I + 1) / 2)); + std::vector gs((size_t)I * ((D + 31) / 32)), us(gs.size()), ds((size_t)D * ((I + 31) / 32)); + std::vector x((size_t)S * D), g((size_t)S * I), u(g.size()), want(x.size()), got(x.size()); + for (size_t j = 0; j < gw.size(); j++) { gw[j] = (unsigned char)(j * 13 + 2); uw[j] = (unsigned char)(j * 7 + 5); } + for (size_t j = 0; j < dw.size(); j++) dw[j] = (unsigned char)(j * 11 + 3); + for (size_t j = 0; j < gs.size(); j++) { gs[j] = 121 + j % 4; us[j] = 122 + j % 3; } + for (size_t j = 0; j < ds.size(); j++) ds[j] = 123 + j % 4; + for (size_t j = 0; j < x.size(); j++) x[j] = (int(j % 17) - 8) * 0.125f; + for (int pass = 0; pass < 2; pass++) { + mxfp4_ref(g.data(), x.data(), gw.data(), gs.data(), S, D, I); + mxfp4_ref(u.data(), x.data(), uw.data(), us.data(), S, D, I); + for (size_t j = 0; j < g.size(); j++) + g[j] = b1 * tanhf(g[j] / b1) * (1.f / (1.f + expf(-g[j]))) * b2 * tanhf(u[j] / b2); + mxfp4_ref(want.data(), g.data(), dw.data(), ds.data(), S, I, D); + if (!coli_cuda_expert_mxfp4(got.data(), x.data(), gw.data(), gs.data(), uw.data(), us.data(), + dw.data(), ds.data(), S, D, I, b1, b2)) return 1; + for (size_t j = 0; j < got.size(); j++) { + if (!std::isfinite(got[j]) || fabsf(got[j] - want[j]) > 2e-5f + 2e-4f * fabsf(want[j])) { + printf("FAIL S=%d D=%d I=%d row-element=%zu want=%g got=%g\n", S, D, I, j, want[j], got[j]); + return 1; + } + } + /* Reused host addresses now contain different expert bytes. */ + for (size_t j = 0; j < gw.size(); j++) gw[j] ^= 0x88; + for (size_t j = 0; j < ds.size(); j++) ds[j]++; + } + got[0] = 123.f; + if (coli_cuda_expert_mxfp4(got.data(), x.data(), gw.data(), gs.data(), uw.data(), us.data(), + dw.data(), ds.data(), S, D, I, 0.f, b2) || got[0] != 123.f) return 1; + printf("ok SiTU expert S=%d D=%d I=%d b1=%g b2=%g\n", S, D, I, b1, b2); + return 0; +} +int main() { + int n = 0, device = 0; + if (cudaGetDeviceCount(&n) != cudaSuccess || !n) { puts("SKIP no CUDA device"); return 0; } + if (!coli_cuda_init(&device, 1)) return 1; + int failed = check_case(1, 64, 128, 1.5f, 2.5f) | + check_case(3, 33, 65, 4.f, 3.f) | + check_case(2, 129, 31, 0.5f, 0.75f); + coli_cuda_shutdown(); + if (!coli_cuda_init(&device, 1)) return 1; + failed |= check_case(1, 64, 128, 1.5f, 2.5f); + coli_cuda_shutdown(); + return failed; +} diff --git a/c/tests/test_olmoe_cache_index.c b/c/tests/test_olmoe_cache_index.c index 6556e2d15..8a2271734 100644 --- a/c/tests/test_olmoe_cache_index.c +++ b/c/tests/test_olmoe_cache_index.c @@ -40,6 +40,150 @@ static void check_lookup_scaling(int cap) { free_cache(&m); } +/* The first read of a scenario stays open until the test releases it; every + * later read returns at once. g_io_loads counts them all. */ +static pthread_mutex_t g_io_mx = PTHREAD_MUTEX_INITIALIZER; +static pthread_cond_t g_io_cv = PTHREAD_COND_INITIALIZER; +static int g_io_loads, g_io_open, g_io_release; + +static void held_load(Model *m, int layer, int eid, Slot *s) { + (void)m; (void)layer; (void)eid; (void)s; + pthread_mutex_lock(&g_io_mx); + if (g_io_loads++ == 0) { + g_io_open = 1; + pthread_cond_broadcast(&g_io_cv); + while (!g_io_release) pthread_cond_wait(&g_io_cv, &g_io_mx); + } + pthread_mutex_unlock(&g_io_mx); +} + +static int io_get(int *v) { + pthread_mutex_lock(&g_io_mx); int x = *v; pthread_mutex_unlock(&g_io_mx); + return x; +} + +static void io_release(void) { + pthread_mutex_lock(&g_io_mx); + g_io_release = 1; + pthread_cond_broadcast(&g_io_cv); + pthread_mutex_unlock(&g_io_mx); +} + +/* expert_get marks the expert routed under g_pilot_mx and keeps the lock until + * it either waits or starts its own read: seeing the mark means it got there. */ +static int routed(Model *m, int eid) { + pthread_mutex_lock(&g_pilot_mx); + int r = m->ehit && m->ehit[0][eid]; + pthread_mutex_unlock(&g_pilot_mx); + return r; +} + +typedef struct { Model *m; int eid, done; Slot *got; } Req; + +static void *demand_thread(void *p) { + Req *r = p; expert_get(r->m, 0, r->eid, &r->got); + __atomic_store_n(&r->done, 1, __ATOMIC_RELEASE); return NULL; +} + +static void *prefetch_thread(void *p) { + Req *r = p; pilot_realload(r->m, 0, r->eid); + __atomic_store_n(&r->done, 1, __ATOMIC_RELEASE); return NULL; +} + +/* Bounded wait: a lost wakeup must fail the test, not hang the suite. */ +static void wait_done(Req *r, pthread_t t, const char *who) { + for (int ms = 0; ms < 5000 && !__atomic_load_n(&r->done, __ATOMIC_ACQUIRE); ms++) sleep_ms(1); + if (!__atomic_load_n(&r->done, __ATOMIC_ACQUIRE)) { + fprintf(stderr, "FAIL %s:%d: %s never returned\n", __FILE__, __LINE__, who); + exit(1); + } + pthread_join(t, NULL); +} + +static int copies(Model *m, int eid) { + LCache *lc = &m->cache[0]; int n = 0; + for (int i = 0; i < lc->n; i++) n += lc->slots[i].eid == eid; + return n; +} + +static void init_loader(Model *m) { + init_cache(m, 8, 4); + m->c.inter = 4; m->c.hidden = 4; + m->is_queued = calloc(8, 1); m->is_pinned = calloc(8, 1); + g_io_loads = g_io_open = g_io_release = 0; + g_test_expert_load = held_load; +} + +static void free_loader(Model *m) { + LCache *lc = &m->cache[0]; + for (int i = 0; i < lc->n; i++) { free(lc->slots[i].g); free(lc->slots[i].gs); } + if (m->ehit) { free(m->ehit[0]); free(m->ehit); } + free(m->is_queued); free(m->is_pinned); + free_cache(m); + g_test_expert_load = NULL; +} + +/* The prefetcher is reading expert 3 when the forward pass routes to it: the + * forward pass must take the prefetched copy, not read it a second time. */ +static void check_demand_joins_prefetch(void) { + Model m; init_loader(&m); + m.is_queued[3] = 1; + Req pf = { &m, 3, 0, NULL }, dm = { &m, 3, 0, NULL }; + pthread_t tp, td; + pthread_create(&tp, NULL, prefetch_thread, &pf); + for (int ms = 0; ms < 5000 && !io_get(&g_io_open); ms++) sleep_ms(1); + CHECK(io_get(&g_io_open), "prefetch never started its read"); + pthread_create(&td, NULL, demand_thread, &dm); + for (int ms = 0; ms < 5000 && !routed(&m, 3); ms++) sleep_ms(1); + io_release(); + wait_done(&pf, tp, "pilot_realload"); + wait_done(&dm, td, "expert_get"); + CHECK(g_io_loads == 1, "expert 3 read %d times, prefetch already had it in flight", g_io_loads); + CHECK(copies(&m, 3) == 1, "expert 3 resident in %d slots", copies(&m, 3)); + CHECK(dm.got && dm.got == slot_indexed(&m, 0, 3), "forward pass got a slot the index does not name"); + free_loader(&m); +} + +/* The forward pass is reading expert 4 when a prefetch for it comes up: the + * prefetch must drop the request at once, without a read or a wait. */ +static void check_prefetch_skips_demand(void) { + Model m; init_loader(&m); + Req dm = { &m, 4, 0, NULL }; + pthread_t td; + pthread_create(&td, NULL, demand_thread, &dm); + for (int ms = 0; ms < 5000 && !io_get(&g_io_open); ms++) sleep_ms(1); + CHECK(io_get(&g_io_open), "forward pass never started its read"); + pthread_mutex_lock(&g_pilot_mx); m.is_queued[4] = 1; pthread_mutex_unlock(&g_pilot_mx); + pilot_realload(&m, 0, 4); + int loads = io_get(&g_io_loads); + CHECK(loads == 1, "expert 4 read %d times, forward pass already had it in flight", loads); + CHECK(m.is_queued[4] == 0, "dropped prefetch stayed queued"); + io_release(); + wait_done(&dm, td, "expert_get"); + CHECK(copies(&m, 4) == 1, "expert 4 resident in %d slots", copies(&m, 4)); + free_loader(&m); +} + +/* Two callers route to expert 5 at once: the second waits for the first read + * and must be woken by its publish. */ +static void check_demand_joins_demand(void) { + Model m; init_loader(&m); + Req a = { &m, 5, 0, NULL }, b = { &m, 5, 0, NULL }; + pthread_t ta, tb; + pthread_create(&ta, NULL, demand_thread, &a); + for (int ms = 0; ms < 5000 && !io_get(&g_io_open); ms++) sleep_ms(1); + CHECK(io_get(&g_io_open), "first caller never started its read"); + pthread_mutex_lock(&g_pilot_mx); m.ehit[0][5] = 0; pthread_mutex_unlock(&g_pilot_mx); + pthread_create(&tb, NULL, demand_thread, &b); + for (int ms = 0; ms < 5000 && !routed(&m, 5); ms++) sleep_ms(1); + io_release(); + wait_done(&a, ta, "first expert_get"); + wait_done(&b, tb, "second expert_get"); + CHECK(g_io_loads == 1, "expert 5 read %d times by two callers", g_io_loads); + CHECK(a.got && a.got == b.got, "two callers got different slots for expert 5"); + free_loader(&m); +} + int main(void) { Model m; init_cache(&m, 6, 2); LCache *lc = &m.cache[0]; lc->n = 2; cache_publish(&m, 0, &lc->slots[0], 1); @@ -62,6 +206,9 @@ int main(void) { free_cache(&m); check_lookup_scaling(44); check_lookup_scaling(219); + check_demand_joins_prefetch(); + check_prefetch_skips_demand(); + check_demand_joins_demand(); if (failures) { fprintf(stderr,"olmoe cache index: %d failure(s)\n", failures); return 1; } puts("olmoe cache index: ok"); return 0; diff --git a/c/tests/test_olmoe_chat_template.py b/c/tests/test_olmoe_chat_template.py new file mode 100644 index 000000000..3c3d781a7 --- /dev/null +++ b/c/tests/test_olmoe_chat_template.py @@ -0,0 +1,134 @@ +#!/usr/bin/env python3 +"""Il renderer di OLMoE nel gateway, contro il chat_template del checkpoint. + +Il template di OLMoE vive dentro tokenizer_config.json, non in un chat_template.jinja +a se'. Estrai il campo `chat_template` in un file e passalo qui. bos_token ed eos_token +sono lo stesso marcatore, "|||IP_ADDRESS|||". + +Se manca il template o jinja2, il test si dichiara SALTATO invece di passare: un +test che non ha trovato il suo riferimento non ha verificato niente, e dirlo +verde sarebbe peggio -- percio' un salto esce con codice 2, distinto dallo 0 di +un confronto riuscito. + +RIFERIMENTO (scaricato 2026-09-10): + repo allenai/OLMoE-1B-7B-0125-Instruct + file campo `chat_template` estratto da tokenizer_config.json + sha256 fe689ffbd6a4e2d0532d7480696b065b10e0e1eff3f9b9fc4bea415761e4bf4a (del .jinja estratto) + hf download allenai/OLMoE-1B-7B-0125-Instruct tokenizer_config.json # poi estrai chat_template + +USO: + python3 tests/test_olmoe_chat_template.py --template PATH/olmoe-chat_template.jinja +""" +import argparse +import sys +from pathlib import Path + +BOUNDARY = "|||IP_ADDRESS|||" # bos_token == eos_token + +CASES = { + "un turno utente": [{"role": "user", "content": "ciao"}], + "sistema piu' utente": [{"role": "system", "content": "Sei conciso."}, + {"role": "user", "content": "capitale della Francia?"}], + "assistant in cronologia": [{"role": "user", "content": "1+1?"}, + {"role": "assistant", "content": "2"}, + {"role": "user", "content": "e 2+2?"}], +} + + +def reference(template_text, *, messages, add_generation_prompt=True): + import jinja2 + environment = jinja2.Environment(trim_blocks=False, lstrip_blocks=False, + extensions=["jinja2.ext.loopcontrols"]) + rendered = environment.from_string(template_text) + return rendered.render(messages=messages, add_generation_prompt=add_generation_prompt, + bos_token=BOUNDARY, eos_token=BOUNDARY) + + +def show(label, ours, theirs): + print(f"FAIL {label}") + for index, (a, b) in enumerate(zip(ours.splitlines(), theirs.splitlines())): + if a != b: + print(f" prima differenza alla riga {index + 1}") + print(f" gateway: {a!r}") + print(f" template: {b!r}") + return + print(f" lunghezze diverse: gateway {len(ours)}, template {len(theirs)}") + print(f" coda gateway: {ours[-120:]!r}") + print(f" coda template: {theirs[-120:]!r}") + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--template", type=Path, required=True) + arguments = parser.parse_args() + + if not arguments.template.exists(): + print(f"SKIP: manca {arguments.template}; il riferimento non c'e' e " + f"questo test non ha verificato nulla") + return 2 # salto != successo (vedi docstring) + try: + import jinja2 # noqa: F401 + except ImportError: + print("SKIP: jinja2 non installato; senza non c'e' riferimento") + return 2 # salto != successo (vedi docstring) + + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + import openai_server + + openai_server.ARCH = "olmoe" + template_text = arguments.template.read_text(encoding="utf-8") + + failures = 0 + for label, messages in CASES.items(): + theirs = reference(template_text, messages=messages) + ours = openai_server.render_chat_olmoe(messages) + if ours == theirs: + print(f"ok {label}") + else: + show(label, ours, theirs) + failures += 1 + + # Prosecuzione: l'ultimo turno assistant e' da CONTINUARE. Il template chiude anche + # l'ultimo turno con eos_token, quindi la forma aperta e' quel rendering MENO l'eos + # finale e senza cue -- lo stesso taglio del terminatore delle famiglie ChatML. + aperto = [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + produced = openai_server.render_chat_for_arch(aperto, add_generation_prompt=False) + closed = reference(template_text, messages=aperto, add_generation_prompt=False) + # L'eos da togliere DEVE esserci nel riferimento: toglierlo "se c'e'" sarebbe un no-op + # silenzioso se il template cambiasse convenzione, e il confronto perderebbe senso. + if not closed.endswith(BOUNDARY): + print(f"FAIL prosecuzione: il riferimento non finisce con l'eos {BOUNDARY!r} da " + f"togliere -- convenzione del template cambiata? coda: {closed[-40:]!r}") + failures += 1 + expected = closed + else: + expected = closed[:-len(BOUNDARY)] + if produced == expected: + print("ok prosecuzione: turno aperto = template(add_generation_prompt=False) " + "senza l'eos finale") + else: + show("prosecuzione", produced, expected) + failures += 1 + if not produced.endswith("La capitale e'"): + print(f"FAIL prosecuzione: il prompt non finisce sull'apertura del client: " + f"{produced[-60:]!r}") + failures += 1 + if produced.endswith(BOUNDARY): + print("FAIL prosecuzione: il turno resta chiuso con l'eos") + failures += 1 + # Controllo negativo: col ramo normale lo stesso scambio DEVE finire sulla cue. + if not openai_server.render_chat_for_arch(aperto).endswith("<|assistant|>\n"): + print("FAIL prosecuzione: il ramo normale non emette piu' il prompt di generazione") + failures += 1 + + print() + if failures: + print(f"TEST FAIL ({failures} casi)") + return 1 + print("template OLMoE: il gateway e' identico al riferimento") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/c/tests/test_olmoe_dot_i8_16.c b/c/tests/test_olmoe_dot_i8_16.c new file mode 100644 index 000000000..b6373060c --- /dev/null +++ b/c/tests/test_olmoe_dot_i8_16.c @@ -0,0 +1,88 @@ +/* olmoe's dot_i8_16 (c/olmoe.c, ARM NEON / AVX2 / SSE4.1 variants): must be + * bit-for-bit identical to a plain scalar int8 dot product. + * + * Unlike qwen36's float GEMV SSE4.1 tiers (gsgemv.h/qgemv.h, tolerance-tested + * because float addition is not associative), dot_i8_16 is pure integer + * arithmetic -- sign-extend to int16, madd, horizontal sum, no rounding at + * any step -- so every variant's own comment in olmoe.c claims exactness + * ("Bit-for-bit identical to the AVX2 version above"). That claim previously + * had no automated test backing it in this diff; this is it. + * + * Does NOT test matmul_q's IDOT path (that's tests/test_olmoe_matmul_q.c's + * job, and deliberately non-exact -- activation quantization -- see that + * file's header comment for why the two must not be conflated). + * + * Build: make -C c tests/test_olmoe_dot_i8_16 + * make -C c tests/test_olmoe_dot_i8_16_sse41 (forces the SSE4.1 body) + */ +#define OLMOE_TESTING 1 +#define main coli_olmoe_main_unused +#include "../olmoe.c" +#undef main + +#include +#include + +static int32_t dot_i8_16_scalar_ref(const int8_t *a, const int8_t *b) { + int32_t acc = 0; + for (int i = 0; i < 16; i++) acc += (int32_t)a[i] * (int32_t)b[i]; + return acc; +} + +int main(void) { +#if !defined(HAVE_FAST_DOT_I8) + printf("test_olmoe_dot_i8_16: no fast dot_i8_16 compiled in this build " + "(scalar-only -- nothing to test here)\n"); + return 0; +#else + int fails = 0; + const int N = 10000; + + /* dot_i8_16's SSE4.1 body loads a full 16-byte SSE register from a+8/b+8 + * and only consumes the low 8 bytes -- exactly like production callers' + * larger buffers (e.g. matmul_q's `xi[4096]`), a 16-element array here + * would make that load read 8 bytes past the end. Pad so the same call + * pattern stays in-bounds for THIS test without touching dot_i8_16 itself. */ + srand(12345); + for (int t = 0; t < N; t++) { + int8_t a[24] = {0}, b[24] = {0}; + for (int i = 0; i < 16; i++) { + a[i] = (int8_t)(rand() % 256 - 128); + b[i] = (int8_t)(rand() % 256 - 128); + } + int32_t got = dot_i8_16(a, b); + int32_t want = dot_i8_16_scalar_ref(a, b); + if (got != want) { + if (fails < 10) fprintf(stderr, "MISMATCH random t=%d: got %d want %d\n", t, got, want); + fails++; + } + } + + /* Edge cases: all-zero, and the saturation-adjacent extremes (-128 has no + * positive counterpart in int8, the one place a sign-extend bug would + * show up first). */ + int8_t zero[24] = {0}, minv[24] = {0}, maxv[24] = {0}; + for (int i = 0; i < 16; i++) { minv[i] = -128; maxv[i] = 127; } + struct { const char *name; const int8_t *a; const int8_t *b; } edge[] = { + { "zero.zero", zero, zero }, + { "min.min", minv, minv }, + { "min.max", minv, maxv }, + { "max.max", maxv, maxv }, + }; + for (size_t e = 0; e < sizeof(edge) / sizeof(edge[0]); e++) { + int32_t got = dot_i8_16(edge[e].a, edge[e].b); + int32_t want = dot_i8_16_scalar_ref(edge[e].a, edge[e].b); + if (got != want) { + fprintf(stderr, "MISMATCH edge %s: got %d want %d\n", edge[e].name, got, want); + fails++; + } + } + + if (fails) { + fprintf(stderr, "test_olmoe_dot_i8_16: %d failures\n", fails); + return 1; + } + printf("test_olmoe_dot_i8_16: ALL PASS (0 failures), %d random pairs + 4 edge cases, bit-exact\n", N); + return 0; +#endif +} diff --git a/c/tests/test_olmoe_serve_framing.c b/c/tests/test_olmoe_serve_framing.c index 41858edd7..546e9af45 100644 --- a/c/tests/test_olmoe_serve_framing.c +++ b/c/tests/test_olmoe_serve_framing.c @@ -175,6 +175,14 @@ static void test_bad_frame_stops_before_the_next_command(void) fclose(output); } +static void test_max_tokens_is_a_ceiling(void) +{ + assert(coli_serve_budget(2, 1024, 4096, 0) == 1024); + assert(coli_serve_budget(3500, 1024, 4096, 0) == 596); + assert(coli_serve_budget(4096, 1, 4096, 0) == -1); + assert(coli_serve_budget(4096, 0, 4096, 1) == 0); +} + int main(void) { test_submit_moves_exact_payload_into_queue(); @@ -182,6 +190,7 @@ int main(void) test_errors_are_byte_exact_and_frames_are_consumed(); test_invalid_submit_does_not_parse_its_payload_as_control(); test_bad_frame_stops_before_the_next_command(); + test_max_tokens_is_a_ceiling(); puts("olmoe serve framing baseline: ok"); return 0; } diff --git a/c/tests/test_openai_server.py b/c/tests/test_openai_server.py index 75e691e7c..8695bc9dc 100644 --- a/c/tests/test_openai_server.py +++ b/c/tests/test_openai_server.py @@ -16,14 +16,20 @@ from pathlib import Path from openai_server import (APIError, APIHandler, APIServer, ClientCancelled, + CONTINUATION_FAMILIES, DEFAULT_CHAT_STOP_SEQUENCES, END, GenerationScheduler, READY, Engine, InklingStreamSplit, StopFilter, ThinkingStreamSplit, - _engine_error, _image_bytes_from_url, cap_for_arch, conversation_cache_slot, model_arch, + _engine_error, _image_bytes_from_url, cap_for_arch, + conversation_cache_slot, model_arch, generation_options, parse_tool_calls, parse_dsv4_tool_calls, parse_arch_tool_calls, parse_k3_tool_calls, parse_qwen38_tool_calls, - read_engine_turn, render_chat, render_chat_kimi, render_chat_olmoe, - render_chat_qwen38, render_chat_v4, _dsv4_tool_calls, serve, - split_thinking_reply, + read_engine_turn, render_chat, render_chat_for_arch, + render_chat_glm53, render_chat_inkling, render_chat_kimi, + render_chat_olmoe, + render_chat_qwen38, render_chat_v4, render_chat_dsv41, + _dsv4_tool_calls, serve, + resolve_generation_prompt, split_thinking_reply, + starts_in_reasoning, stop_policy, tune_child_env) @@ -255,6 +261,42 @@ def test_kimi_json_fallback_for_unparseable_arguments(self): "name": "fn", "arguments": "not json"}}]}]) self.assertIn("J 2 8\nfnnot json", prompt) + def test_glm_renders_a_tool_call_whose_arguments_are_not_an_object(self): + """`arguments` that parses but is not an object must not kill the request. + + Both GLM renderers already tolerate `arguments` that does not parse at + all -- the except branch sets {} -- and every sibling renderer (Kimi + above, Qwen3.8, DeepSeek V4/V4.1) renders the call without arguments + rather than failing. Only the "parsed, but not an object" case reached + .items(), raised AttributeError, and came back as HTTP 500 "The colibri + engine failed to process the request." on a request the engine never + saw. + """ + import openai_server as srv + for arguments in ('[1, 2]', '"text"', '5', [1, 2], 7): + with self.subTest(arguments=arguments): + messages = [ + {"role": "user", "content": "run it"}, + {"role": "assistant", "content": "", "tool_calls": [ + {"id": "x", "type": "function", + "function": {"name": "fn", "arguments": arguments}}]}, + {"role": "tool", "content": "done"}, + ] + glm = render_chat(list(messages)) + self.assertIn("fn", glm) + self.assertNotIn("", glm) + glm53 = srv.render_chat_glm53(list(messages)) + self.assertIn("fn", glm53) + self.assertNotIn("", glm53) + # An object still renders its arguments, on both renderers. + renders = [{"role": "assistant", "content": "", "tool_calls": [ + {"id": "x", "type": "function", + "function": {"name": "fn", "arguments": '{"city": "Rome"}'}}]}] + self.assertIn("cityRome", + render_chat(list(renders))) + self.assertIn("cityRome", + srv.render_chat_glm53(list(renders))) + def test_kimi_still_rejects_unknown_roles(self): with self.assertRaisesRegex(APIError, "Unsupported role"): render_chat_kimi([{"role": "critic", "content": "hm"}]) @@ -417,6 +459,33 @@ def test_tool_choice_function_that_is_not_an_object_is_a_400(self): render_chat(list(messages), tools=list(tools), tool_choice=choice)) + def test_tool_function_that_is_not_an_object_is_a_400(self): + """The same slip as tool_choice, on tools[] this time. + + generation_options already has 400 "Tool function must be an object". + The GLM and DeepSeek declaration blocks then did fn.items() on + {"type": "function", "function": "search"} and raised AttributeError + before that 400 ran, so do_POST answered 500 "engine failed". + """ + import openai_server as srv + messages = [{"role": "user", "content": "hi"}] + for function in ("search", ["search"], 5, True): + with self.subTest(function=function): + tools = [{"type": "function", "function": function}] + with self.assertRaises(APIError) as caught: + generation_options({"tools": tools}, 8) + self.assertEqual(caught.exception.status, 400) + self.assertEqual(caught.exception.param, "tools.0.function") + for render in (render_chat, srv.render_chat_glm53, render_chat_kimi, + render_chat_v4, srv.render_chat_dsv41): + try: + render(list(messages), tools=list(tools)) + except APIError as error: + self.assertEqual(error.status, 400) + well = [{"type": "function", "function": {"name": "search"}}] + generation_options({"tools": well}, 8) + self.assertIn('"name": "search"', render_chat(list(messages), tools=well)) + def test_coli_temp_is_the_default_for_requests_that_omit_temperature(self): with patch.dict("openai_server.os.environ", {"COLI_TEMP": "0.25"}): self.assertEqual(generation_options({}, 8)[1], 0.25) @@ -532,7 +601,107 @@ def test_occupied_port_fails_before_engine_start(self): listener.close() +class GenerationMetricsTest(unittest.TestCase): + def setUp(self): + self.server = APIServer(("127.0.0.1", 0), FakeEngine(), "test") + self.addCleanup(self.server.server_close) + + def test_records_first_output_once_and_preserves_text_and_stats(self): + output = [] + with patch("openai_server.time.monotonic", side_effect=[10, 10.5, 13]): + stats = self.server.generate("prompt", 4, 0, 1, output.append) + self.assertEqual(output, ["Hé", "llo"]) + self.assertEqual(stats["completion_tokens"], 2) + metrics = self.server.scheduler.prometheus() + self.assertIn("colibri_scheduler_first_output_seconds_sum 0.5\n", metrics) + self.assertIn("colibri_scheduler_first_output_seconds_count 1\n", metrics) + self.assertIn("colibri_scheduler_engine_call_seconds_sum 3.0\n", metrics) + self.assertIn("colibri_scheduler_engine_call_seconds_count 1\n", metrics) + + def test_empty_output_does_not_count_but_tool_output_does(self): + text, tools = [], [] + def generate(prompt, maximum, temperature, top_p, on_text, **kwargs): + on_text("") + kwargs["on_tool"]("") + kwargs["on_tool"]("tool payload") + on_text("tail") + return {"completion_tokens": 2} + with patch.object(self.server.engine, "generate", side_effect=generate), \ + patch("openai_server.time.monotonic", side_effect=[10, 12, 15]): + self.server.generate("prompt", 4, 0, 1, text.append, on_tool=tools.append) + self.assertEqual(text, ["", "tail"]) + self.assertEqual(tools, ["", "tool payload"]) + metrics = self.server.scheduler.prometheus() + self.assertIn("colibri_scheduler_first_output_seconds_sum 2.0\n", metrics) + self.assertIn("colibri_scheduler_first_output_seconds_count 1\n", metrics) + + def test_error_and_cancellation_before_output_do_not_invent_first_output(self): + for error in (RuntimeError("failed"), ClientCancelled()): + with patch.object(self.server.engine, "generate", side_effect=error), \ + patch("openai_server.time.monotonic", side_effect=[10, 14]): + with self.assertRaises(type(error)): + self.server.generate("prompt", 4, 0, 1, lambda text: None) + metrics = self.server.scheduler.prometheus() + self.assertIn("colibri_scheduler_first_output_seconds_count 0\n", metrics) + self.assertIn("colibri_scheduler_engine_call_seconds_sum 8.0\n", metrics) + self.assertIn("colibri_scheduler_engine_call_seconds_count 2\n", metrics) + + def test_failure_after_output_keeps_both_observations(self): + def generate(prompt, maximum, temperature, top_p, on_text): + on_text("partial") + raise RuntimeError("failed after output") + with patch.object(self.server.engine, "generate", side_effect=generate), \ + patch("openai_server.time.monotonic", side_effect=[10, 11, 12]): + with self.assertRaises(RuntimeError): + self.server.generate("prompt", 4, 0, 1, lambda text: None) + metrics = self.server.scheduler.prometheus() + self.assertIn("colibri_scheduler_first_output_seconds_count 1\n", metrics) + self.assertIn("colibri_scheduler_engine_call_seconds_count 1\n", metrics) + + class SchedulerTest(unittest.TestCase): + def test_engine_failure_is_not_a_completed_request(self): + scheduler = GenerationScheduler() + with self.assertRaisesRegex(RuntimeError, "engine failed"): + with scheduler.admit(): + raise RuntimeError("engine failed") + stats = scheduler.snapshot() + self.assertEqual(stats["failed"], 1) + self.assertEqual(stats["completed"], 0) + self.assertEqual(stats["active"], 0) + with scheduler.admit(): + pass + self.assertEqual(scheduler.snapshot()["completed"], 1) + + def test_prometheus_histograms_measure_admission_and_slot_occupancy(self): + scheduler = GenerationScheduler() + with patch("openai_server.time.monotonic", side_effect=[10, 10.25, 10.25, 12.25]): + with scheduler.admit(): + active = scheduler.prometheus() + self.assertIn("colibri_scheduler_active 1\n", active) + self.assertIn("colibri_scheduler_slot_duration_seconds_count 0\n", active) + metrics = scheduler.prometheus() + self.assertIn("# TYPE colibri_scheduler_completed_total counter\n", metrics) + self.assertIn("colibri_scheduler_completed_total 1\n", metrics) + self.assertIn("colibri_scheduler_queue_wait_seconds_sum 0.25\n", metrics) + self.assertIn('colibri_scheduler_queue_wait_seconds_bucket{le="0.1"} 0\n', metrics) + self.assertIn('colibri_scheduler_queue_wait_seconds_bucket{le="0.5"} 1\n', metrics) + self.assertIn('colibri_scheduler_queue_wait_seconds_bucket{le="+Inf"} 1\n', metrics) + self.assertIn("colibri_scheduler_slot_duration_seconds_sum 2.0\n", metrics) + self.assertIn("colibri_scheduler_slot_duration_seconds_count 1\n", metrics) + + def test_failed_and_cancelled_admissions_are_timed(self): + scheduler = GenerationScheduler() + for error in (RuntimeError("failed"), ClientCancelled()): + with self.assertRaises(type(error)): + with scheduler.admit(): + raise error + metrics = scheduler.prometheus() + self.assertIn("colibri_scheduler_failed_total 1\n", metrics) + self.assertIn("colibri_scheduler_cancelled_total 1\n", metrics) + self.assertIn("colibri_scheduler_completed_total 0\n", metrics) + self.assertIn("colibri_scheduler_slot_duration_seconds_count 2\n", metrics) + def test_admits_up_to_capacity_without_serializing(self): scheduler = GenerationScheduler(max_queue=0, queue_timeout=1, capacity=2) with scheduler.admit() as first: @@ -550,6 +719,75 @@ def test_rejects_when_waiting_queue_is_full(self): self.assertEqual(caught.exception.code, "queue_full") self.assertEqual(scheduler.snapshot()["rejected"], 1) + def test_any_slot_request_uses_capacity_not_reserved_by_older_waiters(self): + scheduler = GenerationScheduler(capacity=2) + # State immediately after both slots are released, before the older + # slot-0 waiter reacquires the condition lock. + older = (object(), 0) + scheduler.queue.append(older) + with patch.object(scheduler.condition, "wait", side_effect=AssertionError("unused free slot")): + with scheduler.admit() as (_, slot): + self.assertEqual(slot, 1) + self.assertEqual(list(scheduler.queue), [older]) + self.assertEqual(scheduler.free_slots, {0, 1}) + + def test_older_any_slot_and_same_slot_waiters_keep_priority(self): + for older_slot, requested in ((None, None), (None, 1), (0, 0)): + with self.subTest(older_slot=older_slot, requested=requested): + scheduler = GenerationScheduler(capacity=2) + older = (object(), older_slot) + scheduler.queue.append(older) + with patch.object(scheduler.condition, "wait", + side_effect=lambda _timeout: scheduler.queue.remove(older)) as wait: + with scheduler.admit(slot=requested) as (_, slot): + self.assertEqual(slot, 0 if requested is None else requested) + wait.assert_called_once() + self.assertEqual(scheduler.snapshot()["queued"], 0) + + def test_full_queue_still_admits_unreserved_free_slot(self): + for requested in (None, 1): + with self.subTest(requested=requested): + scheduler = GenerationScheduler(max_queue=1, capacity=2) + with scheduler.admit(slot=0): + older = (object(), 0) + scheduler.queue.append(older) + try: + with scheduler.admit(slot=requested) as (_, slot): + self.assertEqual(slot, 1) + self.assertEqual(scheduler.snapshot()["active"], 2) + self.assertEqual(list(scheduler.queue), [older]) + finally: + scheduler.queue.remove(older) + self.assertEqual(scheduler.snapshot()["rejected"], 0) + self.assertEqual(scheduler.snapshot()["completed"], 2) + + def test_full_queue_does_not_bypass_older_slot_reservations(self): + for older_slot, requested in ((None, None), (None, 1), (0, 0)): + with self.subTest(older_slot=older_slot, requested=requested): + scheduler = GenerationScheduler(max_queue=1, capacity=2) + older = (object(), older_slot) + scheduler.queue.append(older) + with self.assertRaises(APIError) as caught: + with scheduler.admit(slot=requested): + self.fail("bypassed older waiter") + self.assertEqual(caught.exception.code, "queue_full") + self.assertEqual(list(scheduler.queue), [older]) + self.assertEqual(scheduler.snapshot()["rejected"], 1) + + def test_zero_queue_rejects_busy_pinned_slot_with_spare_capacity(self): + scheduler = GenerationScheduler(max_queue=0, queue_timeout=0.01, capacity=2) + with scheduler.admit(slot=0): + with self.assertRaises(APIError) as caught: + with scheduler.admit(slot=0): + self.fail("busy pinned slot admitted") + self.assertEqual(caught.exception.code, "queue_full") + with scheduler.admit(slot=1) as (_, slot): + self.assertEqual(slot, 1) + stats = scheduler.snapshot() + self.assertEqual((stats["rejected"], stats["timed_out"], stats["queued"]), (1, 0, 0)) + self.assertEqual((stats["admitted"], stats["completed"], stats["active"]), (2, 2, 0)) + self.assertIn("colibri_scheduler_queue_wait_seconds_count 2\n", scheduler.prometheus()) + def test_times_out_and_cancels_queued_requests(self): scheduler = GenerationScheduler(max_queue=2, queue_timeout=0.02) with scheduler.admit(): @@ -564,6 +802,81 @@ def test_times_out_and_cancels_queued_requests(self): self.assertEqual(stats["timed_out"], 1) self.assertEqual(stats["cancelled"], 1) + def test_queue_deadline_wins_when_slot_becomes_free(self): + for released_at in (0.9, 1.0, 1.1): + with self.subTest(released_at=released_at): + scheduler = GenerationScheduler(queue_timeout=1) + now = [0.0] + with patch("openai_server.time.monotonic", side_effect=lambda: now[0]): + holder = scheduler.admit() + holder.__enter__() + def release(_timeout): + now[0] = released_at + holder.__exit__(None, None, None) + with patch.object(scheduler.condition, "wait", side_effect=release): + if released_at < 1: + with scheduler.admit(): + pass + else: + with self.assertRaises(APIError) as caught: + with scheduler.admit(): + pass + self.assertEqual(caught.exception.code, "queue_timeout") + stats = scheduler.snapshot() + expected = 2 if released_at < 1 else 1 + self.assertEqual((stats["admitted"], stats["completed"]), (expected, expected)) + self.assertEqual((stats["active"], stats["queued"], stats["timed_out"]), + (0, 0, int(released_at >= 1))) + self.assertIn(f"colibri_scheduler_slot_duration_seconds_count {expected}\n", + scheduler.prometheus()) + with scheduler.admit(): + pass + + def test_cancelled_request_does_not_acquire_a_free_slot(self): + scheduler = GenerationScheduler() + with self.assertRaises(ClientCancelled): + with scheduler.admit(lambda: True): + self.fail("cancelled request admitted") + stats = scheduler.snapshot() + self.assertEqual((stats["active"], stats["queued"], stats["admitted"], stats["cancelled"]), + (0, 0, 0, 1)) + self.assertIn("colibri_scheduler_queue_wait_seconds_count 0\n", scheduler.prometheus()) + with scheduler.admit(): + pass + + def test_cancellation_wins_when_a_waiting_slot_becomes_free(self): + scheduler = GenerationScheduler(queue_timeout=1) + waiting = threading.Event() + cancelled = threading.Event() + outcomes = [] + holder = scheduler.admit() + holder.__enter__() + def is_cancelled(): + waiting.set() + return cancelled.is_set() + def run(): + try: + with scheduler.admit(is_cancelled): + outcomes.append("admitted") + except ClientCancelled: + outcomes.append("cancelled") + thread = threading.Thread(target=run) + thread.start() + observed = waiting.wait(1) + # Publish cancellation and release capacity under the same lock, so + # the waiter must observe both on its next scheduling pass. + with scheduler.condition: + cancelled.set() + holder.__exit__(None, None, None) + thread.join(2) + self.assertTrue(observed) + self.assertFalse(thread.is_alive()) + self.assertEqual(outcomes, ["cancelled"]) + stats = scheduler.snapshot() + self.assertEqual((stats["admitted"], stats["completed"], stats["cancelled"]), (1, 1, 1)) + self.assertEqual((stats["active"], stats["queued"]), (0, 0)) + self.assertIn("colibri_scheduler_slot_duration_seconds_count 1\n", scheduler.prometheus()) + def test_counts_admitted_client_cancellation_without_completion(self): scheduler = GenerationScheduler(max_queue=0, queue_timeout=1) with self.assertRaises(ClientCancelled): @@ -680,6 +993,11 @@ def __init__(self, on_write): self.returncode = None def write(self, data): + # `_write_all` (production) hands every write a `memoryview` slice, + # first write included -- the fakes below pattern-match frame bytes + # (`frame.split()`, equality against a literal), so normalise here + # rather than asking each one to know about the view. + data = bytes(data) self.writes.append(data) self.on_write(self, data) return len(data) @@ -1161,6 +1479,101 @@ def respond(process, frame): self.assertEqual(process.writes[-1].split(), [b"STOP", request_id]) +def _capture_frames(body, path="/v1/completions"): + """Sends `body` to `path` against a FakeProcess-backed Engine/APIServer + and returns (status, parsed_response, frames_written_to_the_engine). + Shared by the test classes below so the harness lives in one place. + """ + frames = [] + + def respond(process, frame): + frames.append(frame) + rid = frame.split()[1] + process.stdout.feed(b"DATA " + rid + b" 5\nHello\n") + process.stdout.feed(b"DONE " + rid + b" STAT 1 2.5 0 1.0 4 0\n") + + process = FakeProcess(respond) + with patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + server = APIServer(("127.0.0.1", 0), engine, "test-model", "secret", 16) + thread = threading.Thread(target=server.serve_forever, args=(0.01,), daemon=True) + thread.start() + try: + data = json.dumps(body).encode() + headers = {"Authorization": "Bearer secret", "Content-Type": "application/json"} + request = Request(f"http://127.0.0.1:{server.server_port}{path}", + data=data, headers=headers) + with urlopen(request, timeout=2) as response: + status = response.status + parsed = json.load(response) + finally: + server.scheduler.close() + server.shutdown() + server.server_close() + thread.join(timeout=2) + engine.close() + return status, parsed, frames + + +class BaseWireContractTest(unittest.TestCase): + """Pins the SUBMIT frame written for a request that uses none of the + optional fields; any change here is a wire-format change and must be + deliberate. + """ + + def test_no_new_fields_request_emits_the_base_submit_header(self): + _, _, frames = _capture_frames( + {"model": "test-model", "prompt": "Complete me", "temperature": 0, "max_tokens": 4}) + self.assertEqual(frames, [b"SUBMIT 1 0 11 4 0 0.9\nComplete me\n"]) + + +class SeedOptionTest(unittest.TestCase): + """`generation_options()` accepts a `seed` field without raising.""" + + def test_seed_is_accepted_by_generation_options(self): + generation_options({"seed": 1234, "prompt": "hi"}, 16) # must not raise + + +class SeedWireFrameTest(unittest.TestCase): + """`seed` is accepted and ignored. A stub-response equality check alone + is vacuous here (the scripted `respond` closure inside `_capture_frames` + always returns the same canned text regardless of any request field) -- + the real proof is that the byte-exact SUBMIT frames the dispatcher + writes to the engine process (see DispatcherTest above) never carry the + seed value at all, seeded or not. + """ + + def test_seed_accepted_and_absent_from_submit_frame(self): + base = {"model": "test-model", "prompt": "Complete me", "temperature": 0, "max_tokens": 4} + status_plain, body_plain, frames_plain = _capture_frames(base) + status_seeded, body_seeded, frames_seeded = _capture_frames({**base, "seed": 1234}) + self.assertEqual(status_plain, 200) + self.assertEqual(status_seeded, 200) + self.assertEqual(body_seeded["choices"][0], body_plain["choices"][0]) + # Each call uses a freshly-constructed Engine, so both first requests are + # assigned request id "1" -- the wire frames are directly byte-comparable, + # no field needs normalizing. Comparing the whole frame list (not just + # the first frame) closes "reaches no wire frame" literally: if `seed` + # ever leaked onto any frame, this equality would break. + self.assertEqual(frames_seeded, frames_plain) + + def test_seed_accepted_and_absent_from_chat_submit_frame(self): + # Same proof as above, on /v1/chat/completions: generation_options() + # is shared by both endpoints, but the SUBMIT frame is built from + # the chat-rendered prompt, so this is not implied by the completions + # case above -- a divergence between the two call sites would only + # show up here. + base = {"model": "test-model", "messages": [{"role": "user", "content": "Hi"}], + "temperature": 0, "max_tokens": 4} + status_plain, body_plain, frames_plain = _capture_frames(base, path="/v1/chat/completions") + status_seeded, body_seeded, frames_seeded = _capture_frames( + {**base, "seed": 1234}, path="/v1/chat/completions") + self.assertEqual(status_plain, 200) + self.assertEqual(status_seeded, 200) + self.assertEqual(body_seeded["choices"][0], body_plain["choices"][0]) + self.assertEqual(frames_seeded, frames_plain) + + class CapSentinelShimTest(unittest.TestCase): # #379 cap-sentinel shim, arch-keyed (#386 r2, F3): an absent cap is # "platform-auto" only for the glm engine (colibri.c coli_resolve_cap); @@ -1275,6 +1688,40 @@ def test_engine_consumes_planned_cap_without_leaking_private_env(self): self.assertNotIn("COLI_PLAN_CAP", child_env) self.assertEqual(child_env["KEEP"], "yes") + def test_v41_ram_plan_reaches_engine_argv(self): + model = self._model("deepseek_v41") + for settings, expected_ram in (({"RAM_GB": "120", "CTX": "8192"}, 120), + ({"CTX": "8192"}, 0), + ({"RAM_GB": "auto", "CTX": "8192"}, 0)): + with self.subTest(settings=settings): + process = FakeProcess(lambda _process, _frame: None) + with patch("resource_plan.build_plan", return_value={ + "tiers": {"ram": {"cache_slots_per_layer": 96}}}) as planner, \ + patch("openai_server.subprocess.Popen", return_value=process) as popen: + engine = Engine("deepseek_v41", model, env=settings) + engine.close() + planner.assert_called_once_with(model, ram_gb=expected_ram, + context=8192, gpu_indices=[]) + self.assertEqual(popen.call_args[0][0], ["deepseek_v41", "96"]) + + def test_v41_explicit_and_calibrated_caps_bypass_planning(self): + for cap, env, expected in ((7, {}, 7), (0, {}, 0), + (None, {"COLI_PROFILE_CAP": "12"}, 12), + (None, {"COLI_PLAN_CAP": "24"}, 24)): + with self.subTest(cap=cap, env=env), patch("resource_plan.build_plan") as planner: + self.assertEqual(cap_for_arch("deepseek_v41", cap, env, model="model"), + expected) + planner.assert_not_called() + + def test_v41_insufficient_ram_refuses_before_spawning(self): + model = self._model("deepseek_v41") + with patch("resource_plan.build_plan", return_value={ + "tiers": {"ram": {"cache_slots_per_layer": 0}}}), \ + patch("openai_server.subprocess.Popen") as popen: + with self.assertRaisesRegex(ValueError, "one expert slot"): + Engine("deepseek_v41", model, env={"RAM_GB": "8"}) + popen.assert_not_called() + def test_model_arch_reads_model_type(self): self.assertEqual(model_arch(self._model("glm_moe_dsa")), "glm") self.assertEqual(model_arch(self._model("inkling")), "inkling") @@ -1346,6 +1793,41 @@ def test_lists_models_and_checks_auth(self): self.addCleanup(caught.exception.close) self.assertEqual(caught.exception.code, 401) + def test_metrics_counts_http_engine_failure_without_success(self): + before = self.server.scheduler.snapshot() + with patch.object(self.engine, "generate", side_effect=RuntimeError("injected failure")): + with self.assertRaises(HTTPError) as caught: + self.request("/v1/chat/completions", { + "model": "test-model", + "messages": [{"role": "user", "content": "hello"}], "max_tokens": 1}) + self.addCleanup(caught.exception.close) + self.assertEqual(caught.exception.code, 500) + after = self.server.scheduler.snapshot() + self.assertEqual(after["failed"], before["failed"] + 1) + self.assertEqual(after["completed"], before["completed"]) + with self.request("/metrics") as response: + text = response.read().decode() + self.assertIn(f'colibri_scheduler_failed_total {after["failed"]}\n', text) + self.assertIn("colibri_scheduler_active 0\n", text) + + def test_metrics_exposes_prometheus_text_with_auth(self): + with self.request("/metrics") as response: + self.assertEqual(response.headers["Content-Type"], + "text/plain; version=0.0.4; charset=utf-8") + self.assertEqual(response.headers["Cache-Control"], "no-store") + text = response.read().decode() + self.assertIn("colibri_scheduler_capacity 2\n", text) + self.assertIn("# TYPE colibri_scheduler_queue_wait_seconds histogram\n", text) + for key in ("wrong", ""): + with self.assertRaises(HTTPError) as caught: + self.request("/metrics", key=key) + self.addCleanup(caught.exception.close) + self.assertEqual(caught.exception.code, 401) + with self.assertRaises(HTTPError) as caught: + urlopen(self.base + "/metrics", timeout=2) + self.addCleanup(caught.exception.close) + self.assertEqual(caught.exception.code, 401) + def test_health_reports_scheduler_and_kv_slots(self): with self.request("/health") as response: health = json.load(response) @@ -1354,6 +1836,17 @@ def test_health_reports_scheduler_and_kv_slots(self): self.assertIn("queued", scheduler) self.assertEqual(health["kv_slots"], 2) + def test_health_reports_the_continuation_switch(self): + """The web UI shows Continue only when this is true: with the switch off a + trailing assistant turn is answered fresh, which a Continue button would + present as a resumption. Unauthed probes keep the bare liveness shape.""" + for value, expected in (("1", True), ("0", False)): + with patch.dict(os.environ, {"COLI_CONTINUE_ASSISTANT": value}), \ + self.request("/health") as response: + self.assertIs(json.load(response)["continue_assistant"], expected) + with urlopen(self.base + "/health", timeout=2) as response: + self.assertNotIn("continue_assistant", json.load(response)) + def test_profile_requires_auth(self): """/profile is served before require_auth(), so it needs its own gate. @@ -1626,6 +2119,64 @@ def test_a_well_formed_forced_tool_choice_still_runs(self): self.assertEqual(response.status, 200) self.assertIn("You must call the function `search`", self.engine.calls[-1][0]) + def test_tool_with_a_non_object_function_is_a_client_error(self): + """{"type": "function", "function": "search"} on tools[] answered HTTP 500. + + generation_options already has the 400, but /v1/chat/completions renders + first. GLM, GLM-5.3, DeepSeek V4 and V4.1 did fn.items() on a string + and the AttributeError became do_POST's 500 "The colibri engine failed + to process the request." for a request the engine never saw. + """ + for arch in ("glm", "glm53", "kimi", "deepseek_v4", "deepseek_v41", + "olmoe", "inkling", "qwen36", "qwen38"): + for function in ("search", ["search"], 5): + with self.subTest(arch=arch, function=function): + with patch("openai_server.ARCH", arch): + with self.assertRaises(HTTPError) as caught: + self.request("/v1/chat/completions", { + "model": "test-model", + "messages": [{"role": "user", "content": "hi"}], + "tools": [{"type": "function", + "function": function}]}) + self.addCleanup(caught.exception.close) + self.assertEqual(caught.exception.code, 400) + with self.assertRaises(HTTPError) as caught: + self.request("/v1/completions", { + "model": "test-model", "prompt": "hi", + "tools": [{"type": "function", "function": "search"}]}) + self.addCleanup(caught.exception.close) + self.assertEqual(caught.exception.code, 400) + + def test_a_well_formed_tool_function_still_runs(self): + """The read above must not change the shape clients actually send.""" + with self.request("/v1/chat/completions", { + "model": "test-model", + "messages": [{"role": "user", "content": "hi"}], + "tools": [{"type": "function", + "function": {"name": "search"}}]}) as response: + self.assertEqual(response.status, 200) + self.assertIn('"name": "search"', self.engine.calls[-1][0]) + + def test_tool_call_arguments_that_are_not_an_object_do_not_fail_the_request(self): + """A replayed tool call with `arguments: "[1, 2]"` answered HTTP 500. + + render_chat reached `(args or {}).items()` with a parsed list, and the + AttributeError became do_POST's catch-all 500 "The colibri engine failed + to process the request." -- which OpenAI SDKs retry, against an engine + that was never asked anything. + """ + body = {"model": "test-model", "messages": [ + {"role": "user", "content": "run it"}, + {"role": "assistant", "content": "", "tool_calls": [ + {"id": "x", "type": "function", + "function": {"name": "fn", "arguments": "[1, 2]"}}]}, + {"role": "tool", "tool_call_id": "x", "content": "done"}, + {"role": "user", "content": "and now?"}, + ]} + with self.request("/v1/chat/completions", body) as response: + self.assertEqual(response.status, 200) + self.assertEqual(json.load(response)["object"], "chat.completion") + class ClientHangupTest(unittest.TestCase): """A client that disconnects mid-response must not print a traceback. @@ -2054,6 +2605,329 @@ def test_rejects_tool_choice_without_tools(self): generation_options({"messages": [], "tool_choice": "required"}, 128) +class TrailingAssistantTurnTest(unittest.TestCase): + """A trailing `assistant` message is a turn to CONTINUE, not one already finished. + + The gateway used to fold it into a completed turn and append a fresh generation cue, + so the model wrote a second assistant turn and the client's opening was dropped. These + pin what replaced that: the switch that turns continuation on, what the prompt looks + like when it is on, and what is refused rather than silently reshaped. + + The switch is COLI_CONTINUE_ASSISTANT and not a request field on purpose -- a body + extension would only be reachable by hand-written JSON, and the clients that want this + send a message list and nothing else. + """ + + OPEN_TURN = [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + + @staticmethod + def on(): + return patch.dict(os.environ, {"COLI_CONTINUE_ASSISTANT": "1"}) + + @staticmethod + def off(): + return patch.dict(os.environ, {"COLI_CONTINUE_ASSISTANT": "0"}) + + def test_switch_on_leaves_the_turn_open(self): + with self.on(), patch("openai_server.ARCH", "glm53"): + self.assertFalse(resolve_generation_prompt(self.OPEN_TURN, {})) + prompt = render_chat_glm53(self.OPEN_TURN, enable_thinking=True, + add_generation_prompt=False) + # ends INSIDE the turn, on the client's opening -- no new cue after it + self.assertTrue(prompt.endswith("La capitale e'"), prompt[-60:]) + self.assertFalse(prompt.endswith("<|assistant|>")) + self.assertEqual(prompt.count("<|assistant|>"), 1) + + def test_default_is_on(self): + """Continuation is the default now: unset (the ordinary deployment) continues a + trailing assistant turn. Only COLI_CONTINUE_ASSISTANT=0 turns it off.""" + with patch.dict(os.environ, {}, clear=False), patch("openai_server.ARCH", "glm53"): + os.environ.pop("COLI_CONTINUE_ASSISTANT", None) + self.assertFalse(resolve_generation_prompt(self.OPEN_TURN, {})) + + def test_switch_off_restores_old_behaviour(self): + """COLI_CONTINUE_ASSISTANT=0 is the off-switch: a deployment that sets it behaves + exactly as the gateway did before continuation existed -- append a fresh cue.""" + with self.off(), patch("openai_server.ARCH", "glm53"): + self.assertTrue(resolve_generation_prompt(self.OPEN_TURN, {})) + self.assertEqual(render_chat_glm53(self.OPEN_TURN, enable_thinking=True), + render_chat_glm53(self.OPEN_TURN, enable_thinking=True, + add_generation_prompt=True)) + self.assertTrue(render_chat_glm53(self.OPEN_TURN, + enable_thinking=True).endswith("<|assistant|>")) + + def test_no_trailing_assistant_turn_is_untouched_either_way(self): + messages = [{"role": "user", "content": "a"}] + for switch in (self.on(), self.off()): + with switch, patch("openai_server.ARCH", "glm53"): + self.assertTrue(resolve_generation_prompt(messages, {})) + + def test_rejects_trailing_whitespace(self): + """The template strips it, so the model would resume from different bytes than the + ones sent -- the same reason Anthropic's own prefill validator refuses it.""" + messages = [{"role": "user", "content": "a"}, + {"role": "assistant", "content": "La capitale e' "}] + with self.on(), patch("openai_server.ARCH", "glm53"): + with self.assertRaises(APIError): + resolve_generation_prompt(messages, {}) + + def test_rejects_an_empty_continuation(self): + """An empty one ends the prompt on a closed, empty : the + out-of-distribution position #1327 removed.""" + for empty in ("", " ", None): + messages = [{"role": "user", "content": "a"}, + {"role": "assistant", "content": empty}] + with self.on(), patch("openai_server.ARCH", "glm53"): + with self.assertRaises(APIError): + resolve_generation_prompt(messages, {}) + + def test_rejects_tools_and_tool_calls(self): + with self.on(), patch("openai_server.ARCH", "glm53"): + with self.assertRaises(APIError): + resolve_generation_prompt(self.OPEN_TURN, {"tools": ORDER_TOOL}) + calls = [{"role": "user", "content": "a"}, + {"role": "assistant", "content": "x", "tool_calls": [ + {"type": "function", "function": {"name": "f", "arguments": "{}"}}]}] + with self.assertRaises(APIError): + resolve_generation_prompt(calls, {}) + + def test_unimplemented_family_passes_through(self): + """Continuation is on by default, so a family whose renderer has no open-turn shape + yet must render as before -- append the cue -- not reject a request nobody opted into. + Every shipped family is now in CONTINUATION_FAMILIES (Kimi K3 too, via its C `C` + record), so the backstop is exercised with a hypothetical future arch: it must pass + through, not error, the day a new renderer lands before its open-turn shape does.""" + self.assertNotIn("future_family", CONTINUATION_FAMILIES) + with self.on(), patch("openai_server.ARCH", "future_family"): + self.assertTrue(resolve_generation_prompt(self.OPEN_TURN, {})) + + def test_continuation_open_turn_deepseek_v4(self): + """deepseek_v4 has no authoritative vendored jinja template to diff against: + render_chat_v4 is pinned to the official encoding_dsv4.py, and the community + reap-150b template on the Hub diverges on the reasoning-block convention (a bare + for a direct answer vs ). So this is the expected-string + pin the maintainer allows for such families -- the open turn is pinned to a literal + here, with the past-turn-minus-EOS invariant kept as an added check, in both + thinking modes.""" + EOS = "<|end▁of▁sentence|>" + ASSISTANT = "<|Assistant|>" + # The literal open turn, written out so the test does not lean on another renderer + # call for its only expected value -- a bug that corrupts render_chat_v4 and the + # continuation path identically would slip past the comparison below but not this. + # Identical in both thinking modes: the open turn is the shape of the PAST turn, + # which carries no generation cue for enable_thinking to steer. + EXPECTED_OPEN = ("<|begin▁of▁sentence|><|User|>capitale della Francia?" + "<|Assistant|>La capitale e'") + for enable_thinking in (True, False): + cue = ASSISTANT + ("" if enable_thinking else "") + with patch("openai_server.ARCH", "deepseek_v4"): + normal = render_chat_v4(self.OPEN_TURN, enable_thinking=enable_thinking) + cont = render_chat_for_arch(self.OPEN_TURN, enable_thinking=enable_thinking, + add_generation_prompt=False) + self.assertEqual(cont, EXPECTED_OPEN) + # and the invariant tying it to the normal render: past turn without its EOS + cue + self.assertEqual(normal, cont + EOS + cue) + self.assertTrue(cont.endswith("La capitale e'"), cont[-40:]) + self.assertFalse(cont.endswith(EOS)) + + def test_continuation_open_turn_deepseek_v41(self): + """deepseek_v41, like deepseek_v4, is pinned to encoding.py rather than a diffable + vendored jinja template, so it gets the same expected-string pin: the open turn is a + literal here, with the past-turn-minus-EOS invariant kept as an added check, in both + thinking modes. + + Unlike v4, the v41 open turn is NOT identical across the modes: thinking-on carries the + <|System|> Reasoning Effort line and closes the continued turn's (empty) reasoning as + ; thinking-off has neither. The empty is the SAFE + position #1327 is about, not the out-of-distribution one -- it is followed by the + client's content ('La capitale e''), never left dangling at the end of the prompt, + which resolve_generation_prompt guarantees by refusing an empty continuation.""" + EOS = "<|end▁of▁sentence|>" + ASSISTANT = "<|Assistant|>" + EFFORT = ("<|System|>Reasoning Effort: 75 (range 1-100, the higher the value, the " + "more thorough the reasoning)\n\n") + # The literal open turns, written out so the test does not lean on another renderer + # call for its only expected value (see the deepseek_v4 test), one per thinking mode. + EXPECTED_OPEN = { + True: ("<|begin▁of▁sentence|>" + EFFORT + "<|User|>capitale della Francia?" + "<|Assistant|>La capitale e'"), + False: ("<|begin▁of▁sentence|><|User|>capitale della Francia?" + "<|Assistant|>La capitale e'"), + } + for enable_thinking in (True, False): + cue = ASSISTANT + ("" if enable_thinking else "") + with patch("openai_server.ARCH", "deepseek_v41"): + normal = render_chat_dsv41(self.OPEN_TURN, enable_thinking=enable_thinking) + cont = render_chat_for_arch(self.OPEN_TURN, enable_thinking=enable_thinking, + add_generation_prompt=False) + self.assertEqual(cont, EXPECTED_OPEN[enable_thinking]) + # and the invariant tying it to the normal render: past turn without its EOS + cue + self.assertEqual(normal, cont + EOS + cue) + self.assertTrue(cont.endswith("La capitale e'"), cont[-40:]) + self.assertFalse(cont.endswith(EOS)) + + def test_continuation_open_turn_inkling(self): + """inkling uses its own markers, and render_chat_inkling deliberately deviates from + the template's generation cue (it prefills <|content_text|> in the thinking-off case + to force content mode, and defaults thinking off), so it is pinned with an + expected-string test: the open turn is pinned to a literal here, with the + past-turn-minus-terminators invariant kept as an added check, in both thinking modes.""" + END = "<|end_message|><|content_model_end_sampling|>" + for enable_thinking in (True, False): + # eff is 0.9 with thinking on, 0.0 off; the off cue prefills the content channel + eff = "0.9" if enable_thinking else "0" + cue = "<|message_model|>" + ("" if enable_thinking else "<|content_text|>") + # The literal open turn, written out so the test does not lean on another renderer + # call for its only expected value (see the deepseek_v4 test). The two modes differ + # only in the system effort line; both end on the prefilled content channel. + expected = ("<|message_system|><|content_text|>Thinking effort level: " + eff + + "<|end_message|><|message_user|><|content_text|>capitale della Francia?" + "<|end_message|><|message_model|><|content_text|>La capitale e'") + with patch("openai_server.ARCH", "inkling"): + normal = render_chat_inkling(self.OPEN_TURN, enable_thinking=enable_thinking) + cont = render_chat_for_arch(self.OPEN_TURN, enable_thinking=enable_thinking, + add_generation_prompt=False) + self.assertEqual(cont, expected) + # and the invariant tying it to the normal render: past turn without terminators + cue + self.assertEqual(normal, cont + END + cue) + self.assertTrue(cont.endswith("La capitale e'"), cont[-40:]) + self.assertFalse(cont.endswith("<|end_message|>")) + + def test_only_the_final_assistant_turn_is_opened(self): + """Across every continuation family: only the TRAILING assistant turn is opened, and + each earlier turn renders exactly as it does in a completed conversation. The + single-turn fixtures elsewhere cannot see this -- their assistant turn is trivially + last -- but the terminator is dropped by a per-family `index == last` check, written + out by hand in each renderer; a family that lost that check would open every assistant + turn, and nothing else in the suite would notice. + + Family-agnostic on purpose, because the open-turn SHAPE is not uniform: qwen36 injects + an empty , GLM carries no per-turn terminator at all, ChatML drops an + <|im_end|>. What IS uniform is that the completed render (a fresh generation cue + appended) and the open render share their entire prefix up to the final turn -- so the + SECOND user turn must survive into their common prefix. If the first assistant turn + lost its terminator in the open render, that prefix would break right after it, before + this text. Checked in both thinking modes; the set drives the loop so a newly added + family is covered the day it joins. + + Kimi K3 is excluded: render_chat_for_arch returns its engine-side K3CHAT1 wire, not a + string prompt, so this string-level invariant doesn't apply -- its open turn is pinned + at the token level in tests/test_k3_chat_tools.c against the tiny tokenizer instead.""" + multi = [{"role": "user", "content": "1+1?"}, + {"role": "assistant", "content": "2"}, + {"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + for arch in sorted(CONTINUATION_FAMILIES): + if arch == "kimi": + continue + for enable_thinking in (True, False): + where = (arch, enable_thinking) + with patch("openai_server.ARCH", arch): + completed = render_chat_for_arch(multi, enable_thinking=enable_thinking) + opened = render_chat_for_arch(multi, enable_thinking=enable_thinking, + add_generation_prompt=False) + common = os.path.commonprefix([opened, completed]) + # the prior assistant turn (and its terminator) rendered identically: the turn + # AFTER it survives into the shared prefix + self.assertIn("capitale della Francia?", common, where) + # and only the last turn is open -- the prompt ends on the client's opening + self.assertTrue(opened.endswith("La capitale e'"), (where, opened[-40:])) + + def test_splitter_starts_in_content_mode_on_a_continued_turn(self): + """Measured on glm53 int4, CPU: content '' with reasoning_chars 10 and 109, + clean stop, and a byte-correct open turn on the wire. The model was fine; the + splitter was primed from enable_thinking alone, so it waited for a the + prompt had already passed and filed the whole answer as reasoning. + + starts_in_reasoning's own docstring is about exactly this invariant -- a continued + turn is the third state it did not model. Nothing else in the suite covers it: + every other thinking test runs against a prompt with the generation cue appended, + where enable_thinking really does say where the block was left. + + Pinned to glm53 because that is the only family the switch serves. Left on the + module default (ARCH = "glm") this exercised the continuation path on a family + resolve_generation_prompt refuses, and passed for the wrong reason. + + The family rule itself -- whether a NEW turn starts inside the block -- belongs to + #1278 and is pinned by its own test; asserted here it would only duplicate it. What + this test owns is the axis crossing it: a continued turn opens no block, whatever + the family rule says.""" + with patch("openai_server.ARCH", "glm53"): + self.assertTrue(starts_in_reasoning(True)) # cue: block left open + self.assertFalse(starts_in_reasoning(True, add_generation_prompt=False)) + self.assertFalse(starts_in_reasoning(False, add_generation_prompt=False)) + + # what the engine actually returns after "...The capital of France is": no + # markers, because the turn's is already behind it in the prompt + emitted = " Paris." + reasoning, answer = split_thinking_reply(emitted, enable_thinking=True, + add_generation_prompt=False) + self.assertEqual(answer, emitted) + self.assertEqual(reasoning, "") + # and the bug it replaces, so this test fails if the priming is ever reverted + reasoning, answer = split_thinking_reply(emitted, enable_thinking=True, + add_generation_prompt=True) + self.assertEqual(answer, "") + self.assertEqual(reasoning, emitted) + + def test_rejects_a_continuation_with_nothing_to_continue_from(self): + lone = [{"role": "assistant", "content": "La capitale e'"}] + with self.on(), patch("openai_server.ARCH", "glm53"): + with self.assertRaises(APIError): + resolve_generation_prompt(lone, {}) + + +class TrailingAssistantEndToEndTest(unittest.TestCase): + """End-to-end, through the real HTTP handler and a fake engine (no model weights): a + request whose last message is an `assistant` turn must reach the engine as a CONTINUATION + prompt -- ending on the client's own opening, no generation cue appended -- and the text + the engine generates must come back as the message `content`. + + The TrailingAssistantTurnTest cases pin the prompt string in isolation; none of them prove + the wiring from an HTTP request through resolve_generation_prompt to the engine and back, + which a serve() refactor could silently drop. /v1/messages is a translation layer onto the + same engine path (not a second one), so /v1/chat/completions covers both endpoints. The + generated text arriving as content (not reasoning_content) also shows the continued turn + primes the splitter into content mode over the wire -- the third state #1327 fixed.""" + + OPEN = {"model": "test-model", + "messages": [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}]} + + def _server(self, engine): + server = APIServer(("127.0.0.1", 0), engine, "test-model") + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + self.addCleanup(thread.join, 2) + self.addCleanup(server.server_close) + self.addCleanup(server.shutdown) + self.addCleanup(server.scheduler.close) + return server + + def test_continuation_reaches_the_engine_and_returns_as_content(self): + engine = FakeEngine() + with patch("openai_server.ARCH", "glm53"), \ + patch.dict(os.environ, {"COLI_CONTINUE_ASSISTANT": "1"}): + server = self._server(engine) + conn = http.client.HTTPConnection("127.0.0.1", server.server_port, timeout=3) + self.addCleanup(conn.close) + conn.request("POST", "/v1/chat/completions", body=json.dumps(self.OPEN), + headers={"Content-Type": "application/json"}) + response = conn.getresponse() + status, payload = response.status, response.read() + self.assertEqual(status, 200, payload) + # the engine saw the continuation prompt: it ends on the client's opening, no cue + self.assertEqual(len(engine.calls), 1) + prompt = engine.calls[0][0] + self.assertTrue(prompt.endswith("La capitale e'"), prompt[-60:]) + self.assertFalse(prompt.endswith("<|assistant|>")) + # and the generated text comes back as content, not misfiled as reasoning + message = json.loads(payload)["choices"][0]["message"] + self.assertEqual(message["content"], "Héllo") + self.assertFalse(message.get("reasoning_content")) + + class AllowedHostsTest(unittest.TestCase): """#597: the DNS-rebinding guard must accept operator-trusted reverse-proxy Host values, while still rejecting everything else by default.""" @@ -2580,6 +3454,44 @@ def test_anthropic_stream_announces_close_framing(self): self.assertIn("event: message_stop", raw) self.assertNotIn("", raw) + def test_stream_exit_stops_keepalive_before_releasing_slot(self): + class CancelledEngine(_ExplodingEngine): + def generate(self, *args, **kwargs): + try: + return super().generate(*args, **kwargs) + except RuntimeError: + raise ClientCancelled() + + original_thread = threading.Thread + for path in ("/v1/chat/completions", "/v1/messages"): + for engine_type, outcome in ((_ExplodingEngine, "failed"), + (CancelledEngine, "cancelled")): + with self.subTest(path=path, outcome=outcome): + pumps = [] + def thread_factory(*args, **kwargs): + thread = original_thread(*args, **kwargs) + target = kwargs.get("target") + if getattr(target, "__name__", "") in ("_keepalive", "keepalive"): + stop = next(cell.cell_contents for cell in target.__closure__ + if isinstance(cell.cell_contents, threading.Event)) + pumps.append((thread, stop)) + return thread + server = self._server(engine_type()) + try: + with patch.object(threading, "Thread", side_effect=thread_factory): + status, _ = self._post(self._conn(server), + dict(self.CHAT, stream=True, max_tokens=16), path=path) + self.assertEqual(status, 200) + self.assertEqual(len(pumps), 1) + self.assertFalse(pumps[0][0].is_alive(), "keepalive survived stream exit") + stats = server.scheduler.snapshot() + self.assertEqual(stats[outcome], 1) + self.assertEqual((stats["active"], stats["completed"]), (0, 0)) + finally: + for thread, stop in pumps: + stop.set() + thread.join(2) + def test_engine_failure_after_commit_does_not_splice_a_second_response(self): """Once the 200 is out, a 500 status line would land inside the event stream.""" server = self._server(_ExplodingEngine()) @@ -2852,5 +3764,475 @@ def test_missing_fields_do_not_crash_the_message(self): self.assertIn("the context", text) +class _ShortWritingStream: + """A fake stdin that hands back at most `chunk` bytes per write() call, + forcing `_write_all` to loop -- the production pipe does this on a + signal landing mid-write or a full pipe buffer on a large IMAGE frame.""" + + def __init__(self, chunk=3): + self.chunk = chunk + self.received = bytearray() + + def write(self, data): + piece = bytes(data)[:self.chunk] + self.received.extend(piece) + return len(piece) + + +class _ScriptedStream: + """A fake stdin whose write() answers each entry in `script` in turn -- + an int number of bytes actually taken, or `None` for "took nothing" -- + then takes everything it is offered once the script runs out.""" + + def __init__(self, script): + self.script = list(script) + self.received = bytearray() + self.calls = [] + + def write(self, data): + data = bytes(data) + self.calls.append(data) + answer = self.script.pop(0) if self.script else len(data) + if answer is None: + return None + taken = data[:answer] + self.received.extend(taken) + return len(taken) + + +class _CountingLock: + """A `threading.Lock`-alike that counts `with` acquisitions -- used to + pin that IMAGE and SUBMIT share exactly one `write_lock` acquisition.""" + + def __init__(self): + self._lock = threading.Lock() + self.acquisitions = 0 + + def __enter__(self): + self.acquisitions += 1 + return self._lock.__enter__() + + def __exit__(self, *exc_info): + return self._lock.__exit__(*exc_info) + + +class WriteAllTest(unittest.TestCase): + """`_write_all` (extracted from the inline server->engine stdin writes): + loop on a short write until the whole frame is sent, and fail closed -- + never spin -- on the two shapes that are not progress. + + `_write_all` is imported locally in each test method here, not at module + scope: it does not exist on base, and a module-level import of it would + make this whole test file fail to collect when overlaid on base product + code (the procedure the base-pin tests below rely on).""" + + def test_short_writes_reassemble_to_the_full_frame(self): + from openai_server import _write_all + stream = _ShortWritingStream(chunk=3) + frame = b"SUBMIT 1 0 5 3 0.25 0.9\nHello\n" + _write_all(stream, frame, "SUBMIT") + self.assertEqual(bytes(stream.received), frame) + + def test_a_short_first_write_still_receives_the_correct_remainder(self): + from openai_server import _write_all + # The first call must not be special-cased to the unsliced buffer: + # every call, including the first, offers a memoryview starting at + # the bytes not yet sent. + stream = _ScriptedStream([2]) + _write_all(stream, b"CANCEL 42\n", "CANCEL") + self.assertEqual(bytes(stream.received), b"CANCEL 42\n") + self.assertEqual(len(stream.calls), 2) # the short first write forced a second + + def test_none_return_fails_closed_instead_of_spinning(self): + from openai_server import _write_all + stream = _ScriptedStream([None]) + with self.assertRaisesRegex(RuntimeError, "failed to write SUBMIT to the engine"): + _write_all(stream, b"SUBMIT 1 0 1 1 1 1\nx\n", "SUBMIT") + + def test_zero_return_fails_closed_instead_of_spinning(self): + from openai_server import _write_all + stream = _ScriptedStream([0]) + with self.assertRaisesRegex(RuntimeError, "failed to write STOP to the engine"): + _write_all(stream, b"STOP 1\n", "STOP") + + +class WriteFailureHTTPTest(unittest.TestCase): + """A checked engine-stdin write that fails must reach the client as the + named 500 engine_error while the response is still uncommitted, and must + never splice anything into a stream that has already committed -- it + just ends (#597 item 3's `_fail`, exercised here by a genuine broken + pipe on the engine's stdin, through the real `Engine`, rather than by a + fake that raises `RuntimeError` directly -- a real `BrokenPipeError` is + a `ConnectionError`, and it is exactly that unwrapped subclass that a + correct checked write must keep away from do_POST's client-hangup + handler).""" + + def _serve(self, process): + with patch("openai_server.ARCH", "glm"), \ + patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + server = APIServer(("127.0.0.1", 0), engine, "test-model", None, 16, kv_slots=1) + # A short poll_interval means server.shutdown() (addCleanup, below) + # returns almost immediately instead of paying up to the default + # 0.5s poll -- these tests do no waiting of their own, so that 0.5s + # would be pure teardown overhead, not a wait under test. + thread = threading.Thread(target=server.serve_forever, args=(0.01,), daemon=True) + thread.start() + self.addCleanup(thread.join, 2) + self.addCleanup(server.server_close) + self.addCleanup(server.shutdown) + self.addCleanup(server.scheduler.close) + self.addCleanup(engine.close) + return server + + def _raw_post(self, server, path, body): + """POST over a plain socket and read until the peer closes, so the + literal status line(s) on the wire are visible -- `urlopen` strips + the status line into `response.status` before handing back `.read()`, + so asserting on `.read()` alone cannot tell "one status line" from + "a second one spliced into the body".""" + payload = json.dumps(body).encode() + request = (f"POST {path} HTTP/1.1\r\nHost: 127.0.0.1\r\n" + f"Content-Type: application/json\r\nContent-Length: {len(payload)}\r\n" + f"Connection: close\r\n\r\n").encode() + payload + sock = socket.create_connection(("127.0.0.1", server.server_port), 2) + sock.sendall(request) + sock.settimeout(2) + chunks = [] + try: + while True: + chunk = sock.recv(4096) + if not chunk: + break + chunks.append(chunk) + except socket.timeout: + pass + sock.close() + return b"".join(chunks) + + def test_dead_engine_submit_is_a_named_500_engine_error_not_silence(self): + def respond(process, frame): + if frame.split()[0] == b"SUBMIT": + raise BrokenPipeError("broken pipe") + + server = self._serve(FakeProcess(respond)) + body = json.dumps({"model": "test-model", "prompt": "hi"}).encode() + request = Request(f"http://127.0.0.1:{server.server_port}/v1/completions", + data=body, headers={"Content-Type": "application/json"}) + with self.assertRaises(HTTPError) as caught: + urlopen(request, timeout=2) + self.assertEqual(caught.exception.code, 500) + payload = json.load(caught.exception) + self.assertEqual(payload["error"]["code"], "engine_error") + + def test_dead_engine_submit_is_a_named_500_on_the_chat_streaming_path(self): + # Same failure, but through /v1/chat/completions with stream: true -- + # the tool-sideband/ThinkingStreamSplit wrapping chat streaming builds + # around engine.generate() must not swallow or reshape a RuntimeError + # that fires before anything is committed (nothing here ever reaches + # that wrapping: the SUBMIT write fails before the first ACCEPT/DATA). + def respond(process, frame): + if frame.split()[0] == b"SUBMIT": + raise BrokenPipeError("broken pipe") + + server = self._serve(FakeProcess(respond)) + body = json.dumps({"model": "test-model", "stream": True, + "messages": [{"role": "user", "content": "hi"}]}).encode() + request = Request(f"http://127.0.0.1:{server.server_port}/v1/chat/completions", + data=body, headers={"Content-Type": "application/json"}) + with self.assertRaises(HTTPError) as caught: + urlopen(request, timeout=2) + self.assertEqual(caught.exception.code, 500) + payload = json.load(caught.exception) + self.assertEqual(payload["error"]["code"], "engine_error") + + def test_uncommitted_stop_write_failure_is_a_named_500_engine_error(self): + # The dead-SUBMIT tests above cover the write _write_frame never gets + # a chance to make (the connection is refused before the first + # request even lands). This is the case _write_frame's own docstring + # is actually about: a CANCEL/STOP write that fails while the + # response is still uncommitted -- non-streaming, so nothing commits + # until the whole answer is assembled, and the STOP write happens + # well before that. + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 3\nhi \n") + process.stdout.feed(b"DATA " + request_id + b" 4\nSTOP\n") + elif fields[0] == b"STOP": + raise BrokenPipeError("broken pipe") + + server = self._serve(FakeProcess(respond)) + body = json.dumps({"model": "test-model", "prompt": "hi", + "stop": "STOP"}).encode() + request = Request(f"http://127.0.0.1:{server.server_port}/v1/completions", + data=body, headers={"Content-Type": "application/json"}) + with self.assertRaises(HTTPError) as caught: + urlopen(request, timeout=2) + self.assertEqual(caught.exception.code, 500) + payload = json.load(caught.exception) + self.assertEqual(payload["error"]["code"], "engine_error") + + def test_write_failure_reaching_the_committed_stream_ends_it_cleanly(self): + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + # "hi " clears StopFilter's hold and reaches the client; the + # stop sequence itself never does -- feeding them as two + # frames pins the STOP write to the SECOND one, after the + # stream has already committed on the first. + process.stdout.feed(b"DATA " + request_id + b" 3\nhi \n") + process.stdout.feed(b"DATA " + request_id + b" 4\nSTOP\n") + elif fields[0] == b"STOP": + raise BrokenPipeError("broken pipe") + + server = self._serve(FakeProcess(respond)) + raw = self._raw_post(server, "/v1/completions", + {"model": "test-model", "prompt": "hi", "stream": True, + "stop": "STOP"}) + # Exactly one status line on the whole wire -- the original 200. A + # response that spliced a second status line (or any framed error) + # into the already-committed SSE body would show up here as 2. + self.assertEqual(raw.count(b"HTTP/1.1"), 1) + self.assertIn(b"200", raw.split(b"\r\n", 1)[0]) + # The committed token reached the client... + self.assertIn(b'"hi "', raw) + # ...and nothing else did: no error object, no terminal [DONE] + # (generate() never returned normally). + self.assertNotIn(b"engine_error", raw) + self.assertNotIn(b"data: [DONE]", raw) + + +class PendingMapCleanupTest(unittest.TestCase): + """The pending-map entry is dropped on every failed engine write -- + SUBMIT, IMAGE, CANCEL or STOP -- including one forced by a checked + CANCEL/STOP write that fails closed with no `OSError` (a `None`/zero + return). A write that never reached the engine gets no DONE/ERROR back, + so nothing else would ever clear the slot; without this it sits behind + for `close()`/`_fail_pending` to find stale. + + This is narrower than "every exit": a raise from inside a decode + callback (`on_text`/`on_accept`/`on_tool`/`on_echo`), or the duplicate- + ACCEPT guard, still leaves the entry behind, as on `dev`; that is not + changed here.""" + + def test_a_failed_cancel_write_does_not_leave_the_pending_entry_behind(self): + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 1\nx\n") + elif fields[0] == b"CANCEL": + raise BrokenPipeError("broken pipe") + + process = FakeProcess(respond) + with patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + with self.assertRaisesRegex(RuntimeError, "failed to write CANCEL to the engine"): + engine.generate("hello", 8, 0.7, 0.9, lambda _: None, cancelled=lambda: True) + with engine.pending_lock: + self.assertEqual(engine.pending, {}) + engine.close() + + def test_a_failed_stop_write_does_not_leave_the_pending_entry_behind(self): + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 1\nx\n") + elif fields[0] == b"STOP": + raise BrokenPipeError("broken pipe") + + process = FakeProcess(respond) + with patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + output = [] + with self.assertRaisesRegex(RuntimeError, "failed to write STOP to the engine"): + engine.generate("hello", 8, 0.7, 0.9, output.append, + stopped=lambda: output == ["x"]) + with engine.pending_lock: + self.assertEqual(engine.pending, {}) + engine.close() + + def test_a_cancel_write_that_takes_no_bytes_still_drops_the_pending_entry(self): + # Not an OSError: the pipe accepts the call and answers 0, the same + # "took nothing" shape _write_all treats as a fail-closed write. If + # _write_frame's cleanup only ran for OSError, this entry would + # survive -- the pop has to be keyed on failure of the write, not on + # which failure shape it took. + class ZeroOnCancelProcess(FakeProcess): + def write(self, data): + text = bytes(data) + fields = text.split() + if fields[:1] == [b"CANCEL"]: + self.writes.append(text) + # Base ignores write()'s return value entirely, so a bare + # `return 0` here would leave base's generate() waiting + # forever for an ERROR/DONE it never actually asked for + # (the CANCEL it believes it sent never reached the + # engine) -- a hang, not a failure, on the base-overlay + # procedure. Feeding the terminal frame here means base + # fails in seconds instead; at head the RuntimeError from + # the zero-byte write fires first, so this frame arrives + # after the pending entry is already gone and is a no-op. + self.stdout.feed(b"ERROR " + fields[1] + b" CANCELLED\n") + return 0 + return super().write(data) + + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 1\nx\n") + + process = ZeroOnCancelProcess(respond) + with patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + with self.assertRaisesRegex(RuntimeError, "failed to write CANCEL to the engine"): + engine.generate("hello", 8, 0.7, 0.9, lambda _: None, cancelled=lambda: True) + with engine.pending_lock: + self.assertEqual(engine.pending, {}) + engine.close() + + def test_a_failed_image_write_names_the_image_frame_and_drops_the_pending_entry(self): + # generate()'s outer handler wraps both the IMAGE and the SUBMIT + # write with one except OSError -- an OSError on the IMAGE write + # specifically must still name "IMAGE", not fall through to the + # handler's own default "SUBMIT" label, or one frame kind would + # report under two different names depending on failure shape + # (_write_all's own None/zero path already says "IMAGE"; an OSError + # is the likelier real failure and must match). + class BrokenPipeOnImageProcess(FakeProcess): + def write(self, data): + if bytes(data).split()[:1] == [b"IMAGE"]: + raise BrokenPipeError("broken pipe") + return super().write(data) + + process = BrokenPipeOnImageProcess(lambda _process, _frame: None) + with patch("openai_server.ARCH", "glm53"), \ + patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm53", "model") + + class FakePatches: + def tobytes(self): + return bytes(range(8)) + + with self.assertRaisesRegex(RuntimeError, "failed to write IMAGE to the engine"): + engine.generate("hello", 8, 0.7, 0.9, lambda _: None, image=(FakePatches(), 2, 2)) + with engine.pending_lock: + self.assertEqual(engine.pending, {}) + engine.close() + + +class PlainRequestFrameOrderTest(unittest.TestCase): + """A request using none of the checked-write machinery's new surface + (no image, no grammar, no logprobs, no pin) must still put byte-identical + SUBMIT/STOP and SUBMIT/CANCEL frames on the wire, in the same order, as + the base module -- literals captured by running the base module's own + Engine.generate() against this same FakeProcess harness.""" + + def test_stop_flow_matches_base(self): + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 1\nx\n") + elif fields[0] == b"STOP": + process.stdout.feed(b"DONE " + request_id + b" STAT 1 1 0 1 2 0\n") + + process = FakeProcess(respond) + with patch("openai_server.ARCH", "glm"), \ + patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + output = [] + engine.generate("hello", 8, 0.7, 0.9, output.append, + stopped=lambda: output == ["x"]) + engine.close() + self.assertEqual(process.writes, [b"SUBMIT 1 0 5 8 0.7 0.9\nhello\n", b"STOP 1\n"]) + + def test_cancel_flow_matches_base(self): + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 1\nx\n") + elif fields[0] == b"CANCEL": + process.stdout.feed(b"ERROR " + request_id + b" CANCELLED\n") + + process = FakeProcess(respond) + with patch("openai_server.ARCH", "glm"), \ + patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm", "model") + disconnected = False + + def sink(text): + nonlocal disconnected + disconnected = True + + with self.assertRaises(ClientCancelled): + engine.generate("hello", 8, 0.7, 0.9, sink, cancelled=lambda: disconnected) + engine.close() + self.assertEqual(process.writes, [b"SUBMIT 1 0 5 8 0.7 0.9\nhello\n", b"CANCEL 1\n"]) + + def test_image_and_submit_frames_match_base_under_one_lock_acquisition(self): + request_id = None + + def respond(process, frame): + nonlocal request_id + fields = frame.split() + if fields[0] == b"SUBMIT": + request_id = fields[1] + process.stdout.feed(b"DATA " + request_id + b" 1\nx\n") + process.stdout.feed(b"DONE " + request_id + b" STAT 1 1 0 1 2 0\n") + + process = FakeProcess(respond) + with patch("openai_server.ARCH", "glm53"), \ + patch("openai_server.subprocess.Popen", return_value=process): + engine = Engine("glm53", "model") + counting_lock = _CountingLock() + engine.write_lock = counting_lock + + blob = bytes(range(12)) + + class FakePatches: + def tobytes(self): + return blob + + output = [] + engine.generate("hello", 8, 0.7, 0.9, output.append, image=(FakePatches(), 2, 2)) + engine.close() + self.assertEqual(process.writes, [ + b"IMAGE 1 12 2 2\n" + blob + b"\n", + b"SUBMIT 1 0 5 8 0.7 0.9\nhello\n", + ]) + # IMAGE must reach the wire before SUBMIT, and both under the SAME + # lock acquisition -- another request's IMAGE could otherwise land + # between this one's IMAGE and its SUBMIT. + self.assertEqual(counting_lock.acquisitions, 1) + + if __name__ == "__main__": unittest.main() diff --git a/c/tests/test_openai_tools_fallback_e2e.py b/c/tests/test_openai_tools_fallback_e2e.py new file mode 100644 index 000000000..8e2045aad --- /dev/null +++ b/c/tests/test_openai_tools_fallback_e2e.py @@ -0,0 +1,226 @@ +"""Prompt-injected tool translation for families with no native tool syntax. + +OLMoE and Qwen3.6 have no tool tokens in their chat templates, so the gateway +answers 400 for `tools[]` and for a `role: "tool"` turn rather than invent a +format. `COLI_TOOL_FALLBACK=1` opts into a translation that writes the +declaration, the prior assistant calls and the tool results as ordinary turns, +in the GLM wire format `parse_tool_calls()` already reads back (#1378). + +These tests pin both halves of that switch: the default still refuses, and with +the flag set a full two-turn agent loop completes through +`/v1/chat/completions`. Runs against a mock engine speaking the SERVE wire +protocol, so no checkpoint is needed. +""" +import json +import os +import socket +import subprocess +import sys +import tempfile +import unittest +import urllib.error +import urllib.request +from pathlib import Path + +HERE = Path(__file__).resolve().parent +SERVER = HERE.parent / "openai_server.py" +sys.path.insert(0, str(HERE.parent)) + +# The call the injected preamble asks for, in the format parse_tool_calls reads. +CALL = ("get_weather" + "locationRome" + "unitcelsius" + "") + +MOCK_ENGINE = r'''#!/usr/bin/env python3 +import sys, os +out, inp = sys.stdout.buffer, sys.stdin.buffer +out.write(b"\x01\x01READY\x01\x01\n" + b"STAT 0 0 0 0 0\n"); out.flush() + +CALL = ("get_weather" + "locationRome" + "unitcelsius" + "") + +def reply(rid, text): + data = text.encode("utf-8") + out.write(("DATA %s %d\n" % (rid, len(data))).encode() + data + b"\n"); out.flush() + out.write(("DONE %s STAT %d 1.0 50.0 10.0 42 0\n" % (rid, len(text.split()))).encode()) + out.flush() + +while True: + line = inp.readline() + if not line: break + f = line.decode().strip().split() + if not f or f[0] != "SUBMIT": continue + rid, plen = f[1], int(f[3]) + prompt = inp.read(plen).decode("utf-8", "replace"); inp.read(1) + with open(os.environ["MOCK_LOG"], "a") as log: + log.write(prompt + "\n\x00\n") + if "" in prompt: + reply(rid, "25 degrees and sunny in Rome.") + elif "weather in Rome" in prompt: + reply(rid, CALL) + else: + reply(rid, "Hello from the mock engine.") +''' + +TOOLS = [{"type": "function", "function": { + "name": "get_weather", + "description": "Current weather for a city", + "parameters": {"type": "object", + "properties": {"location": {"type": "string"}, + "unit": {"type": "string"}}, + "required": ["location"]}}}] + + +@unittest.skipUnless(os.name == "posix", + "the mock engine is a shebang script the gateway execs directly; " + "Windows CreateProcess cannot run it. The gateway logic under test " + "is platform-independent and covered by the POSIX CI jobs.") +class _FallbackBase(unittest.TestCase): + """Boots the gateway for one arch, with COLI_TOOL_FALLBACK under test.""" + + arch = None + # config.json model_type for the family (family_registry.model_types), which + # is not always the --arch id: qwen36's checkpoints say "qwen3_5_moe". + model_type = None + fallback = "1" + + @classmethod + def setUpClass(cls): + if cls.arch is None: + raise unittest.SkipTest("base class") + cls.tmp = tempfile.TemporaryDirectory() + (Path(cls.tmp.name) / "config.json").write_text( + json.dumps({"model_type": cls.model_type or cls.arch}), encoding="utf-8") + mock = Path(cls.tmp.name) / "mock_engine.py" + mock.write_text(MOCK_ENGINE) + mock.chmod(0o755) + cls.mock_log = Path(cls.tmp.name) / "prompts.log" + cls.mock_log.touch() + with socket.socket() as probe: + probe.bind(("127.0.0.1", 0)) + cls.port = probe.getsockname()[1] + env = dict(os.environ, MOCK_LOG=str(cls.mock_log), + COLI_TOOL_FALLBACK=cls.fallback) + env.pop("COLI_API_KEY", None) + cls.server = subprocess.Popen( + [sys.executable, str(SERVER), "--model", cls.tmp.name, + "--engine", str(mock), "--arch", cls.arch, "--port", str(cls.port)], + env=env, stderr=subprocess.DEVNULL) + cls.base = f"http://127.0.0.1:{cls.port}/v1" + for _ in range(100): + try: + with urllib.request.urlopen(cls.base + "/models", timeout=2): + pass + return + except OSError: + if cls.server.poll() is not None: + raise RuntimeError("gateway exited during startup") + import time + time.sleep(0.1) + raise RuntimeError("gateway did not come up") + + @classmethod + def tearDownClass(cls): + if getattr(cls, "server", None) is None: + return + cls.server.terminate() + cls.server.wait(timeout=5) + cls.tmp.cleanup() + + def post(self, body): + req = urllib.request.Request( + self.base + "/chat/completions", json.dumps(body).encode(), + {"Content-Type": "application/json"}) + with urllib.request.urlopen(req, timeout=30) as resp: + return json.loads(resp.read()) + + def model_id(self): + with urllib.request.urlopen(self.base + "/models", timeout=5) as resp: + return json.loads(resp.read())["data"][0]["id"] + + +class _TwoTurnLoop(_FallbackBase): + """The two-turn loop the issue asks for, with the flag on.""" + + def test_tool_call_parsed_back(self): + out = self.post({"model": self.model_id(), + "messages": [{"role": "user", "content": "weather in Rome?"}], + "tools": TOOLS, "temperature": 0, "max_tokens": 128}) + choice = out["choices"][0] + self.assertEqual(choice["finish_reason"], "tool_calls") + calls = choice["message"].get("tool_calls") or [] + self.assertEqual(len(calls), 1) + self.assertEqual(calls[0]["function"]["name"], "get_weather") + self.assertEqual(json.loads(calls[0]["function"]["arguments"]), + {"location": "Rome", "unit": "celsius"}) + + def test_second_turn_completes(self): + mid = self.model_id() + msgs = [{"role": "user", "content": "weather in Rome?"}] + out = self.post({"model": mid, "messages": msgs, "tools": TOOLS, + "temperature": 0, "max_tokens": 128}) + calls = out["choices"][0]["message"]["tool_calls"] + msgs.append({"role": "assistant", "content": None, "tool_calls": calls}) + msgs.append({"role": "tool", "tool_call_id": calls[0]["id"], + "content": json.dumps({"temp_c": 25})}) + out2 = self.post({"model": mid, "messages": msgs, "tools": TOOLS, + "temperature": 0, "max_tokens": 128}) + self.assertIn("25", out2["choices"][0]["message"]["content"] or "") + + prompts = self.mock_log.read_text() + self.assertIn("", prompts) # declaration was injected + self.assertIn("get_weather", prompts) # prior call replayed + self.assertIn("", prompts) # result reached the prompt + + +class OlmoeToolFallbackE2E(_TwoTurnLoop): + arch = "olmoe" + + +class Qwen36ToolFallbackE2E(_TwoTurnLoop): + arch = "qwen36" + model_type = "qwen3_5_moe" + + +class _RefusedByDefault(_FallbackBase): + """With the flag unset the gateway must still refuse, not half-support.""" + + fallback = "0" + + def _expect_400(self, body): + with self.assertRaises(urllib.error.HTTPError) as caught: + self.post(body) + self.assertEqual(caught.exception.code, 400) + return json.loads(caught.exception.read()) + + def test_tools_refused(self): + err = self._expect_400({"model": self.model_id(), + "messages": [{"role": "user", "content": "hi"}], + "tools": TOOLS, "max_tokens": 16}) + self.assertIn("COLI_TOOL_FALLBACK", json.dumps(err)) + + def test_tool_role_refused(self): + self._expect_400({"model": self.model_id(), + "messages": [{"role": "user", "content": "hi"}, + {"role": "tool", "content": "18C"}], + "max_tokens": 16}) + + +class OlmoeToolRefusedByDefault(_RefusedByDefault): + arch = "olmoe" + + +class Qwen36ToolRefusedByDefault(_RefusedByDefault): + arch = "qwen36" + model_type = "qwen3_5_moe" + + +# The shared base classes must not run as tests themselves. +del _FallbackBase, _TwoTurnLoop, _RefusedByDefault + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_openai_tools_v41_e2e.py b/c/tests/test_openai_tools_v41_e2e.py index 59cd3a379..56d4b0dc4 100644 --- a/c/tests/test_openai_tools_v41_e2e.py +++ b/c/tests/test_openai_tools_v41_e2e.py @@ -95,9 +95,11 @@ def setUpClass(cls): cls.port = probe.getsockname()[1] env = dict(os.environ, MOCK_LOG=str(cls.mock_log)) env.pop("COLI_API_KEY", None) + # This protocol mock has no weights to plan; choose its cap explicitly. cls.server = subprocess.Popen( [sys.executable, str(SERVER), "--model", cls.tmp.name, - "--engine", str(mock), "--arch", "deepseek_v41", "--port", str(cls.port)], + "--engine", str(mock), "--arch", "deepseek_v41", "--port", str(cls.port), + "--cap", "8"], env=env, stderr=subprocess.DEVNULL) cls.base = f"http://127.0.0.1:{cls.port}/v1" for _ in range(100): diff --git a/c/tests/test_oracle.c b/c/tests/test_oracle.c new file mode 100644 index 000000000..57b23cbf1 --- /dev/null +++ b/c/tests/test_oracle.c @@ -0,0 +1,83 @@ +#include +#include +#include "../oracle.h" + +#define CHECK(c) do { if (!(c)) { \ + fprintf(stderr,"%s:%d: %s\n",__FILE__,__LINE__,#c); return 1; \ +} } while (0) + +int main(void) { + OracleRef r; + const char *good="{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3]}"; + CHECK(oracle_ref_parse(good,4,1,&r)); + CHECK(r.np==1 && r.nfull==2 && r.prompt[0]==1 && r.full[1]==2 && r.tf[1]==3); + oracle_ref_free(&r); + CHECK(oracle_ref_parse("{\"prompt_ids\":[1],\"full_ids\":[1,2]}",4,0,&r)); + CHECK(r.tf==NULL); + oracle_ref_free(&r); + + /* Missing/truncated/wrongly typed predictions used to reach tf[i] unchecked. */ + const char *bad[]={ + "{}", + "[]", + "{\"prompt_ids\":[1],\"full_ids\":[1,2]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3,1]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":{\"x\":2,\"y\":3}}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3]", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3,]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3]} garbage", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2 3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\" [2,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2,3],\"tf_pred\":[1,1]}", + "{\"prompt_ids\":[1,2],\"full_ids\":[1],\"tf_pred\":[2]}", + "{\"prompt_ids\":[2],\"full_ids\":[1,2],\"tf_pred\":[2,3]}", + "{\"prompt_ids\":[],\"full_ids\":[1,2],\"tf_pred\":[2,3]}", + "{\"prompt_ids\":[-1],\"full_ids\":[-1,2],\"tf_pred\":[2,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,4],\"tf_pred\":[2,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[true,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[\"2\",3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2.5,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[1e999,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[NaN,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[+2,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[02,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[2.,3]}", + "{\"prompt_ids\":[1],\"full_ids\":[1,2],\"tf_pred\":[0x2,3]}", + }; + for (size_t i=0; i +#include +#include +#include +#include +#if defined(__AVX2__) && defined(__FMA__) +#include +#endif + +#include "../qgemv.h" + +static int fails = 0; +#define CHECK(cond, msg) do { \ + if (cond) { printf(" ok %s\n", msg); } \ + else { printf(" FAIL %s\n", msg); fails++; } \ +} while (0) + +/* Deterministic across libc implementations, unlike rand(). */ +static uint32_t rng_state = 0x9E3779B9u; +static uint32_t rng_next(void) { rng_state = rng_state * 1664525u + 1013904223u; return rng_state; } +static float rng_f32(void) { return (float)((int32_t)(rng_next() >> 8) - 8388608) / 8388608.0f; } + +/* --------------------------------------------------------------------------- + * VERBATIM copy of matmul_q as it shipped before the SSE4.1 tier was added. + * Do not tidy, reformat or "improve" this: its value is that it is the old + * arithmetic, character for character. If this drifts, the test proves nothing. + * ------------------------------------------------------------------------- */ +static void reference_q_gemv(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) { +#if defined(__AVX2__) && defined(__FMA__) + #pragma omp parallel for schedule(static) if(O >= 256) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps(); + __m256 a2 = _mm256_setzero_ps(), a3 = _mm256_setzero_ps(); + int i = 0; + for (; i + 32 <= I; i += 32) { + __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i)); + __m128i b1 = _mm_loadu_si128((const __m128i*)(w + i + 16)); + a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0); + a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1); + a2 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+16), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a2); + a3 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+24), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a3); + } + a0 = _mm256_add_ps(_mm256_add_ps(a0,a1), _mm256_add_ps(a2,a3)); + __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1)); + s = _mm_add_ps(s, _mm_movehl_ps(s,s)); + s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1)); + float acc = _mm_cvtss_f32(s); + for (; i < I; i++) acc += x[i] * (float)w[i]; + y[o] = acc * scale[o]; + } +#else + #pragma omp parallel for schedule(static) + for (int o = 0; o < O; o++) { + const int8_t *w = q + (int64_t)o * I; + float acc = 0.f; + for (int i = 0; i < I; i++) acc += x[i] * (float)w[i]; + y[o] = acc * scale[o]; + } +#endif +} + +/* Fills one random case, runs both kernels, compares the results. + * The AVX2 and scalar tiers stay exact: memcmp on raw float bits, per the + * engine's byte-identical requirement. The SSE4.1 tier's tree reduction is + * not bit-reproducible against the scalar reference -- float addition is + * not associative -- so that tier alone is checked by max relative error. */ +static int rows_match(int I, int O) { + float *x = malloc((size_t)I * sizeof *x); + int8_t *q = malloc((size_t)O * I); + float *sc = malloc((size_t)O * sizeof *sc); + float *y = malloc((size_t)O * sizeof *y); + float *ref = malloc((size_t)O * sizeof *ref); + if (!x || !q || !sc || !y || !ref) { fprintf(stderr, "out of memory\n"); exit(2); } + + for (int i = 0; i < I; i++) x[i] = rng_f32(); + for (int64_t i = 0; i < (int64_t)O * I; i++) q[i] = (int8_t)(rng_next() >> 24); + for (int o = 0; o < O; o++) sc[o] = rng_f32() * 0.01f; + + /* Poison both outputs differently: a kernel that skips a row must not pass + * by leaving stale bytes that happen to agree. */ + memset(y, 0x5A, (size_t)O * sizeof *y); + memset(ref, 0xA5, (size_t)O * sizeof *ref); + + reference_q_gemv(ref, x, q, sc, I, O); + matmul_q(y, x, q, sc, I, O); + +#if defined(__SSE4_1__) && !(defined(__AVX2__) && defined(__FMA__)) + /* Combined absolute+relative tolerance (numpy allclose-style), not pure + * relative error: a tree-reduced SSE4.1 sum and a sequential scalar sum + * disagree by float-rounding noise, and when a reference output happens + * to land near zero that noise alone blows the relative error past any + * pure-relative threshold. Measured worst-case noise across all shapes + * below is ~4.2e-6 absolute; the 1e-5 floor gives headroom above that + * without loosening the check for the far more common non-near-zero + * outputs, where 1e-4 relative still rules. */ + int equal = 1; + for (int o = 0; o < O; o++) { + float diff = fabsf(y[o] - ref[o]); + float denom = fabsf(ref[o]); + if (diff > 1e-5f + 1e-4f * denom) { equal = 0; break; } + } +#else + int equal = memcmp(y, ref, (size_t)O * sizeof *y) == 0; +#endif + free(x); free(q); free(sc); free(y); free(ref); + return equal; +} + +int main(void) { + printf("test_qgemv: plain int8 GEMV matches its reference\n"); + + /* ---- P1: the shapes the engine actually runs ------------------------- */ + CHECK(rows_match(2048, 512), "P1 lm_head-like shape (I=2048, O=512) matches"); + CHECK(rows_match(512, 2048), "P1 dense shape (I=512, O=2048) matches"); + + /* ---- P2: row count around the OpenMP parallel-for threshold ---------- */ + CHECK(rows_match(2048, 515), "P2 O=515 (just over the parallel threshold) matches"); + CHECK(rows_match(2048, 6), "P2 O=6 (few rows) matches"); + CHECK(rows_match(2048, 1), "P2 O=1 (single row) matches"); + + /* ---- P4: I not a multiple of the unroll width leaves a final chunk --- */ + CHECK(rows_match(2000, 512), "P4 I=2000 (final chunk of 16) matches"); + + /* ---- P5: a short final chunk, folded in by the unconditional tail ---- */ + CHECK(rows_match(1990, 512), "P5 I=1990 (final chunk of 6) matches"); + CHECK(rows_match(1, 4), "P5 I=1 (no full unrolled chunk at all) matches"); + + printf("\n%s (%d failure%s)\n", fails ? "TEST FAIL" : "ALL PASS", fails, fails == 1 ? "" : "s"); + return fails ? 1 : 0; +} diff --git a/c/tests/test_qt_addrow.c b/c/tests/test_qt_addrow.c index 46c3673ca..6cac5986c 100644 --- a/c/tests/test_qt_addrow.c +++ b/c/tests/test_qt_addrow.c @@ -4,31 +4,35 @@ * fmt 0/4/5 explicitly, then fall through assuming a PER-ROW scale (t->s[row]) followed by * fmt=1/2/3 (qt_addrow) or fmt=0/1/2/3/4/5 (qt_matvec_rows, via an if/else-if chain ending * in a bare `else`) -- nothing stopped fmt=6 (E8/IQ3, t->s is a FIXED 4-byte tag, not O - * floats) or fmt=8 (fp8-e4m3-b128, t->s holds per-128x128-block floats, not O; t->q4 is - * NULL) from reaching that fall-through. For fmt=8 specifically this SIGSEGVs: t->s[row] - * overreads (silently, usually not fatal on its own), then the untouched tail computes - * `t->q4+(int64_t)row*((I+3)/4)` on a NULL t->q4 and dereferences it. For fmt=6 it silently - * misreads the real E8/IQ3 lattice bytes as int2-packed data (same bug SHAPE as #298's CUDA - * absorb-kernel fix, which is why this file's own fmt=4/5 branches exist -- fmt=6 was simply - * missed). Both functions now refuse loudly (exit(1), naming the function and the fmt) for - * any fmt they don't explicitly handle, matching qt_resolve_fmt's own "refuse rather than - * misread" discipline. + * floats) from reaching that fall-through, where it silently misreads the real E8/IQ3 + * lattice bytes as int2-packed data (same bug SHAPE as #298's CUDA absorb-kernel fix, which + * is why this file's own fmt=4/5 branches exist -- fmt=6 was simply missed). fmt=6 has no + * decoder in either function and still refuses loudly (exit(1), naming the function and the + * fmt), matching qt_resolve_fmt's own "refuse rather than misread" discipline. * - * This file: (1) proves the refusal fires for fmt=6 and fmt=8 through BOTH functions - * (fork+pipe+waitpid, this suite's established house pattern for exit(1)-terminated paths -- - * see tests/test_fp8_load.c's expect_refuse/expect_stamp_refuse); (2) proves every format - * BOTH functions still legitimately handle (0/1/2/3/4/5) produces byte-identical results - * against an independently-written reference dequantizer (qt_dequant_row_ref below -- NOT - * copy-pasted from qt_addrow/qt_matvec_rows, restructured as a single per-element loop per - * format, so a real regression in either the guard's placement or the untouched per-fmt math - * would show up here, not just tautologically re-run the same code). Reachability note (not - * a scope excuse, just context): both functions serve ONLY the kv_b absorb path - * (attention_rows/decode call sites), and tools/repack_fp8_passthrough.py deliberately - * excludes kv_b_proj from fmt=8 repacking -- so this fires only via a hand-slotted or - * ambiguous-collision container, not this repo's own tooling's own output. Crash-instead-of- - * refuse is still a real defect (spec I6: loud failure, every refusal names its condition), - * and the fmt=8 QT surface these functions can now be handed is one this same PR pair - * created. */ + * fmt=8 (fp8-e4m3-b128) USED to hit this same fall-through and SIGSEGV (t->s[row] overread, + * then a NULL t->q4 dereference -- see git history for the pre-fix account) but now has its + * own explicit branch in both functions (the absorb-decode commit this file accompanies): + * t->s holds per-128x128-block scales, t->q8 holds raw e4m3 bytes (t->q4 stays NULL), decoded + * through the same e4m3 LUT / block-scale geometry (quant.h's e4m3_decode/fp8_nblk/FP8_BLOCK) + * that matmul_fp8 already exercises on the dense/expert path. This file now (1) proves fmt=8 + * produces correct, tolerance-bounded output through BOTH functions -- an exact-dequant check + * against an independent per-element reference (qt_dequant_row_ref's new fmt==8 branch, + * disjoint code shape from qt_addrow/qt_matvec_rows' block-batched loops) plus a direct parity + * check against matmul_fp8 (the proven non-absorb fp8 reference) on a kv_b-shaped tensor; (2) + * proves the refusal still fires for fmt=6 through both functions (fork+pipe+waitpid, this + * suite's established house pattern for exit(1)-terminated paths -- see + * tests/test_fp8_load.c's expect_refuse/expect_stamp_refuse); (3) proves every other format + * both functions handle (0/1/2/3/4/5) produces byte-identical results against the same + * independent reference dequantizer, so a real regression in either the guard's placement or + * the untouched per-fmt math would show up here, not just tautologically re-run the same code. + * Reachability note (not a scope excuse, just context): both functions serve ONLY the kv_b + * absorb path (attention_rows/decode call sites), and the repo's own repack tool mints + * exactly the container these arms decode -- tools/repack_fp8_passthrough.py emits kv_b_proj + * (kind "kvb", in its RESIDENT_KINDS/STAMPABLE_KINDS sets) as byte-preserved fmt=8 with a + * stamped scale sidecar, and attention_rows' `int absorb = kvs || ...` makes absorb + * unbypassable on the batched serving path. These arms are the decode support that + * container needs. */ #define main coli_glm_main_unused #include "../colibri.c" #undef main @@ -48,13 +52,22 @@ static uint64_t rng = 0xA11CE5EEDF00Dull; static uint8_t rndbyte(void){ rng ^= rng << 13; rng ^= rng >> 7; rng ^= rng << 17; return (uint8_t)(rng & 0xFF); } static float rndsmallf(void){ rng ^= rng << 13; rng ^= rng >> 7; rng ^= rng << 17; return ((int64_t)(rng & 0xFFF) - 0x800) / (float)0x800; } /* [-1,1)-ish, small magnitude */ +/* fmt=8 fixtures must avoid the two NaN byte patterns (0x7F/0xFF, see quant.h's E4M3_LUT + * comment) -- a NaN weight is a real, policy-accepted outcome in production, but it would + * make this file's relative-error comparisons meaningless (NaN != NaN), same reasoning + * tests/test_fp8_load.c's rndbyte_nonan documents. */ +static uint8_t rndbyte_nonan(void){ + for(;;){ uint8_t b=rndbyte(); if(b!=0x7F && b!=0xFF) return b; } +} /* ---- independent reference dequantizer: W[row,:] as floats, one element per loop * iteration -- deliberately NOT the same code shape as qt_addrow/qt_matvec_rows (those * unroll pairs for fmt=2/3, split low/high planes for fmt=5) so this genuinely * cross-checks the production math, not just the production code running twice. ---- */ +static void qt_dequant_row_ref_fmt8(const QT *t, int row, float *out); static void qt_dequant_row_ref(const QT *t, int row, float *out){ int I=t->I; + if(t->fmt==8){ qt_dequant_row_ref_fmt8(t,row,out); return; } if(t->fmt==0){ const float *w=t->qf+(int64_t)row*I; for(int i=0;ifmt==4){ const uint8_t *w=t->q4+(int64_t)row*((I+1)/2); int gs=t->gs, ng=(I+gs-1)/gs; @@ -80,6 +93,22 @@ static void qt_dequant_row_ref(const QT *t, int row, float *out){ { const uint8_t *w=t->q4+(int64_t)row*((I+3)/4); for(int i=0;i>2]; int v=(b>>((i&3)*2))&3; out[i]=((int)v-2)*s; } } } +/* fmt=8 (fp8-e4m3-b128) independent reference: a flat per-ELEMENT loop, deliberately NOT + * the block-batched-accumulation shape qt_addrow/qt_matvec_rows and matmul_fp8 all share -- + * this recomputes each element's block index and looks up its scale independently, on every + * iteration, rather than walking block-by-block. Reuses quant.h's e4m3_decode/fp8_nblk/ + * FP8_BLOCK constants deliberately: those ARE the declared format geometry under test here + * (the same LUT/geometry every fmt=8 consumer in the tree shares), not implementation + * detail this reference should reinvent. */ +static void qt_dequant_row_ref_fmt8(const QT *t, int row, float *out){ + int I=t->I; + int64_t nblkI=fp8_nblk(I), blkO=(int64_t)row/FP8_BLOCK; + const float *scl=t->s+blkO*nblkI; + for(int i=0;iq8[(int64_t)row*I+i])*scl[bi]; + } +} /* ---- fixture builders: one per format, deterministic pseudo-random payload ---- */ static void fill_fmt0(QT *t, int O, int I){ @@ -126,6 +155,17 @@ static void fill_fmt5(QT *t, int O, int I){ for(int i=0;iq4[i]=rndbyte(); for(int i=0;is[i]=0.025f+0.0003f*(float)i; } +/* kv_b-shaped fmt=8 fixture: I is left caller-chosen so callers can pick both the + * real contraction dimension (kv_lora_rank=512, GLM-5.2's MLA latent width) and + * small tail-covering shapes. */ +static void fill_fmt8(QT *t, int O, int I){ + t->fmt=8; t->O=O; t->I=I; t->gs=0; + int64_t nblkO=fp8_nblk(O), nblkI=fp8_nblk(I), nblk=nblkO*nblkI; + t->q8=(int8_t*)malloc((size_t)O*I); + t->s=(float*)malloc((size_t)nblk*sizeof(float)); + for(int64_t i=0;i<(int64_t)O*I;i++) t->q8[i]=(int8_t)rndbyte_nonan(); + for(int64_t i=0;is[i]=0.01f+0.001f*(float)i; +} static void free_qt(QT *t){ free(t->qf); free(t->q8); free(t->q4); free(t->s); memset(t,0,sizeof *t); } /* ---- byte-identity: qt_addrow / qt_matvec_rows vs the independent reference, every @@ -185,22 +225,113 @@ static void test_byte_identity_all_formats(void){ memset(&t,0,sizeof t); fill_fmt3(&t,4,17); check_addrow_identity(&t,"fmt=3"); check_matvec_identity(&t,"fmt=3"); free_qt(&t); memset(&t,0,sizeof t); fill_fmt4(&t,4,40,16);check_addrow_identity(&t,"fmt=4"); check_matvec_identity(&t,"fmt=4"); free_qt(&t); memset(&t,0,sizeof t); fill_fmt5(&t,4,130); check_addrow_identity(&t,"fmt=5"); check_matvec_identity(&t,"fmt=5"); free_qt(&t); + /* fmt=8: small single-block-both-dims shape (mirrors the other formats' O=4-ish + * defaults) plus a kv_b-shaped multi-block shape (O=192 -> nblkO=2 with a partial + * tail block; I=512=kv_lora_rank -> nblkI=4, exact blocks) so both the partial-row- + * block and the multi-column-block paths through the new branches are exercised. */ + memset(&t,0,sizeof t); fill_fmt8(&t,6,17); check_addrow_identity(&t,"fmt=8 (single block)"); check_matvec_identity(&t,"fmt=8 (single block)"); free_qt(&t); + memset(&t,0,sizeof t); fill_fmt8(&t,192,512); check_addrow_identity(&t,"fmt=8 (kv_b-shaped, multi-block)"); check_matvec_identity(&t,"fmt=8 (kv_b-shaped, multi-block)"); free_qt(&t); + /* nblkI>=2 WITH a partial COLUMN tail: I=200 -> nblkI=2 with a 72-wide tail block, + * O=130 -> nblkO=2 with a 2-row tail. The kv_b-shaped case above has I=512 (exact + * column blocks), so the per-row scale STRIDE (nblkI) and the bi block-scale index + * were only ever exercised at shapes where flooring/misdeviating them is invisible. + * Mutation-checked: corrupting the block-scale index math (nblkI=I/FP8_BLOCK, the + * floor) passes every fmt=8 case above and fails exactly this one. */ + memset(&t,0,sizeof t); fill_fmt8(&t,130,200); check_addrow_identity(&t,"fmt=8 (partial column tail)"); check_matvec_identity(&t,"fmt=8 (partial column tail)"); free_qt(&t); +} + +/* ---- fmt=8 absorb-path vs the proven non-absorb matmul_fp8 reference (quant.h) ---- + * qt_matvec_rows(t,row,1,x,&y) and matmul_fp8's per-output-row computation are the SAME + * mathematical quantity (y[row] = sum_i x[i]*dequant(W[row,i])) computed with the exact + * same block-scale geometry and the exact same accumulation order (float-accumulate within + * a 128-wide block, double-accumulate the per-block partials) -- both were written to + * mirror that convention deliberately (see this PR's qt_matvec_rows fmt=8 branch comment + * in colibri.c). Compiled in the same translation unit with the same flags, so the two are + * expected to agree tightly; the tolerance below is a guard against incidental FP-contraction + * differences between the two call sites, not evidence of a real algorithmic mismatch. */ +static void test_fmt8_matmul_fp8_parity(void){ + enum { O=192, I=512 }; /* kv_b-shaped: I=kv_lora_rank=512, O spans a partial row-block */ + QT t; memset(&t,0,sizeof t); fill_fmt8(&t,O,I); + float *x=(float*)malloc((size_t)I*sizeof(float)); + for(int i=0;i1e-6f ? ae/fabsf(want) : ae; + if(rel > 1e-5f){ + printf("FAIL fmt=8 matmul_fp8 parity: row=%d got=%.9g want=%.9g rel=%.3g\n", + row,(double)y,(double)want,(double)rel); + fails++; + } + } + free(x); free(yref); free_qt(&t); +} + +/* ---- fmt=8 NaN propagation (policy pin -- see quant.h's "NaN POLICY" note): a NaN + * weight byte (0x7F/0xFF) decodes to a real IEEE NaN and is left to PROPAGATE, relying + * on the tested downstream sampler net (tests/test_logit_nan.c), never scrubbed at the + * weight level. The identity checks above deliberately exclude NaN bytes + * (rndbyte_nonan) -- and would pass NaN lanes silently anyway (NaN > eps compares + * false) -- so this pins the absorb-path behavior explicitly against the independent + * reference: qt_addrow poisons exactly the accumulator lanes whose reference dequant is + * NaN (per-element accumulate, both NaN byte codes), qt_matvec_rows poisons the whole + * dot product of any row containing one, and clean rows/lanes of the same tensor stay + * NaN-free and tolerance-identical to the reference. ---- */ +static void test_fmt8_nan_propagation(void){ + enum { O=130, I=200 }; /* same tail-covering shape as the identity case above */ + QT t; memset(&t,0,sizeof t); fill_fmt8(&t,O,I); + /* one NaN of EACH byte code, in different row-blocks and different column blocks; + * (129,199) lands in the tail row-block x tail column-block corner */ + const int nan_row[2]={3,129}, nan_col[2]={5,199}; + t.q8[(int64_t)nan_row[0]*I+nan_col[0]]=(int8_t)0x7F; + t.q8[(int64_t)nan_row[1]*I+nan_col[1]]=(int8_t)0xFF; + float *ref=(float*)malloc((size_t)I*sizeof(float)); + float *acc=(float*)malloc((size_t)I*sizeof(float)); + float *x=(float*)malloc((size_t)I*sizeof(float)); + for(int i=0;i1e-6f?ae/fabsf(want):ae; + if(rel>1e-5f){ printf("FAIL fmt=8 NaN: qt_addrow row=%d i=%d clean lane got=%.9g want=%.9g\n",row,i,(double)acc[i],(double)want); fails++; } + } + } + float y=0.f; qt_matvec_rows(&t,row,1,x,&y); + CHECK(isnan(y)); /* dot product over a NaN-bearing row is NaN, like the reference sum */ + } + /* a clean row of the SAME tensor stays NaN-free through both functions */ + { int row=4; float y=0.f; + qt_dequant_row_ref(&t,row,ref); + for(int i=0;is[0]=v; } + +static void call_addrow_fmt8_nan(void){ + QT t; memset(&t,0,sizeof t); fill_fmt8(&t,6,17); poison_scale(&t,(float)NAN); + float acc[17]; memset(acc,0,sizeof acc); + qt_addrow(&t,0,1.f,acc); +} +static void call_matvec_fmt8_nan(void){ + QT t; memset(&t,0,sizeof t); fill_fmt8(&t,6,17); poison_scale(&t,(float)NAN); + static float x[17]; for(int i=0;i<17;i++) x[i]=rndsmallf(); + float y=0.f; qt_matvec_rows(&t,0,1,x,&y); +} +/* A ZERO block scale is valid data, not corruption: block scales are amax/448, + * so an all-zero block legitimately carries one. Both arms must decode it to + * zeros in place -- not refuse it, and not leave the accumulator untouched. + * These two assertions exist to keep a zero refusal from being re-added. */ +static void test_fmt8_zero_scale_decodes_to_zeros(void){ + enum { O=6, I=17 }; + { QT t; memset(&t,0,sizeof t); fill_fmt8(&t,O,I); poison_scale(&t,0.f); + float acc[I]; for(int i=0;i +#include + +static int fails; +static void check(int ok, const char *what) { + if (!ok) { fails++; printf(" FAIL: %s\n", what); } +} + +/* residency table for the callback: level per expert id */ +static int8_t g_lvl[64]; +static int lvl_table(void *ctx, int e) { (void)ctx; return g_lvl[e]; } + +/* softmax row where expert e has mass proportional to (E - e): rank == id. */ +static void ranked_row(float *pr, int E) { + float sum = 0; for (int e = 0; e < E; e++) { pr[e] = (float)(E - e); sum += pr[e]; } + for (int e = 0; e < E; e++) pr[e] /= sum; +} + +static int has(const int *idx, int K, int e) { for (int k = 0; k < K; k++) if (idx[k] == e) return 1; return 0; } + +int main(void) { + enum { E = 16, K = 4 }; + float pr[E]; uint8_t keep[E]; int idx[K]; float val[K]; RouteStats st; + ranked_row(pr, E); memset(keep, 1, sizeof keep); + + /* nothing resident: the plain top-K, no substitution, full agreement, zero KL */ + memset(g_lvl, 0, sizeof g_lvl); memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[0]==0 && idx[1]==1 && idx[2]==2 && idx[3]==3, "no_resident_expert_means_the_true_top_k"); + check(st.swaps == 0 && st.agree_hit == K && st.agree_tot == K, "no_substitution_reports_full_agreement"); + check(st.kl_n == 1 && st.kl_sum == 0.0, "identical_choice_has_zero_kl"); + check(val[0] == pr[0] && val[3] == pr[3], "val_carries_the_raw_mass_of_each_chosen_expert"); + + /* a RAM-resident expert inside the window replaces the unresident tail */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[7] = 1; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[0]==0 && idx[1]==1, "true_top_j_is_taken_first"); + check(idx[2]==7 && idx[3]==2, "resident_expert_in_window_fills_before_the_unresident_ranking"); + check(st.swaps == 1 && st.swaps_vram == 0 && st.agree_hit == 3, "one_substitution_is_counted_once_and_not_as_vram"); + check(st.kl_sum > 0.0, "a_substitution_shows_up_as_positive_kl"); + + /* the sacred top-J is taken even when a lower rank is resident and it is not */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[2] = 1; g_lvl[3] = 1; g_lvl[4] = 1; g_lvl[5] = 1; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[0]==0 && idx[1]==1 && idx[2]==2 && idx[3]==3, "unresident_top_j_is_never_displaced"); + check(st.swaps == 0, "resident_experts_already_in_the_top_k_are_not_substitutions"); + + /* VRAM outranks RAM inside the window, whatever their rank order */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[5] = 1; g_lvl[9] = 2; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[2]==9 && idx[3]==5, "vram_resident_fills_before_ram_resident"); + check(st.swaps == 2 && st.swaps_vram == 1, "swaps_to_vram_are_counted_separately"); + + /* residency outside the top-M window is ignored */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[12] = 2; g_lvl[15] = 1; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(!has(idx, K, 12) && !has(idx, K, 15) && idx[2]==2 && idx[3]==3, "resident_expert_past_the_window_is_not_chosen"); + + /* ROUTE_J=0: even rank 0 may be replaced; ROUTE_J=K: nothing may */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[6] = 1; g_lvl[7] = 1; g_lvl[8] = 1; g_lvl[9] = 1; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 0, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[0]==6 && idx[1]==7 && idx[2]==8 && idx[3]==9 && st.swaps == 4, "route_j_zero_lets_residents_take_every_slot"); + memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, K, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[0]==0 && idx[3]==3 && st.swaps == 0, "route_j_equal_to_k_disables_substitution"); + + /* the group mask is respected: an excluded expert is never ranked or chosen */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[5] = 2; memset(&st, 0, sizeof st); + keep[1] = 0; keep[5] = 0; + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(!has(idx, K, 1) && !has(idx, K, 5), "masked_experts_are_never_chosen_even_when_resident"); + check(idx[0]==0 && idx[1]==2 && idx[2]==3 && idx[3]==4, "masked_ranking_shifts_the_true_top_k"); + memset(keep, 1, sizeof keep); + + /* ROUTE_ALPHA scales only the substitute's mass, before moe() renormalises */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[7] = 1; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.f, 0.5f, lvl_table, NULL, idx, val, &st); + check(idx[2]==7 && val[2] == pr[7]*0.5f && val[0] == pr[0] && val[3] == pr[2], "alpha_halves_the_substitute_and_nothing_else"); + + /* ROUTE_P: the window is the mass, not M. Ranks 0..4 hold 70/136 = 0.515 of the + * mass, so P=0.5 closes the window at rank 4: rank 4 is inside, rank 5 outside. */ + memset(g_lvl, 0, sizeof g_lvl); g_lvl[4] = 1; g_lvl[5] = 1; memset(&st, 0, sizeof st); + route_select(pr, keep, E, K, 2, 12, 0.5f, 1.f, lvl_table, NULL, idx, val, &st); + check(has(idx, K, 4) && !has(idx, K, 5), "route_p_bounds_the_window_by_cumulative_mass"); + + /* fewer eligible experts than K: the remainder is marked -1, not garbage */ + memset(g_lvl, 0, sizeof g_lvl); memset(&st, 0, sizeof st); + memset(keep, 0, sizeof keep); keep[3] = 1; keep[9] = 1; + route_select(pr, keep, E, K, 2, 12, 0.f, 1.f, lvl_table, NULL, idx, val, &st); + check(idx[0]==3 && idx[1]==9 && idx[2]==-1 && idx[3]==-1 && st.slots == 2, "short_candidate_list_pads_with_minus_one"); + + if (fails) { printf("test_qwen36_cache_route: %d fallimenti\n", fails); return 1; } + printf("test_qwen36_cache_route: ok\n"); + return 0; +} diff --git a/c/tests/test_qwen36_chat_template.py b/c/tests/test_qwen36_chat_template.py new file mode 100644 index 000000000..f3014f5d2 --- /dev/null +++ b/c/tests/test_qwen36_chat_template.py @@ -0,0 +1,161 @@ +#!/usr/bin/env python3 +"""Il renderer di Qwen3.6 nel gateway, contro chat_template.jinja ufficiale. + +Il gateway rende i prompt a mano invece di far girare jinja a ogni richiesta, e +quella scelta si paga in un modo solo: la copia scritta a mano puo' scostarsi +dall'originale senza che nessuno se ne accorga, perche' il modello risponde +comunque. Qui il template vero viene reso con jinja2 e confrontato byte per byte +con quello che produce il gateway, senza strumenti (che il motore qwen36 non +espone) e sui due rami del blocco di ragionamento, piu' il turno aperto della +prosecuzione. + +Se manca il template o jinja2, il test si dichiara SALTATO invece di passare: un +test che non ha trovato il suo riferimento non ha verificato niente, e dirlo +verde sarebbe peggio che non averlo -- percio' un salto esce con codice 2, +distinto dallo 0 di un confronto riuscito. + +RIFERIMENTO (scaricato 2026-09-10): + repo Qwen/Qwen3.6-35B-A3B (repo vendor; il container colibri dava 404) + file chat_template.jinja + sha256 e84f32a23fdda27689f868aa4a1a5621f41133e51a48d7f3efcbea2839574259 + hf download Qwen/Qwen3.6-35B-A3B chat_template.jinja + +USO: + python3 tests/test_qwen36_chat_template.py --template PATH/chat_template.jinja +""" +import argparse +import json +import sys +from pathlib import Path + +CASES = { + "un turno utente": { + "messages": [{"role": "user", "content": "ciao"}], + }, + "sistema piu' utente": { + "messages": [{"role": "system", "content": "Sei conciso."}, + {"role": "user", "content": "capitale della Francia?"}], + }, + "assistant in cronologia": { + "messages": [{"role": "user", "content": "1+1?"}, + {"role": "assistant", "content": "2"}, + {"role": "user", "content": "e 2+2?"}], + }, +} + +THINKING = (True, False) + + +def reference(template_text, *, messages, enable_thinking=True, add_generation_prompt=True): + import jinja2 + + def raise_exception(message): + raise RuntimeError(message) + + environment = jinja2.Environment(trim_blocks=False, lstrip_blocks=False, + extensions=["jinja2.ext.loopcontrols"]) + environment.filters["tojson"] = ( + lambda value, ensure_ascii=False, **kw: json.dumps(value, ensure_ascii=ensure_ascii)) + environment.globals["raise_exception"] = raise_exception + rendered = environment.from_string(template_text) + return rendered.render(messages=messages, add_generation_prompt=add_generation_prompt, + enable_thinking=enable_thinking) + + +def show(label, ours, theirs): + print(f"FAIL {label}") + for index, (a, b) in enumerate(zip(ours.splitlines(), theirs.splitlines())): + if a != b: + print(f" prima differenza alla riga {index + 1}") + print(f" gateway: {a!r}") + print(f" template: {b!r}") + return + print(f" lunghezze diverse: gateway {len(ours)}, template {len(theirs)}") + print(f" coda gateway: {ours[-120:]!r}") + print(f" coda template: {theirs[-120:]!r}") + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--template", type=Path, required=True) + arguments = parser.parse_args() + + if not arguments.template.exists(): + print(f"SKIP: manca {arguments.template}; il riferimento non c'e' e " + f"questo test non ha verificato nulla") + return 2 # salto != successo (vedi docstring) + try: + import jinja2 # noqa: F401 + except ImportError: + print("SKIP: jinja2 non installato; senza non c'e' riferimento") + return 2 # salto != successo (vedi docstring) + + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + import openai_server + + openai_server.ARCH = "qwen36" + template_text = arguments.template.read_text(encoding="utf-8") + + failures = 0 + for enable_thinking in THINKING: + for label, case in CASES.items(): + name = f"{label} [thinking={enable_thinking}]" + theirs = reference(template_text, messages=case["messages"], + enable_thinking=enable_thinking) + ours = openai_server.render_chat_qwen(case["messages"], + enable_thinking=enable_thinking) + if ours == theirs: + print(f"ok {name}") + else: + show(name, ours, theirs) + failures += 1 + + # Prosecuzione: l'ultimo turno assistant e' da CONTINUARE. Come qwen38, ChatML chiude + # ogni turno con <|im_end|> e il template non ha un ramo di continuazione, quindi la + # forma aperta e' il suo add_generation_prompt=False MENO il <|im_end|>\n finale. Qui in + # piu' il turno da proseguire e' DOPO l'ultima domanda, e li' il template scrive il blocco + # (una cronologia piu' vecchia lo perde): la resa aperta lo tiene. + aperto = [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + produced = openai_server.render_chat_for_arch(aperto, enable_thinking=True, + add_generation_prompt=False) + closed = reference(template_text, messages=aperto, add_generation_prompt=False) + # Il terminatore da togliere DEVE esserci nel riferimento: toglierlo "se c'e'" sarebbe un + # no-op silenzioso se il template cambiasse convenzione, e il confronto perderebbe senso. + TERM = "<|im_end|>\n" + if not closed.endswith(TERM): + print(f"FAIL prosecuzione: il riferimento non finisce col terminatore {TERM!r} da " + f"togliere -- convenzione del template cambiata? coda: {closed[-40:]!r}") + failures += 1 + expected = closed + else: + expected = closed[:-len(TERM)] + if produced == expected: + print("ok prosecuzione: turno aperto = template(add_generation_prompt=False) " + "senza il <|im_end|> finale") + else: + show("prosecuzione", produced, expected) + failures += 1 + if not produced.endswith("La capitale e'"): + print(f"FAIL prosecuzione: il prompt non finisce sull'apertura del client: " + f"{produced[-60:]!r}") + failures += 1 + if produced.rstrip("\n").endswith("<|im_end|>"): + print("FAIL prosecuzione: il turno resta chiuso col terminatore") + failures += 1 + # Controllo negativo: col ramo normale lo stesso scambio DEVE finire sulla cue. + if not openai_server.render_chat_for_arch( + aperto, enable_thinking=True).endswith("\n"): + print("FAIL prosecuzione: il ramo normale non emette piu' il prompt di generazione") + failures += 1 + + print() + if failures: + print(f"TEST FAIL ({failures} casi)") + return 1 + print("template Qwen3.6: il gateway e' identico al riferimento") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/c/tests/test_qwen36_ctx.c b/c/tests/test_qwen36_ctx.c index 8a906ed3c..c2f2e6ef0 100644 --- a/c/tests/test_qwen36_ctx.c +++ b/c/tests/test_qwen36_ctx.c @@ -125,10 +125,27 @@ static void case_ceiling(void) { "unset Q36_MAXT falls back to the conservative default"); } +/* #1641: max_tokens is a ceiling, not a target. The gateway's default budget + * (8192, the whole default context) used to be refused on every request + * without max_tokens and on every `coli chat` message; now it is clamped to + * the room the context leaves, and only a prompt that does not fit is refused. + * A read-only logprobs request (max_tokens 0) may fill the context exactly. */ +static void case_budget(void) { + ck(qwen36_serve_budget(2, 8192, 8192, 0) == 8190, "default budget clamped to the room left by the prompt"); + ck(qwen36_serve_budget(100, 50, 8192, 0) == 50, "a budget that fits is untouched"); + ck(qwen36_serve_budget(8191, 1, 8192, 0) == 1, "one token of room is enough"); + ck(qwen36_serve_budget(8192, 1, 8192, 0) == -1, "a prompt that leaves no room is refused"); + ck(qwen36_serve_budget(9000, 1, 8192, 0) == -1, "a prompt longer than the context is refused"); + ck(qwen36_serve_budget(0, 1, 8192, 0) == -1, "an empty prompt is refused"); + ck(qwen36_serve_budget(8192, 0, 8192, 1) == 0, "read-only: the prompt may fill the context"); + ck(qwen36_serve_budget(8193, 0, 8192, 1) == -1, "read-only: past the context is still refused"); +} + int main(void) { case_layout(); case_growth(); case_ceiling(); + case_budget(); if (fails) { printf("FAILED %d\n", fails); return 1; } printf("OK test_qwen36_ctx\n"); return 0; diff --git a/c/tests/test_qwen36_dense_batch.c b/c/tests/test_qwen36_dense_batch.c index 69288725c..d513e814d 100644 --- a/c/tests/test_qwen36_dense_batch.c +++ b/c/tests/test_qwen36_dense_batch.c @@ -48,21 +48,34 @@ static void one_shape(int S, int I, int O) { free(x);free(q);free(sc);free(ref);free(got); } -static void clear_qdw(void) { - for(int i=0;ish_g); qw_free(&l->sh_u); qw_free(&l->sh_d); } static void shared_case(const char *format,int quantized) { enum { S=12,D=64,I=32 }; Model m;memset(&m,0,sizeof(m));m.c.hidden=D;m.c.shared_inter=I; Layer l;memset(&l,0,sizeof(l)); - l.sh_g=falloc((int64_t)I*D);l.sh_u=falloc((int64_t)I*D); - l.sh_d=falloc((int64_t)D*I);l.sh_gate=falloc(D); - for(int64_t i=0;i<(int64_t)I*D;i++){l.sh_g[i]=input_value(i,2);l.sh_u[i]=input_value(i,3);} - for(int64_t i=0;i<(int64_t)D*I;i++)l.sh_d[i]=input_value(i,4); + l.sh_g.w=falloc((int64_t)I*D);l.sh_u.w=falloc((int64_t)I*D); + l.sh_d.w=falloc((int64_t)D*I);l.sh_gate=falloc(D); + l.sh_g.I=D;l.sh_g.O=I; l.sh_u.I=D;l.sh_u.O=I; l.sh_d.I=I;l.sh_d.O=D; + for(int64_t i=0;i<(int64_t)I*D;i++){((float*)l.sh_g.w)[i]=input_value(i,2);((float*)l.sh_u.w)[i]=input_value(i,3);} + for(int64_t i=0;i<(int64_t)D*I;i++)((float*)l.sh_d.w)[i]=input_value(i,4); for(int i=0;i 3\n",format,S,S*3); - clear_qdw();free(l.sh_g);free(l.sh_u);free(l.sh_d);free(l.sh_gate); + clear_qw(&l);free(l.sh_gate); free(x);free(seed);free(ref);free(got);free(g);free(u);free(hh); } diff --git a/c/tests/test_qwen36_dense_idot.c b/c/tests/test_qwen36_dense_idot.c new file mode 100644 index 000000000..c506c81ae --- /dev/null +++ b/c/tests/test_qwen36_dense_idot.c @@ -0,0 +1,128 @@ +/* The dense trunk's integer path (COLI_DENSE_IDOT, COLI_DENSE_BITS=4). + * + * Pins, with no model: + * - dense_act_i8's vector path equals the scalar lrintf contract bit for bit + * (same rounding, same scale) and its block sums are exact; + * - pack_int4_g64_planar stores what the K1b layout expects: nibble v+8, + * lo nibbles = elements 0..31 of the block, hi = 32..63, one scale per 64; + * - matmul_d through the int4 grouped kernel and through the int8 integer + * kernel both agree with the f32 reference within what an int8 activation + * and an int4 weight allow, on every row of a random matrix, for one row + * (decode) and for several rows (prefill); + * - with neither flag the dispatch is the one before: the int8 f32-activation + * kernel, byte-identical to its own output. */ +#include +#include +#include +#include + +#define main qwen36_main_unused +#include "../qwen36.c" +#undef main +#include "../compat.h" /* setenv: MinGW has none */ + +static int fails; +static void ck(int ok, const char *what) { if (ok) { printf(" ok %s\n", what); return; } printf(" FAIL %s\n", what); fails++; } + +static unsigned g_seed = 4242; +static float rnd(void) { g_seed = g_seed * 1103515245u + 12345u; return ((g_seed >> 8) & 0xFFFF) / 32768.f - 1.f; } + +static void ref_matmul(float *y, const float *x, const float *W, int S, int I, int O) { + for (int s = 0; s < S; s++) for (int o = 0; o < O; o++) { + double a = 0; for (int i = 0; i < I; i++) a += (double)x[(size_t)s * I + i] * W[(size_t)o * I + i]; + y[(size_t)s * O + o] = (float)a; + } +} +static double rel_gap(const float *a, const float *b, int n) { + double worst = 0, scale = 1e-6; + for (int i = 0; i < n; i++) { double d = fabs((double)a[i] - b[i]); if (d > worst) worst = d; if (fabs((double)b[i]) > scale) scale = fabs((double)b[i]); } + return worst / scale; +} + +int main(void) { + enum { I = 256, O = 96, S = 5 }; + printf("activation quantizer\n"); + { + float x[I]; int8_t a[I], b[I]; int32_t sums[I / 64]; + for (int t = 0; t < 50; t++) { + float mag = (t % 5 == 0) ? 1e-3f : 3.f; + for (int i = 0; i < I; i++) x[i] = rnd() * mag; + float sa = dense_act_i8(x, I, a, sums); + float sb = qrow_i8(x, b, I); + if (sa != sb || memcmp(a, b, I)) { ck(0, "vector quantizer equals the scalar contract"); break; } + for (int g = 0; g < I / 64; g++) { int32_t s = 0; for (int k = 0; k < 64; k++) s += a[g * 64 + k]; if (s != sums[g]) { ck(0, "block sums are exact"); break; } } + if (t == 49) ck(1, "vector quantizer equals the scalar lrintf contract on 50 random rows, block sums exact"); + } + } + printf("int4 planar packing\n"); + { + float *W = malloc((size_t)O * I * sizeof(float)); + for (size_t i = 0; i < (size_t)O * I; i++) W[i] = rnd(); + uint8_t *q4 = malloc((size_t)O * I / 2); float *sg = malloc((size_t)O * (I / 64) * sizeof(float)); + pack_int4_g64_planar(W, q4, sg, O, I); + int bad = 0; + for (int o = 0; o < O && !bad; o++) for (int g = 0; g < I / 64 && !bad; g++) { + const uint8_t *blk = q4 + (size_t)o * (I / 2) + g * 32; + float s = sg[(size_t)o * (I / 64) + g]; + for (int k = 0; k < 64; k++) { + int nib = k < 32 ? (blk[k] & 0xF) : (blk[k - 32] >> 4); + float w = W[(size_t)o * I + g * 64 + k]; + int want = (int)lrintf(w / s); if (want > 7) want = 7; if (want < -8) want = -8; + if (nib - 8 != want) { bad = 1; break; } + } + } + ck(!bad, "nibble v+8 in the planar order, one absmax/7 scale per block of 64"); + free(W); free(q4); free(sg); + } + printf("matmul_d dispatch\n"); + { + float *W = malloc((size_t)O * I * sizeof(float)), *x = malloc((size_t)S * I * sizeof(float)); + float *ref = malloc((size_t)S * O * sizeof(float)), *y = malloc((size_t)S * O * sizeof(float)), *y0 = malloc((size_t)S * O * sizeof(float)); + for (size_t i = 0; i < (size_t)O * I; i++) W[i] = rnd(); + for (size_t i = 0; i < (size_t)S * I; i++) x[i] = rnd() * 2.f; + ref_matmul(ref, x, W, S, I, O); + /* the classic path (no flags): int8 rows, f32 activations. + * dense_idot_on caches its answer: this TU reads the env once, so + * set it before the first call (matmul_d, via qw_quantize below). */ + setenv("COLI_DENSE_IDOT", "0", 1); setenv("COLI_DENSE_BITS", "8", 1); + QW w = {0}; qw_quantize(W, I, O, NULL, &w); + matmul_d(y0, x, &w, S, I, O); + double g0 = rel_gap(y0, ref, S * O); + ck(g0 < 2e-2, "classic int8 path within 2% of the f32 reference (per-row int8)"); + matmul_d(y, x, &w, 1, I, O); + ck(!memcmp(y, y0, (size_t)O * sizeof(float)), "one row and the first of a batch agree on the classic path"); + qw_free(&w); + free(W); free(x); free(ref); free(y); free(y0); + } + { + /* fresh process-wide state for the flag readers: emulate by direct calls */ + float *W = malloc((size_t)O * I * sizeof(float)), *x = malloc((size_t)S * I * sizeof(float)); + float *ref = malloc((size_t)S * O * sizeof(float)), *y = malloc((size_t)S * O * sizeof(float)); + for (size_t i = 0; i < (size_t)O * I; i++) W[i] = rnd(); + for (size_t i = 0; i < (size_t)S * I; i++) x[i] = rnd() * 2.f; + ref_matmul(ref, x, W, S, I, O); + int8_t *q = malloc((size_t)O * I); float *sc = malloc((size_t)O * sizeof(float)); + for (int o = 0; o < O; o++) { /* the per-row int8 the engine builds at load */ + float am = 0.f; for (int i = 0; i < I; i++) { float a = fabsf(W[(size_t)o * I + i]); if (a > am) am = a; } + float s = am > 1e-12f ? am / 127.f : 1.f; sc[o] = s; + for (int i = 0; i < I; i++) { int v = (int)lrintf(W[(size_t)o * I + i] / s); if (v > 127) v = 127; if (v < -127) v = -127; q[(size_t)o * I + i] = (int8_t)v; } + } + int8_t *xq = malloc((size_t)S * I); float sx[S]; int32_t *xsg = malloc((size_t)S * (I / 64) * sizeof(int32_t)); + for (int s = 0; s < S; s++) sx[s] = dense_act_i8(x + (size_t)s * I, I, xq + (size_t)s * I, xsg + (size_t)s * (I / 64)); + matmul_q_idot(y, xq, sx, q, sc, S, I, O); + ck(rel_gap(y, ref, S * O) < 3e-2, "int8 weights x int8 activations within 3% of the f32 reference, 5 rows"); + matmul_q_idot(y, xq, sx, q, sc, 1, I, O); + ck(rel_gap(y, ref, O) < 3e-2, "same for one row"); + uint8_t *q4 = malloc((size_t)O * I / 2); float *sg = malloc((size_t)O * (I / 64) * sizeof(float)); + pack_int4_g64_planar(W, q4, sg, O, I); + matmul_i4p_grouped_idot(y, xq, sx, xsg, q4, sg, S, I, O, 64); + ck(rel_gap(y, ref, S * O) < 8e-2, "int4 blocks of 64 x int8 activations within 8% of the f32 reference, 5 rows"); + float y1[O]; + matmul_i4p_grouped_idot(y1, xq, sx, xsg, q4, sg, 1, I, O, 64); + ck(!memcmp(y1, y, (size_t)O * sizeof(float)), "the one-row path and the row tile give the same bytes (the K1b contract)"); + free(W); free(x); free(ref); free(y); free(q); free(sc); free(xq); free(xsg); free(q4); free(sg); + } + if (fails) { printf("test_qwen36_dense_idot: %d failure(s)\n", fails); return 1; } + printf("OK test_qwen36_dense_idot: the dense trunk's integer path\n"); + return 0; +} diff --git a/c/tests/test_qwen36_dnproj_batch.c b/c/tests/test_qwen36_dnproj_batch.c new file mode 100644 index 000000000..60ed8f4d0 --- /dev/null +++ b/c/tests/test_qwen36_dnproj_batch.c @@ -0,0 +1,140 @@ +/* Run DeltaNet itself: batching must preserve outputs, conv history and state. */ +#define main qwen_main_unused +#include "../qwen36.c" +#undef main + +/* Define COLI_DNPROJ_REAL_CUDA and link --wrap=coli_cuda_matmul to run + * the same state-parity test against the real CUDA backend on Linux. */ +#ifdef COLI_DNPROJ_REAL_CUDA +#include "../backend_cuda.h" +int __real_coli_cuda_matmul(ColiCudaTensor **,float *,const float *,const void *, + const float *,int,int,int,int,int,int); +#define dn_fake_matmul __real_coli_cuda_matmul +#define DN_MATMUL __wrap_coli_cuda_matmul +#else +/* Wrap the shared fake so this test can inject a failure at a block boundary. */ +#define coli_cuda_matmul dn_fake_matmul +#include "qwen36_fake_cuda.h" +#undef coli_cuda_matmul +#define DN_MATMUL coli_cuda_matmul +#endif +static int calls, fail_call, max_rows, last_rows; +int DN_MATMUL(ColiCudaTensor **t,float *y,const float *x,const void *w, + const float *sc,int fmt,int S,int I,int O,int dev,int gs) { + calls++; if (S > max_rows) max_rows = S; last_rows = S; + if (calls == fail_call) { for(int i=0;iDN_rec = calloc(1,sizeof(float *)); m->DN_conv = calloc(1,sizeof(float *)); + m->DN_rec[0] = calloc(VH*KD*VD,sizeof(float)); m->DN_conv[0] = calloc(C*(CK-1),sizeof(float)); +} +static void clear_state(Model *m) { + memset(m->DN_rec[0],0,VH*KD*VD*sizeof(float)); + memset(m->DN_conv[0],0,C*(CK-1)*sizeof(float)); +} +static void agree(const float *a,const float *b,int n) { + for(int i=0;i 2"); + return failed!=0; +} diff --git a/c/tests/test_qwen36_encode_oom.c b/c/tests/test_qwen36_encode_oom.c new file mode 100644 index 000000000..1849bd670 --- /dev/null +++ b/c/tests/test_qwen36_encode_oom.c @@ -0,0 +1,147 @@ +/* push_id and bpe_piece's growth reallocs were unchecked: a failed realloc left the + * buffer NULL and the very next store wrote through it. Injects a real allocation + * failure (not just a code read) and asserts the engine refuses loudly instead of + * crashing on a NULL write, mirroring tests/test_798_guards.c's shadow-realloc + * technique -- qwen38.c's own push_id/bpe_piece already guard this exact pattern + * (q38_encode_realloc / q38_encode_oom); qwen36.c never got the equivalent fix. */ +#ifndef _GNU_SOURCE +#define _GNU_SOURCE +#endif +#include +#include +#include +#ifndef _WIN32 +#include +#include +#endif + +#ifndef _WIN32 +/* Forward-declare the shadow under its own name before the macro below hijacks every + * "realloc" token that follows, mirroring tests/test_798_guards.c's technique. */ +static void *test_realloc_seam(void *p, size_t n); +#define realloc test_realloc_seam +#endif + +#define main qwen36_main_unused +#include "../qwen36.c" +#undef main + +#ifndef _WIN32 +#undef realloc + +static long g_realloc_n = 0, g_realloc_fail_at = -1; + +static void *test_realloc_seam(void *p, size_t n){ + g_realloc_n++; + if (g_realloc_n == g_realloc_fail_at) return NULL; + return realloc(p, n); +} +static void reset_seam(void){ g_realloc_n = 0; g_realloc_fail_at = -1; } +#endif + +static int g_nfails = 0; +static void check(int cond, const char *what){ + if (!cond) { printf("FAIL: %s\n", what); g_nfails++; } +} + +#ifndef _WIN32 +/* fork/pipe/waitpid harness, mirrors tests/test_798_guards.c's inline idiom */ +static void run_forked(void (*fn)(void), int *exit_code, char *errbuf, size_t errbuf_sz) { + int pipefd[2]; + if (pipe(pipefd) != 0) { *exit_code = -1; errbuf[0] = 0; return; } + pid_t pid = fork(); + if (pid < 0) { *exit_code = -1; errbuf[0] = 0; return; } + if (pid == 0) { + dup2(pipefd[1], 2); close(pipefd[0]); close(pipefd[1]); + fn(); + _exit(42); /* reaching here means fn() did NOT refuse -- a bug, not a crash */ + } + close(pipefd[1]); + size_t off = 0; ssize_t n; + while (off < errbuf_sz - 1 && (n = read(pipefd[0], errbuf + off, errbuf_sz - 1 - off)) > 0) off += (size_t)n; + errbuf[off] = 0; + close(pipefd[0]); + int status = 0; waitpid(pid, &status, 0); + *exit_code = WIFEXITED(status) ? WEXITSTATUS(status) : -1; +} + +/* push_id: cap starts at 2, so the 3rd push (n==cap) is the realloc growth call -- + * the first and only realloc in a fresh child, so ordinal 1 targets it precisely. */ +static void fn_push_id_grow(void) { + int *ids = malloc(2 * sizeof(int)); + int n = 0, cap = 2; + push_id(&ids, &n, &cap, 10); + push_id(&ids, &n, &cap, 20); + push_id(&ids, &n, &cap, 30); /* n==cap(2) here: triggers the realloc */ + free(ids); +} + +/* bpe_piece: scap starts at 16, so a 17-byte piece triggers exactly one realloc (the + * byte-symbol table growth) before the merge loop is ever reached -- an empty + * (never-loaded) merge table is never consulted on the failing path. */ +static void fn_bpe_piece_grow(void) { + build_byte_sym(); + int cap = 16, n = 0; int *ids = malloc((size_t)cap * sizeof(int)); + const char *piece = "abcdefghijklmnopq"; /* 17 bytes */ + bpe_piece(piece, (int)strlen(piece), &ids, &n, &cap); + free(ids); +} + +static void test_push_id_oom(void) { + reset_seam(); + g_realloc_fail_at = 1; + int exit_code; char err[512]; + run_forked(fn_push_id_grow, &exit_code, err, sizeof(err)); + check(exit_code == 1, "push_id: exits(1) on injected realloc failure"); + check(strstr(err, "OOM reallocating token id buffer") != NULL, + "push_id: message names the buffer"); + reset_seam(); +} + +static void test_bpe_piece_oom(void) { + reset_seam(); + g_realloc_fail_at = 1; + int exit_code; char err[512]; + run_forked(fn_bpe_piece_grow, &exit_code, err, sizeof(err)); + check(exit_code == 1, "bpe_piece: exits(1) on injected realloc failure"); + check(strstr(err, "OOM reallocating BPE symbol buffer") != NULL, + "bpe_piece: message names the buffer"); + reset_seam(); +} +#endif + +/* ---- controls: same growth paths, no injection -- must succeed normally ---- */ +static void test_push_id_control(void) { + int *ids = malloc(2 * sizeof(int)); + int n = 0, cap = 2; + push_id(&ids, &n, &cap, 10); + push_id(&ids, &n, &cap, 20); + push_id(&ids, &n, &cap, 30); + check(n == 3 && cap >= 3 && ids[0] == 10 && ids[1] == 20 && ids[2] == 30, + "push_id control: grows and stores correctly without injection"); + free(ids); +} + +static void test_bpe_piece_control(void) { + build_byte_sym(); + int cap = 16, n = 0; int *ids = malloc((size_t)cap * sizeof(int)); + const char *piece = "abcdefghijklmnopq"; + bpe_piece(piece, (int)strlen(piece), &ids, &n, &cap); + check(n == 17, "bpe_piece control: one id per byte with no merge table loaded"); + free(ids); +} + +int main(void) { +#ifndef _WIN32 + test_push_id_oom(); + test_bpe_piece_oom(); +#else + printf("qwen36 encode allocation-failure injection: skipped on Windows (no fork)\n"); +#endif + test_push_id_control(); + test_bpe_piece_control(); + + if (g_nfails) { printf("%d check(s) failed\n", g_nfails); return 1; } + printf("OK\n"); + return 0; +} diff --git a/c/tests/test_qwen36_slot_int8.c b/c/tests/test_qwen36_slot_int8.c new file mode 100644 index 000000000..80d60391e --- /dev/null +++ b/c/tests/test_qwen36_slot_int8.c @@ -0,0 +1,103 @@ +/* Regression gate for wiring slot_ensure_int8() onto #1271's unpack_int4_to_int8 + * (the same branchless AVX2/NEON/scalar kernel load_expert_merged already uses + * for the container's int4 read) instead of its own separate scalar + * nibble-unpack loop. Integer nibble extraction + sign-extend is bit-exact by + * construction (no reduction, no rounding), so this compares the NEW + * slot_ensure_int8 (calling unpack_int4_to_int8) against a copy of the OLD + * scalar loop it replaced -- not just "it compiles", a real regression check + * that the two nibble-unpack formulas produce identical bytes. */ +#define main qwen36_main_unused +#include "../qwen36.c" +#undef main + +static int failures; +#define CHECK(cond, ...) do { if (!(cond)) { \ + fprintf(stderr,"FAIL %s:%d: ",__FILE__,__LINE__); \ + fprintf(stderr,__VA_ARGS__); fputc('\n',stderr); failures++; } } while (0) + +/* Verbatim copy of slot_ensure_int8's pre-#1271 scalar loop (the code this + * change replaces), used here only as the regression oracle. */ +static void old_unpack(int8_t *dst, const uint8_t *p, int64_t len) { + for (int64_t i = 0; i < len; i += 2) { + uint8_t b = p[i >> 1]; + int8_t lo = (int8_t)(b & 0xF); if (lo & 8) lo -= 16; + int8_t hi = (int8_t)((b >> 4) & 0xF); if (hi & 8) hi -= 16; + dst[i] = lo; dst[i + 1] = hi; + } +} + +static uint8_t *random_packed(int64_t nbytes, uint32_t seed) { + uint8_t *p = malloc((size_t)nbytes); + uint32_t x = seed ? seed : 1; + for (int64_t i = 0; i < nbytes; i++) { + x ^= x << 13; x ^= x >> 17; x ^= x << 5; /* xorshift32 */ + p[i] = (uint8_t)x; + } + return p; +} + +/* inter/hidden chosen so ng=inter*hidden and nd=hidden*inter land on both + * sides of the AVX2 16-byte (32-element) vector boundary across the three + * segments (g/u sized ng, d sized nd). ng/nd must stay even: nibble-packing + * always yields ceil(n/2) bytes, and real containers guarantee even + * inter*hidden (SIMD/GS64-block-aligned dims), so old_unpack -- a verbatim + * copy of the pre-#1271 scalar loop, kept unmodified on purpose -- was never + * written to handle an odd total (its last iteration writes dst[i+1] + * unconditionally) and shouldn't be, either: that shape cannot occur. */ +static void one_case(int inter, int hidden, const char *label) { + Model m; memset(&m, 0, sizeof(m)); + m.c.inter = inter; m.c.hidden = hidden; + int64_t ng = (int64_t)inter * hidden, nd = (int64_t)hidden * inter; + CHECK(ng % 2 == 0, "%s: ng must be even (nibble-packed, see one_case)", label); + if (ng % 2) return; + + Slot s; memset(&s, 0, sizeof(s)); + s.g4 = random_packed(ng / 2, 1001); + s.u4 = random_packed(ng / 2, 2002); + s.d4 = random_packed(nd / 2, 3003); + + slot_ensure_int8(&m, &s); + CHECK(s.g != NULL, "%s: slot_ensure_int8 did not populate s->g", label); + if (!s.g) return; + + int8_t *exp_g = malloc((size_t)ng), *exp_u = malloc((size_t)ng), *exp_d = malloc((size_t)nd); + old_unpack(exp_g, s.g4, ng); + old_unpack(exp_u, s.u4, ng); + old_unpack(exp_d, s.d4, nd); + + CHECK(!memcmp(s.g, exp_g, (size_t)ng), "%s: gate segment differs from old scalar loop", label); + CHECK(!memcmp(s.u, exp_u, (size_t)ng), "%s: up segment differs from old scalar loop", label); + CHECK(!memcmp(s.d, exp_d, (size_t)nd), "%s: down segment differs from old scalar loop", label); + printf("qwen36 slot_ensure_int8 exact: %s inter=%d hidden=%d (ng=%lld nd=%lld)\n", + label, inter, hidden, (long long)ng, (long long)nd); + + free(exp_g); free(exp_u); free(exp_d); + free(s.g); /* s.u/s.d point into the same block as s.g (slot_ensure_int8's `w`) */ + free(s.g4); free(s.u4); free(s.d4); +} + +int main(void) { + one_case(4, 8, "tiny, well under one vector"); /* ng=32, nd=32 */ + one_case(16, 16, "exact AVX2 vector boundary"); /* ng=256, nd=256, /16=16 (mult of 16-byte block) */ + one_case(17, 16, "vector body + scalar tail"); /* ng=272 -> nb=136, not mult of 16 */ + one_case(37, 42, "odd, real-expert-shaped"); /* ng=1554, nd=1554 (even: see one_case) */ + one_case(2048, 512, "large, real Qwen3.6 expert scale"); /* ng=1048576, nd=1048576 */ + + /* re-entrancy guard: slot_ensure_int8 must no-op when s->g is already set + * (the "rematerialize only if evicted" contract) -- unrelated to which + * unpack kernel runs, but worth pinning since this function was touched */ + { + Model m; memset(&m, 0, sizeof(m)); m.c.inter = 4; m.c.hidden = 8; + Slot s; memset(&s, 0, sizeof(s)); + s.g4 = random_packed(16, 42); s.u4 = random_packed(16, 43); s.d4 = random_packed(16, 44); + slot_ensure_int8(&m, &s); + int8_t *g_before = s.g; + slot_ensure_int8(&m, &s); /* should be a no-op: s->g already non-NULL */ + CHECK(s.g == g_before, "slot_ensure_int8 re-ran when s->g was already set"); + free(s.g); free(s.g4); free(s.u4); free(s.d4); + } + + if (failures) { fprintf(stderr, "qwen36 slot_ensure_int8: %d failure(s)\n", failures); return 1; } + puts("qwen36 slot_ensure_int8: ok"); + return 0; +} diff --git a/c/tests/test_qwen36_tier_dense.c b/c/tests/test_qwen36_tier_dense.c index 1a63c41d9..72712664d 100644 --- a/c/tests/test_qwen36_tier_dense.c +++ b/c/tests/test_qwen36_tier_dense.c @@ -22,7 +22,7 @@ int main(void) { setenv("COLI_CUDA", "1", 1); setenv("COLI_GPUS", "0", 1); setenv("QT_NO_WARMSTART", "1", 1); setenv("HEAT_FILE", "", 1); setenv("COLI_PLACE", "", 1); /* "" == unset == auto; unsetenv has no UCRT64 shim */ - fake_ndev = 1; fake_uploads = 0; + fake_ndev = 1; fake_uploads = 0; fake_dense_compute = 1; /* room for 8 experts plus a little; each offer is 1 expert worth of bytes */ size_t exp_bytes = 3 * dev_alloc_footprint((size_t)D * IH / 2) + 3 * dev_alloc_footprint((2 * IH + D) / 3 * sizeof(float)); @@ -66,6 +66,29 @@ int main(void) { check(qt_dense_matmul(h2, y, x, I, O) == 1 && fake_matmuls == mm + 2, "so does the second handle"); check(qt_dense_matmul(99, y, x, I, O) == 0 && qt_dense_matmul(-1, y, x, I, O) == 0 && fake_matmuls == mm + 2, "unknown handles are refused without a backend call"); + { + float xb[3 * I], yb[3 * O]; + for (int i = 0; i < 3 * I; i++) xb[i] = (float)(i % 7 - 3); + check(qt_dense_matmul_batch(h, yb, xb, 3, I, O), "batch accepted"); + check(fake_matmul_rows == 3, "all rows sent in one backend call"); + int equal = 1; + for (int row = 0; row < 3; row++) for (int o = 0; o < O; o++) { + float want = 0; + for (int i = 0; i < I; i++) want += xb[row * I + i] * q[o * I + i]; + if (yb[row * O + o] != want * sc[o]) equal = 0; + } + check(equal, "every batch row uses the resident weights and row scales"); + int calls = fake_matmuls; + check(!qt_dense_matmul_batch(h, yb, xb, 0, I, O) && fake_matmuls == calls, + "empty batch declined without a backend call"); + fake_matmul_fail = 1; + check(!qt_dense_matmul_batch(h, yb, xb, 3, I, O), "batch failure requests CPU fallback"); + fake_matmul_fail = 0; + calls = fake_matmuls; + check(!qt_dense_matmul(h, yb, xb, I, O) && fake_matmuls == calls, + "failed handle remains disabled for decode too"); + } + printf(" 4. shutdown releases the matrices\n"); qt_shutdown(); check(qt_dense_count() == 0, "dense handles released at shutdown"); diff --git a/c/tests/test_qwen36_tier_init_failure.c b/c/tests/test_qwen36_tier_init_failure.c new file mode 100644 index 000000000..985ae01b6 --- /dev/null +++ b/c/tests/test_qwen36_tier_init_failure.c @@ -0,0 +1,100 @@ +/* Every failed initialization must unwind only the resources it acquired. */ +#include +#include +#include +#include +#include "../compat.h" +#include "qwen36_fake_cuda.h" +static void *owned[128]; +static int live, mutexes, conds, fail_stage, sync_step; +static void *remember(void *p) { + if(p) { if(live==128) abort(); owned[live++]=p; } + return p; +} +static void *test_malloc(size_t n) { + if(fail_stage==1 && n==32*64*sizeof(float)) return NULL; + return remember(malloc(n)); +} +static void *test_calloc(size_t n,size_t s) { return remember(calloc(n,s)); } +static void test_free(void *p) { + if(p) for(int i=0;iqueued; - pthread_mutex_unlock(&G.mx); - check(qs(0, 0)->resident == 1 && qs(0, 0)->tg != NULL, "the victim of an abandoned swap keeps its tensor and its resident flag"); - check(stale_queued == 0, "no expert may still read as queued after shutdown drained or abandoned the queue"); + check(!G.slot && !G.is_x && !G.waiters, + "shutdown releases storage only after every parked caller has resumed"); } /* ======================================================================== */ diff --git a/c/tests/test_qwen36_tier_release.c b/c/tests/test_qwen36_tier_release.c new file mode 100644 index 000000000..27873845d --- /dev/null +++ b/c/tests/test_qwen36_tier_release.c @@ -0,0 +1,68 @@ +/* Owned tensors must be released before backend shutdown, including standalone + * dense projections. A pending expert group must finish before its weights go. */ +#include +#include +#include "../compat.h" +#define coli_cuda_tensor_free fixture_free +#define coli_cuda_shutdown fixture_shutdown +#define coli_cuda_expert_group_take fixture_take +#include "qwen36_fake_cuda.h" +#undef coli_cuda_tensor_free +#undef coli_cuda_shutdown +#undef coli_cuda_expert_group_take +static int freed, stops, pending, failures; +static ColiCudaTensor *borrowed[3]; +#define CHECK(c) do { if(!(c)){fprintf(stderr,"FAIL %d: %s\n",__LINE__,#c);failures++;} } while(0) +void coli_cuda_tensor_free(ColiCudaTensor *t) { + if(!t) return; + for(int i=0;i<3;i++) CHECK(!pending || t!=borrowed[i]); + freed++; fixture_free(t); +} +void coli_cuda_shutdown(void) { + CHECK(!pending); CHECK(freed==fake_uploads); stops++; +} +const float *coli_cuda_expert_group_take(int device) { pending=0; return NULL; } +static int issue(int device,int count,const float *x) { pending=1; return 1; } +#include "../qwen36_tier.c" +int main(void) { + enum { D=64, I=32 }; + int8_t q[D*D]; float sc[D], x[D], y[D]; + memset(q,1,sizeof q); + for(int i=0;i=0); + CHECK(qt_dnproj_matmul(0,y,x,D,D)); + CHECK(y[0]==D*round); + qt_shutdown(); + CHECK(freed==fake_uploads); CHECK(!G_dnp[0].t && !G_dnp[0].on); + CHECK(!qt_dnproj_matmul(0,y,x,D,D)); CHECK(qt_dense_count()==0); + int before=freed; qt_shutdown(); CHECK(freed==before); CHECK(stops==0); + } + setenv("COLI_CUDA","1",1); setenv("COLI_GPUS","0",1); + setenv("COLI_PLACE","experts=0,lmhead=0,dnproj=0",1); + setenv("QT_NO_WARMSTART","1",1); setenv("HEAT_FILE","",1); + setenv("CUDA_EXPERT_GB","0.0625",1); + for(int round=0;round<2;round++) { + CHECK(qt_init(1,2,D,I,2,1,0,1)); + CHECK(qt_dnproj_init(0,q,sc,D,D,0)); + CHECK(qt_lmhead_init(q,sc,D,D)); + CHECK(qt_dense_init(q,sc,D,D,0)>=0); + uint8_t g[D*I/2]={0}, u[D*I/2]={0}, d[D*I/2]={0}; + qt_note(0,0,g,u,d,sc,sc,sc); qt_fill_wait(); + CHECK(qt_is_resident(0,0)); + borrowed[0]=qs(0,0)->tg; borrowed[1]=qs(0,0)->tu; borrowed[2]=qs(0,0)->td; + fake_issue_hook=issue; + int eid=0; CHECK(qt_issue(0,&eid,1,x)==1); + qt_shutdown(); + CHECK(!pending); CHECK(freed==fake_uploads); CHECK(!G.on); + CHECK(!G.slot && !G.is_x && !G.fill_order && !G.heat0); + CHECK(!G_lmh.t && !G_lmh.on && !G_lmh.dev_ok); + CHECK(!G_dnp[0].t && !G_dnp[0].on); + int before=freed; qt_shutdown(); CHECK(freed==before); + } + printf("tier release: %s (%d uploads, %d frees)\n",failures?"FAIL":"ok",fake_uploads,freed); + return failures!=0; +} diff --git a/c/tests/test_qwen36_tier_rollback.c b/c/tests/test_qwen36_tier_rollback.c new file mode 100644 index 000000000..89a4a922e --- /dev/null +++ b/c/tests/test_qwen36_tier_rollback.c @@ -0,0 +1,58 @@ +/* Fail each matrix upload in turn: unpublished handles must be reclaimed. */ +#include +#include "../compat.h" +#define coli_cuda_tensor_upload fixture_upload +#define coli_cuda_tensor_upload_g fixture_upload_g +#define coli_cuda_tensor_free fixture_free +#include "qwen36_fake_cuda.h" +#undef coli_cuda_tensor_upload +#undef coli_cuda_tensor_upload_g +#undef coli_cuda_tensor_free +static int attempt, fail_at, freed, expected_fmt, unexpected_fmt; +int coli_cuda_tensor_upload(ColiCudaTensor **t,const void *w,const float *s, + int fmt,int I,int O,int dev) { + if(fmt!=expected_fmt) unexpected_fmt=1; + if(++attempt==fail_at) return 0; + return fixture_upload(t,w,s,fmt,I,O,dev); +} +int coli_cuda_tensor_upload_g(ColiCudaTensor **t,const void *w,const float *s, + int fmt,int I,int O,int dev,int gs) { + if(fmt!=expected_fmt) unexpected_fmt=1; + if(++attempt==fail_at) return 0; + return fixture_upload_g(t,w,s,fmt,I,O,dev,gs); +} +void coli_cuda_tensor_free(ColiCudaTensor *t) { if(t) freed++; fixture_free(t); } +#include "../qwen36_tier.c" +int main(void) { + enum { D=64 }; + unsigned char w[D*D]={0}; float sc[D], lut[256]={0}; + const int formats[]={1,2,4,8}; + for(int i=0;iresident || s->queued || s->tg || s->tu || s->td || + G.inflight || G.qn || G.used[0]) { + fprintf(stderr,"rollback failed: fmt=%d matrix=%d uploads=%d frees=%d\n", + expected_fmt,fail_at,fake_uploads,freed); return 1; + } + if(expected_fmt==8 && (s->g4 || s->u4 || s->d4 || s->gs || s->us || s->ds)) { + fputs("FP8 failure retained borrowed slot pointers\n",stderr); return 1; + } + qt_shutdown(); + /* This fixture targets failed uploads; older tier teardown retains + * its host allocations. Release those here for leak sanitizers. */ + free(G.slot); G.slot=NULL; free(G.is_x); G.is_x=NULL; + free(G.fill_order); G.fill_order=NULL; free(G.heat0); G.heat0=NULL; + } + puts("tier upload rollback: ok (12 failure cases: int8, per-row int4, grouped int4, FP8)"); return 0; +} diff --git a/c/tests/test_qwen36_tier_scale_budget.c b/c/tests/test_qwen36_tier_scale_budget.c new file mode 100644 index 000000000..bb2ef81ea --- /dev/null +++ b/c/tests/test_qwen36_tier_scale_budget.c @@ -0,0 +1,34 @@ +/* Separate scale allocations can cross different allocator size classes. */ +#include +#include "../compat.h" +#include "qwen36_fake_cuda.h" +#include "../qwen36_tier.c" +int main(void) { + /* D, Ih, group size (0 = int8), bytes/expert, allowance, admitted experts. + * Constants include three weight and three separate scale allocations. */ + const struct { int D,Ih,gs; size_t bytes,budget; int planned; } cases[]={ + {64,2048,0,458752,917503,1}, + {2048,64,0,450560,901119,1}, + {64,3000,0,655360,1310720,2}, + {64,64,0,49152,98304,2}, + {96,2048,64,385024,770047,1}, + }; + setenv("COLI_CUDA","1",1); setenv("COLI_GPUS","0",1); + setenv("QT_NO_WARMSTART","1",1); setenv("HEAT_FILE","",1); + for(size_t i=0;iresident && qs(0, resident_eid)->tg, - "shutdown_abandons_the_swap_instead_of_freeing_the_victim"); - check(!qs(0, 1)->queued && !qs(0, 1)->resident && G.uploads == 1, - "shutdown_abandons_the_swap_instead_of_uploading_the_incoming_expert"); + check(!G.slot && !G.is_x, "shutdown_releases_tier_storage"); + check(G.uploads == 1 && fake_uploads == 3, + "shutdown_abandons_the_swap_without_uploading_the_incoming_expert"); + + /* A synchronous issue parked behind an upload must not submit new work + * after shutdown wakes it. Hold the inflight predicate explicitly so the + * stop transition is deterministic, independent of upload timing. */ + if(!qt_init(1,2,D,32,2,1,0,1)) return 1; + qt_note(0,0,g4[0],u4[0],d4[0],sc[0],sc[0]+32,sc[0]+64); + qt_fill_wait(); + G_upload_sync=1; + pthread_mutex_lock(&G.mx); + G.inflight=1; + pthread_mutex_unlock(&G.mx); + pthread_t issuer; + if(pthread_create(&issuer,NULL,waiting_issue,NULL)) return 1; + int waiting=0; + for(int i=0;i<500;i++){ + pthread_mutex_lock(&G.mx); + waiting=G.waiters>0; + pthread_mutex_unlock(&G.mx); + if(waiting) break; + struct timespec ts={0,2000000}; nanosleep(&ts,NULL); + } + check(waiting,"issuer parked before stop"); + pthread_mutex_lock(&G.mx); + G.th_stop=1; + pthread_cond_broadcast(&G.cv_take); + pthread_cond_signal(&G.cv); + pthread_mutex_unlock(&G.mx); + pthread_join(issuer,NULL); + check(stopped_issue_mask==0,"stopped issuer must return CPU fallback mask"); + check(!G.issue_open && !G.is_cnt[0],"stopped issuer must not open a GPU group"); + qt_shutdown(); if (fails) { printf("test_qwen36_tier_shutdown: %d fallimenti\n", fails); return 1; } printf("test_qwen36_tier_shutdown: ok\n"); diff --git a/c/tests/test_qwen36_tier_take_error.c b/c/tests/test_qwen36_tier_take_error.c new file mode 100644 index 000000000..e56025045 --- /dev/null +++ b/c/tests/test_qwen36_tier_take_error.c @@ -0,0 +1,34 @@ +/* A failed device must not silently drop an expert or publish a partial sum. */ +#include +#define coli_cuda_expert_group_take fixture_take +#include "qwen36_fake_cuda.h" +#undef coli_cuda_expert_group_take +static int failed=-1, calls[2]; +static float rows[2][2]={{2,4},{8,12}}; +const float *coli_cuda_expert_group_take(int device) { + calls[device]++; + return device==failed?NULL:rows[device]; +} +#include "../qwen36_tier.c" +int main(void) { + G.on=1;G.ndev=2;G.D=2;G.dev[0]=0;G.dev[1]=1; + pthread_mutex_init(&G.mx,NULL);pthread_cond_init(&G.cv_take,NULL); + for(failed=-1;failed<2;failed++){ + G.issue_open=1;G.is_cnt[0]=G.is_cnt[1]=1; + G.is_k[0][0]=0;G.is_k[1][0]=1;calls[0]=calls[1]=0; + float val[2]={0.5f,0.25f},out[2]={10,20}; + int ok=qt_take(3,val,2,out); + float want0=failed<0?13:10,want1=failed<0?25:20; + if(ok!=(failed<0)||out[0]!=want0||out[1]!=want1|| + calls[0]!=1||calls[1]!=1||G.is_cnt[0]||G.is_cnt[1]||G.issue_open){ + fprintf(stderr,"take failed case %d: status=%d output=%g,%g calls=%d,%d\n", + failed,ok,out[0],out[1],calls[0],calls[1]);return 1; + } + } + G.issue_open=1; + if(!qt_take(0,NULL,0,NULL)||G.issue_open) return 1; + G.on=0; + if(!qt_take(0,NULL,0,NULL)||qt_take(1,NULL,0,NULL)) return 1; + pthread_cond_destroy(&G.cv_take);pthread_mutex_destroy(&G.mx); + puts("tier take error: ok");return 0; +} diff --git a/c/tests/test_qwen36_tier_withdraw.c b/c/tests/test_qwen36_tier_withdraw.c new file mode 100644 index 000000000..54dd092b6 --- /dev/null +++ b/c/tests/test_qwen36_tier_withdraw.c @@ -0,0 +1,67 @@ +/* The automatic trunk placement is a prediction the engine measures at startup + * (#1652: on four Tesla M10 every placed component ran slower than the CPU). + * When the probe says the GPU loses, qt_trunk_withdraw() must put every offer + * back on the CPU, lm_head included, and return the bytes the trunk had taken + * to the expert budget, so the warmstart that follows fills that VRAM with + * experts instead. A hand-written COLI_PLACE is the user's word: not auto, not + * withdrawn. Fake CUDA backend, no GPU, no toolkit. */ +#include +#include +#include + +#include "../compat.h" /* setenv: MinGW has none */ +#include "qwen36_fake_cuda.h" + +#include "../qwen36_tier.c" + +static int fails; +static void check(int ok, const char *what) { if (!ok) { printf(" FAIL: %s\n", what); fails++; } } + +enum { NL = 2, NE = 8, D = 64, IH = 32, TOPK = 2 }; + +int main(void) { + setenv("COLI_CUDA", "1", 1); setenv("COLI_GPUS", "0", 1); + setenv("QT_NO_WARMSTART", "1", 1); setenv("HEAT_FILE", "", 1); + setenv("COLI_PLACE", "", 1); /* "" == unset == auto */ + fake_ndev = 1; fake_uploads = 0; + + /* room for 8 experts plus a little; each offer is 1 expert worth of bytes */ + size_t exp_bytes = 3 * dev_alloc_footprint((size_t)D * IH / 2) + 3 * dev_alloc_footprint((2 * IH + D) / 3 * sizeof(float)); + size_t cap = 8 * exp_bytes + exp_bytes / 2; + char gb[64]; snprintf(gb, sizeof gb, "%.15f", (double)cap / 1073741824.0); + setenv("CUDA_EXPERT_GB", gb, 1); + + qt_trunk_offer("lmhead", 0, exp_bytes); + qt_trunk_offer("dnproj", 0, exp_bytes); qt_trunk_offer("dnout", 0, exp_bytes); + qt_trunk_offer("attnproj", 1, exp_bytes); qt_trunk_offer("shexp", 1, exp_bytes); + check(qt_init(NL, NE, D, IH, NE, TOPK, 0, 1), "tier starts (int4 mode, cap == n_experts)"); + check(qt_place_is_auto(), "COLI_PLACE unset: the placement is automatic"); + check(qt_place_of("lmhead", 0) == 0 && qt_place_of("dnproj", 0) == 0 && qt_place_of("dnout", 0) == 0 && + qt_place_of("attnproj", 1) == 0 && qt_place_of("shexp", 1) == 0, "all five offers placed on the one device"); + check(G_lmh.dev_ok, "lm_head has a device before the withdrawal"); + size_t before = G.budget[0]; + check(before + 5 * exp_bytes <= cap && before + 5 * exp_bytes + exp_bytes > cap, + "the expert budget is the allowance minus the five placed offers"); + + qt_trunk_withdraw("test"); + check(qt_place_of("lmhead", 0) == QT_PLACE_CPU && qt_place_of("dnproj", 0) == QT_PLACE_CPU && + qt_place_of("dnout", 0) == QT_PLACE_CPU && qt_place_of("attnproj", 1) == QT_PLACE_CPU && + qt_place_of("shexp", 1) == QT_PLACE_CPU, "after the withdrawal every offer answers CPU"); + check(!G_lmh.dev_ok, "lm_head has no device any more: qt_lmhead_init will refuse"); + check(G.budget[0] == before + 5 * exp_bytes, "the trunk's bytes are back in the expert budget"); + check(G_trunk_bytes[0] == 0, "no trunk bytes are charged to the device"); + check(qt_lmhead_init((const int8_t *)"x", (const float *)"x", 1, 1) == 0, "a later lm_head upload is refused"); + check(fake_uploads == 0, "nothing was uploaded: the withdrawal came before any trunk upload"); + + /* a hand-written list is not automatic: nothing to withdraw, nothing changes */ + setenv("COLI_PLACE", "lmhead=0", 1); + check(!qt_place_is_auto(), "an explicit COLI_PLACE list is not automatic"); + size_t held = G.budget[0]; + qt_trunk_withdraw("test again"); + check(G.budget[0] == held, "withdrawing an explicit placement is a no-op"); + + qt_shutdown(); + if (fails) { printf("test_qwen36_tier_withdraw: %d failure(s)\n", fails); return 1; } + printf("OK test_qwen36_tier_withdraw: the automatic trunk placement can be withdrawn, budget restored\n"); + return 0; +} diff --git a/c/tests/test_qwen36_tokenizer.c b/c/tests/test_qwen36_tokenizer.c new file mode 100644 index 000000000..1eba30322 --- /dev/null +++ b/c/tests/test_qwen36_tokenizer.c @@ -0,0 +1,81 @@ +/* #1653: an added token that directly follows punctuation (`X.<|im_end|>`) must + * still be one token. HF's tokenizers split added tokens out first and run the + * regex pre-tokenizer on the ordinary text between them; encode_text() used to + * look for the special only at the start of each piece, and the punctuation + * rule had already swallowed the `<|`. Pins the split on a tiny tokenizer.json + * so it needs no model: every added token boundary, the whitespace lookahead + * at that boundary, and back-to-back specials. */ +#include +#include +#include + +#define main qwen36_main_unused +#include "../qwen36.c" +#undef main + +static int g_fails = 0; +static void expect(const char *text, const int *want, int nwant){ + int *ids = NULL, n = 0; + encode_text(text, &ids, &n); + int ok = (n == nwant); + for (int i = 0; ok && i < n; i++) ok = (ids[i] == want[i]); + if (!ok) { + g_fails++; + printf("FAIL: %-24s got [", text); + for (int i = 0; i < n; i++) printf("%s%d", i ? " " : "", ids[i]); + printf("] want ["); + for (int i = 0; i < nwant; i++) printf("%s%d", i ? " " : "", want[i]); + printf("]\n"); + } + free(ids); +} +#define EXPECT(text, ...) do { int w[] = {__VA_ARGS__}; expect(text, w, (int)(sizeof w / sizeof w[0])); } while (0) + +int main(void){ + /* Byte-level BPE vocabulary: printable ASCII maps to itself, U+0120 (\xC4\xA0) + * is the space, U+010A (\xC4\x8A) the newline. One merge, the repeated space, + * as in the real Qwen table. Id 7 is the only added token. */ + const char *path = "test_qwen36_tokenizer.json"; + const char *json = + "{\"model\":{\"vocab\":{\"X\":0,\".\":1,\"<\":2,\"|\":3,\"i\":4,\"m\":5,\"Y\":6," + "\"}\":8,\"\\\"\":9,\"\xC4\x8A\":10,\"\xC4\xA0\":11,\"\xC4\xA0\xC4\xA0\":12," + "\"_\":13,\"e\":14,\"n\":15,\"d\":16,\">\":17,\"!\":18,\"\xC4\xA0Y\":19}," + "\"merges\":[\"\xC4\xA0 \xC4\xA0\",\"\xC4\xA0 Y\"]}," + "\"added_tokens\":[{\"id\":7,\"content\":\"<|im_end|>\",\"special\":true}]}"; + FILE *f = fopen(path, "wb"); + if (!f || fwrite(json, 1, strlen(json), f) != strlen(json) || fclose(f)) { printf("FAIL: cannot write %s\n", path); return 1; } + load_tokenizer(path); + remove(path); + + EXPECT("X", 0); + EXPECT("X<|im_end|>", 0, 7); + /* the reporter's cases: punctuation right before the special, +4 tokens before the fix */ + EXPECT("X.", 0, 1); + EXPECT("X.<|im_end|>", 0, 1, 7); + EXPECT("X}<|im_end|>", 0, 8, 7); + EXPECT("X\"<|im_end|>", 0, 9, 7); + EXPECT("X!<|im_end|>", 0, 18, 7); + EXPECT("X.\n<|im_end|>", 0, 1, 10, 7); + /* the special ends the ordinary text: `\s+(?!\S)` must see end-of-input there, + * so two spaces stay one piece (and one merged token), as in HF */ + EXPECT("X <|im_end|>", 0, 11, 7); + EXPECT("X <|im_end|>", 0, 12, 7); + /* the whitespace rules themselves, with and without a special after them: + * `\s+(?!\S)` leaves the last space to the next piece, `\s*[\r\n]+` stops at + * the last newline, a lone space before text is that text's prefix */ + EXPECT(" Y", 11, 19); + EXPECT(" Y<|im_end|>", 11, 19, 7); + EXPECT("X ", 0, 12); + EXPECT("\n Y", 10, 11, 19); + EXPECT("X\n\n<|im_end|>", 0, 10, 10, 7); + /* back to back, and text on both sides */ + EXPECT("<|im_end|><|im_end|>", 7, 7); + EXPECT("X.<|im_end|>Y", 0, 1, 7, 6); + EXPECT("<|im_end|>.X", 7, 1, 0); + /* the special spelled out as ordinary text is NOT the special: it must stay text */ + EXPECT("<|im_en", 2, 3, 4, 5, 13, 14, 15); + + if (g_fails) { printf("test_qwen36_tokenizer: %d failure(s)\n", g_fails); return 1; } + printf("OK test_qwen36_tokenizer: added tokens split before the pre-tokenizer (#1653)\n"); + return 0; +} diff --git a/c/tests/test_qwen36_trunk_dense.c b/c/tests/test_qwen36_trunk_dense.c new file mode 100644 index 000000000..0640990a5 --- /dev/null +++ b/c/tests/test_qwen36_trunk_dense.c @@ -0,0 +1,256 @@ +/* The rest of the dense trunk offered to the VRAM placer: "dnout" (DeltaNet + * out_proj), "attnproj" (q, k, v, o of an attention layer) and "shexp" (the + * shared expert's gate, up, down), placed by name and layer and served from + * VRAM through qt_dense handles kept in the Layer. + * + * What is pinned, on the fake CUDA backend (no GPU, no toolkit): + * - trunk_offer_dense offers exactly the components whose every matrix has + * a dense-i8 copy, with the bytes of those copies; + * - trunk_place_dense uploads what the placer took and keeps one handle per + * matrix, counting matrices and bytes; + * - a placed matrix answers the same GEMV from VRAM as matmul_d does on the + * CPU (same int8 rows, same per-row scales), and a component without a + * handle keeps running matmul_d; + * - the placer prices every one of them as a dense component: with room for + * all, all go; with COLI_PLACE=off nothing is offered at all. + * Include order as in tests/test_qwen36_tier_int8_engine.c: the engine, the + * fake backend, then the tier, so the tier's statics live in this TU. */ +#define main qwen36_main_unused +#include "../qwen36.c" +#undef main + +#include "../compat.h" /* setenv/unsetenv: MinGW has neither */ + +#include "qwen36_fake_cuda.h" + +#include "../qwen36_tier.c" + +static int fails; +static void ck(int ok, const char *what) { + if (ok) { printf(" ok %s\n", what); return; } + printf(" FAIL %s\n", what); + fails++; +} + +/* Small but every dimension distinct, so a swapped I/O would show. */ +enum { NL = 2, D = 48, VH = 2, VD = 8, QH = 2, QD = 16, KVH = 1, KD = 8, SH = 20, NE = 4, IH = 16 }; + +static unsigned g_seed = 12345; +static float rnd(void) { + g_seed = g_seed * 1103515245u + 12345u; + return ((g_seed >> 8) & 0xFFFF) / 32768.f - 1.f; +} +static float *rnd_matrix(int rows, int cols) { + float *w = malloc((size_t)rows * cols * sizeof(float)); + for (size_t i = 0; i < (size_t)rows * cols; i++) w[i] = rnd(); + return w; +} + +/* One GEMV both ways: from VRAM through the handle and on the CPU through + * matmul_d. The fake backend computes exactly what a real one does with these + * bytes (x . int8 row, times the row scale), so the two must agree to float + * accumulation order. */ +static double gemv_gap(int hp1, const QW *w, int I, int O, int *served) { + float *x = rnd_matrix(1, I), *ya = calloc((size_t)O, sizeof(float)), *yb = calloc((size_t)O, sizeof(float)); + *served = qtd(hp1, ya, x, I, O); + matmul_d(yb, x, w, 1, I, O); + double worst = 0, scale = 1e-6; + for (int o = 0; o < O; o++) { + double d = fabs((double)ya[o] - yb[o]); + if (d > worst) worst = d; + if (fabs((double)yb[o]) > scale) scale = fabs((double)yb[o]); + } + free(x); free(ya); free(yb); + return worst / scale; +} + +/* Quantize a freshly-built random matrix straight into *out (the QW the + * Layer holds), then discard the f32 staging buffer -- the same "quantize + * once, free the f32 copy" shape load_tq uses. */ +static void register_qw(int rows, int cols, QW *out) { + float *w = rnd_matrix(rows, cols); + qw_quantize(w, cols, rows, NULL, out); + free(w); +} + +static void build(Model *m) { + memset(m, 0, sizeof *m); + Cfg *c = &m->c; + c->n_layers = NL; c->hidden = D; c->n_experts = NE; c->inter = IH; c->topk = 1; c->expert_gs = 0; + c->q_heads = QH; c->q_head_dim = QD; c->kv_heads = KVH; c->k_head_dim = KD; c->head_dim = KD; c->o_in = QH * KD; + c->dn_vheads = VH; c->dn_vdim = VD; c->shared_inter = SH; + c->is_attn = calloc(NL, 1); c->is_attn[1] = 1; + m->L = calloc(NL, sizeof(Layer)); + Layer *dn = &m->L[0], *at = &m->L[1]; + register_qw(D, VH * VD, &dn->dn_out); + register_qw(SH, D, &dn->sh_g); + register_qw(SH, D, &dn->sh_u); + register_qw(D, SH, &dn->sh_d); + register_qw(QH * QD, D, &at->q); + register_qw(KVH * KD, D, &at->k); + register_qw(KVH * KD, D, &at->v); + register_qw(D, QH * KD, &at->o); + /* the attention layer's shared expert is INCOMPLETE on purpose: gate and + * up have a dense-i8 copy, down does not (never registered), so "shexp" + * for layer 1 must not be offered and its three handles must stay 0 */ + register_qw(SH, D, &at->sh_g); + register_qw(SH, D, &at->sh_u); + /* at->sh_d: left zeroed (q == NULL) -- no dense-i8 copy, never registered */ +} + +int main(void) { + setenv("COLI_CUDA", "1", 1); setenv("COLI_GPUS", "0", 1); + setenv("QT_NO_WARMSTART", "1", 1); setenv("HEAT_FILE", "", 1); + setenv("COLI_PLACE", "", 1); /* "" == unset == auto */ + setenv("CUDA_EXPERT_GB", "1", 1); /* room for everything */ + unsetenv("COLI_DENSE_I8"); + /* the GEMV from VRAM is compared with the CPU's f32-activation kernel, the + * contract the tier uploads: the integer dense path rounds the activation + * and is measured elsewhere (test_qwen36_dense_idot) */ + setenv("COLI_DENSE_IDOT", "0", 1); + fake_ndev = 1; fake_uploads = 0; fake_dense_compute = 1; + + Model m; build(&m); + Layer *dn = &m.L[0], *at = &m.L[1]; + + printf("offers\n"); + int before = G_offer_n; + trunk_offer_dense(&m); + /* dnout(layer 0) + attnproj(layer 1) + shexp(layer 0); NOT shexp(layer 1) */ + ck(G_offer_n - before == 3, "three components offered: dnout, attnproj, shexp of the complete layer only"); + size_t want_attn = qdw_bytes(&at->q) + qdw_bytes(&at->k) + qdw_bytes(&at->v) + qdw_bytes(&at->o); + int seen_attn = 0, seen_dnout = 0, seen_shexp1 = 0; + for (int o = before; o < G_offer_n; o++) { + if (!strcmp(G_offer[o].name, "attnproj") && G_offer[o].layer == 1 && G_offer[o].bytes == want_attn) seen_attn = 1; + if (!strcmp(G_offer[o].name, "dnout") && G_offer[o].layer == 0 && G_offer[o].bytes == qdw_bytes(&dn->dn_out)) seen_dnout = 1; + if (!strcmp(G_offer[o].name, "shexp") && G_offer[o].layer == 1) seen_shexp1 = 1; + } + ck(seen_dnout, "dnout offered for the DeltaNet layer with the bytes of its dense-i8 copy"); + ck(seen_attn, "attnproj offered for the attention layer as the sum of q, k, v, o"); + ck(!seen_shexp1, "a shared expert missing one dense-i8 copy is not offered"); + ck(qdw_bytes(&at->sh_d) == 0, "a matrix without a dense-i8 copy reports zero bytes"); + + printf("placement\n"); + ck(qt_init(NL, NE, D, IH, NE, 1, 0, 1), "tier starts (int4 mode, cap == n_experts)"); + ck(qt_place_of("dnout", 0) == 0 && qt_place_of("attnproj", 1) == 0 && qt_place_of("shexp", 0) == 0, + "every offered component placed on the one device with room"); + ck(qt_place_of("shexp", 1) == QT_PLACE_CPU, "the component never offered stays on the CPU"); + double vram = 0; + int placed = trunk_place_dense(&m, &vram); + ck(placed == 8, "eight matrices placed: dnout, q k v o, and the complete shared expert"); + ck(vram == (double)(qdw_bytes(&dn->dn_out) + want_attn + qdw_bytes(&dn->sh_g) + qdw_bytes(&dn->sh_u) + qdw_bytes(&dn->sh_d)), + "placed bytes are the sum of the dense-i8 copies uploaded"); + ck(fake_uploads == 8, "one upload per placed matrix"); + ck(dn->qth_dnout > 0 && at->qth_q > 0 && at->qth_k > 0 && at->qth_v > 0 && at->qth_o > 0, "handles kept in the Layer"); + ck(dn->qth_shg > 0 && dn->qth_shu > 0 && dn->qth_shd > 0, "shared expert handles kept in the DeltaNet layer"); + ck(at->qth_shg == 0 && at->qth_shu == 0 && at->qth_shd == 0, "no handle for the shared expert that was not offered"); + + printf("GEMV from VRAM == matmul_d on the CPU\n"); + struct { const char *name; int hp1; const QW *w; int I, O; } mats[] = { + {"dn_out", dn->qth_dnout, &dn->dn_out, VH * VD, D}, + {"q", at->qth_q, &at->q, D, QH * QD}, {"k", at->qth_k, &at->k, D, KVH * KD}, + {"v", at->qth_v, &at->v, D, KVH * KD}, {"o", at->qth_o, &at->o, QH * KD, D}, + {"sh_g", dn->qth_shg, &dn->sh_g, D, SH}, {"sh_u", dn->qth_shu, &dn->sh_u, D, SH}, {"sh_d", dn->qth_shd, &dn->sh_d, SH, D}, + }; + for (size_t i = 0; i < sizeof mats / sizeof mats[0]; i++) { + int served = 0; + double gap = gemv_gap(mats[i].hp1, mats[i].w, mats[i].I, mats[i].O, &served); + char what[128]; + snprintf(what, sizeof what, "%s: served from VRAM, relative gap %.2e", mats[i].name, gap); + ck(served && gap < 1e-4, what); + } + { + int served = 1; float x[D] = {0}, y[SH] = {0}; + served = qtd(at->qth_shd, y, x, SH, D); + ck(!served, "a matrix with no handle is not served (the caller runs matmul_d)"); + } + + /* Drive attention itself: this catches an S==1 gate left at any of the + * four projection call sites, which a tier-only test cannot detect. */ + { + enum { S = 5 }; +#ifdef _OPENMP + omp_set_num_threads(1); +#endif + m.max_t = m.kv_cap = S + 1; + m.K = calloc(NL, sizeof(float *)); m.V = calloc(NL, sizeof(float *)); + m.K[1] = calloc((S + 1) * KVH * KD, sizeof(float)); + m.V[1] = calloc((S + 1) * KVH * KD, sizeof(float)); + m.attn_sc = calloc(S + 1, sizeof(float)); + float *x = rnd_matrix(S + 1, D), gpu[S * D], cpu[S * D], fallback[S * D]; + Layer host = *at; + host.qth_q = host.qth_k = host.qth_v = host.qth_o = 0; + attention(&m, &host, 1, x, S, 0, cpu); + int calls = fake_matmuls; + attention(&m, at, 1, x, S, 0, gpu); + ck(fake_matmuls == calls + 4 && fake_matmul_rows == S, + "prefill dispatches all four projections as batches"); + int handles[]={at->qth_q,at->qth_k,at->qth_v,at->qth_o}; + for(int failure=0;failure<4;failure++){ + for(int i=0;i<4;i++) G_dense[handles[i]-1].on=1; + calls=fake_matmuls; + fake_matmul_fail_at=calls+failure+1; + attention(&m,at,1,x,S,0,fallback); + fake_matmul_fail_at=0; + ck(fake_matmuls==calls+4,"each projection attempted once on first failure"); + for(int i=0;i<4;i++) + ck(G_dense[handles[i]-1].on==(i!=failure),"only failed projection disabled"); + for(int pass=0;pass<2;pass++){ + double gap=0, scale=1e-6; + int finite=1; + for(int i=0;i=0) fake_matmul_fail_at=fake_matmuls+failure+1; + attention(&m,at,1,x+2*D,S-2,2,continued+2*D); + fake_matmul_fail_at=0; + attention(&m,at,1,x+S*D,1,S,continued+S*D); + ck(fake_matmuls==calls+(failure<0?12:11), + "segmented prefill and decode keep only the failed projection on CPU"); + double gap=0, scale=1e-6; + int finite=1; + for(int i=0;i<(S+1)*D;i++){ + finite &= isfinite(continued[i]) && isfinite(reference[i]); + gap=fmax(gap,fabs((double)continued[i]-reference[i])); + scale=fmax(scale,fabs(reference[i])); + } + ck(finite && gap/scale<1e-4,"segmented attention outputs match full CPU prefill"); + gap=0; scale=1e-6; finite=1; + for(int i=0;i<(S+1)*KVH*KD;i++){ + finite &= isfinite(m.K[1][i]) && isfinite(m.V[1][i]); + gap=fmax(gap,fabs((double)m.K[1][i]-keys[i])); + gap=fmax(gap,fabs((double)m.V[1][i]-values[i])); + scale=fmax(scale,fmax(fabs(keys[i]),fabs(values[i]))); + } + ck(finite && gap/scale<1e-4,"segmented attention preserves CPU-equivalent KV state"); + } + free(x); free(m.K[1]); free(m.V[1]); free(m.K); free(m.V); free(m.attn_sc); + } + + qt_shutdown(); + if (fails) { printf("test_qwen36_trunk_dense: %d failure(s)\n", fails); return 1; } + printf("OK test_qwen36_trunk_dense: dnout, attnproj and shexp offered, placed and served from VRAM\n"); + return 0; +} diff --git a/c/tests/test_qwen38_brio.py b/c/tests/test_qwen38_brio.py new file mode 100644 index 000000000..d7f1468e1 --- /dev/null +++ b/c/tests/test_qwen38_brio.py @@ -0,0 +1,149 @@ +"""Pinned Qwen3.8 scores must survive unrelated requests and sibling branches. + +Uses the generated tiny fixture and its 64-token byte tokenizer. All prompt +bytes are below 64; no unknown-token aliases can hide a divergent prefix. +QWEN38_TINY selects BF16 or FP8 fixtures. No full checkpoint is needed. +""" +import math +import os +from pathlib import Path +import queue +import subprocess +import sys +import threading +import time +import unittest + +HERE = Path(__file__).resolve().parent.parent +ENGINE = Path(os.environ.get('QWEN38_BINARY', HERE / ('qwen38.exe' if sys.platform == 'win32' else 'qwen38'))) +FIXTURE = Path(os.environ.get('QWEN38_TINY', HERE / 'qwen38_tiny')) + + +class Engine: + def __init__(self): + env = dict(os.environ, SNAP=str(FIXTURE), SERVE='1', + TOK=str(FIXTURE / 'tokenizer.json'), OMP_NUM_THREADS='2', + COLI_NO_OMP_TUNE='1', COLI_CUDA='0', Q38_TRUNK_GPU='0', + Q38_MAXT='128') + self.p = subprocess.Popen([str(ENGINE), '1', '8'], env=env, + stdin=subprocess.PIPE, stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL) + self.frames = queue.Queue() + def read(): + try: + while True: + raw = self.p.stdout.readline() + if not raw: + raise EOFError('engine closed') + fields = raw.decode().replace('\x01', '').split() + if fields and fields[0] in ('ECHO', 'DATA'): + n = int(fields[2]) + if len(self.p.stdout.read(n)) != n or self.p.stdout.read(1) != b'\n': + raise EOFError('short protocol payload') + self.frames.put(fields) + except Exception as exc: + self.frames.put(exc) + self.reader = threading.Thread(target=read, daemon=True) + self.reader.start() + try: + deadline = time.monotonic() + 30 + while self.next(deadline)[0] != 'READY': + pass + except BaseException: + self.close() + raise + + def next(self, deadline): + frame = self.frames.get(timeout=max(.01, deadline - time.monotonic())) + if isinstance(frame, Exception): + raise frame + return frame + + def submit(self, rid, text, count=0, extension=' logprobs=1'): + data = text.encode('ascii') + assert all(b < 64 for b in data) + self.p.stdin.write(f'SUBMIT {rid} 0 {len(data)} {count} 0 1{extension}\n'.encode() + + data + b'\n') + self.p.stdin.flush() + scores = {} + accepted = False + deadline = time.monotonic() + 30 + while True: + f = self.next(deadline) + if f[0] in ('ACCEPT', 'ECHO', 'DATA', 'DONE', 'ERROR'): + assert f[1] == str(rid), f + if f[0] == 'ACCEPT': + assert int(f[2]) == len(data), f + accepted = True + elif f[0] == 'ECHO': + score = float(f[4]) + assert math.isfinite(score), f + scores[int(f[3])] = score + elif f[0] == 'ERROR': + raise AssertionError(f) + elif f[0] == 'DONE': + assert accepted + return scores + + def close(self): + self.p.terminate() + try: + self.p.wait(timeout=5) + except subprocess.TimeoutExpired: + self.p.kill() + self.p.wait(timeout=5) + self.reader.join(timeout=5) + self.p.stdin.close() + self.p.stdout.close() + + +@unittest.skipUnless(ENGINE.is_file() and (FIXTURE / 'tokenizer.json').is_file(), + 'build qwen38 and generate QWEN38_TINY with a byte tokenizer') +class Qwen38Brio(unittest.TestCase): + def engine(self): + e = Engine() + self.addCleanup(e.close) + return e + + def test_stale_pin_after_unrelated_generation(self): + prefix, option = '123 456:', ' 0' + for repetition in range(2): + with self.subTest(repetition=repetition): + e = self.engine() + e.submit(1, prefix, extension=' logprobs=1 pin=1') + warm = e.submit(2, prefix + option) + e.submit(3, '789 012:', count=3, extension='') + after = e.submit(4, prefix + option) + fresh = self.engine().submit(1, prefix + option) + tail = {8, 9} + self.assertEqual(set(warm), tail, 'valid pins must still skip the prefix') + self.assertTrue(tail <= set(after)) + self.assertEqual({p: warm[p] for p in tail}, {p: fresh[p] for p in tail}) + self.assertEqual({p: after[p] for p in tail}, {p: fresh[p] for p in tail}) + + def test_shorter_branch_drops_old_tail(self): + e = self.engine() + e.submit(1, '123 456:', extension=' logprobs=1 pin=1') + e.submit(2, '123 456: 01', extension=' logprobs=1 pin=1') + e.submit(3, '123 456: ') + after = e.submit(4, '123 456: 012') + fresh = self.engine().submit(1, '123 456: 012') + self.assertEqual(set(after), set(range(8, 12))) + self.assertEqual(after, {p: fresh[p] for p in after}) + + def test_nested_pins_and_sibling_branches(self): + e = self.engine() + e.submit(1, '123', extension=' logprobs=1 pin=1') + e.submit(2, '123 456:', extension=' logprobs=1 pin=1') + e.submit(3, '123 456: 0') + # Same shallow pin, a different question overwrites the deeper pin's KV. + sibling = e.submit(4, '123 789: 1') + self.assertEqual(set(sibling), set(range(3, 10))) + returned = e.submit(5, '123 456: 0') + fresh = self.engine().submit(1, '123 456: 0') + self.assertEqual(set(returned), set(range(3, 10))) + self.assertEqual(returned, {p: fresh[p] for p in returned}) + + +if __name__ == '__main__': + unittest.main() diff --git a/c/tests/test_qwen38_chat_template.py b/c/tests/test_qwen38_chat_template.py index 73dd631b8..97ebc8933 100644 --- a/c/tests/test_qwen38_chat_template.py +++ b/c/tests/test_qwen38_chat_template.py @@ -22,7 +22,14 @@ Il template vero viene reso con jinja2 e confrontato byte per byte con quello che produce il gateway. Se manca il template il test si dichiara SALTATO invece di passare: un test che non ha trovato il suo riferimento non ha verificato -niente, e dirlo verde sarebbe peggio che non averlo. +niente, e dirlo verde sarebbe peggio che non averlo -- percio' un salto esce con +codice 2, distinto dallo 0 di un confronto riuscito. + +RIFERIMENTO (scaricato 2026-09-10): + repo Qwen/Qwen3.8-Flash-Next-FP8 + file chat_template.jinja + sha256 c3cf9e34abf4f9e36c2d72165aa9c132d3e2a725b6c2586aaa3a8af9d7a81041 + hf download Qwen/Qwen3.8-Flash-Next-FP8 chat_template.jinja USO: python3 tests/test_qwen38_chat_template.py --template PATH/chat_template.jinja @@ -115,7 +122,8 @@ EFFORTS = ("xhigh", "medium", "low") -def reference(template_text, *, messages, tools=None, reasoning_effort=None): +def reference(template_text, *, messages, tools=None, reasoning_effort=None, + add_generation_prompt=True): import jinja2 def raise_exception(message): @@ -127,7 +135,7 @@ def raise_exception(message): lambda value, ensure_ascii=False, **kw: json.dumps(value, ensure_ascii=ensure_ascii)) environment.globals["raise_exception"] = raise_exception rendered = environment.from_string(template_text) - arguments = {"messages": messages, "add_generation_prompt": True, + arguments = {"messages": messages, "add_generation_prompt": add_generation_prompt, "enable_thinking": True} if tools: arguments["tools"] = tools @@ -157,12 +165,12 @@ def main() -> int: if not arguments.template.exists(): print(f"SKIP: manca {arguments.template}; il riferimento non c'e' e " f"questo test non ha verificato nulla") - return 0 + return 2 # salto != successo (vedi docstring) try: import jinja2 # noqa: F401 except ImportError: print("SKIP: jinja2 non installato; senza non c'e' riferimento") - return 0 + return 2 # salto != successo (vedi docstring) sys.path.insert(0, str(Path(__file__).resolve().parents[1])) import openai_server @@ -185,6 +193,49 @@ def main() -> int: show(name, ours, theirs) failures += 1 + # Prosecuzione: l'ultimo turno assistant e' da CONTINUARE, non uno gia' finito. + # ChatML chiude ogni turno con <|im_end|>, e questo template non ha un ramo di + # continuazione: il suo add_generation_prompt=False toglie solo la cue, il turno resta + # chiuso. La forma aperta e' percio' quel rendering MENO il <|im_end|>\n finale -- la + # posizione in cui il modello si trova mentre scrive un turno, e da cui prosegue. + aperto = [{"role": "user", "content": "capitale della Francia?"}, + {"role": "assistant", "content": "La capitale e'"}] + produced = openai_server.render_chat_for_arch(aperto, enable_thinking=True, + reasoning_effort="low", + add_generation_prompt=False) + closed = reference(template_text, messages=aperto, reasoning_effort="low", + add_generation_prompt=False) + # Il terminatore da togliere DEVE esserci nel riferimento: se il template cambiasse + # convenzione, toglierlo "se c'e'" sarebbe un no-op silenzioso e il confronto perderebbe + # senso. Percio' lo esigiamo prima di tagliarlo. + TERM = "<|im_end|>\n" + if not closed.endswith(TERM): + print(f"FAIL prosecuzione: il riferimento non finisce col terminatore {TERM!r} da " + f"togliere -- convenzione del template cambiata? coda: {closed[-40:]!r}") + failures += 1 + expected = closed + else: + expected = closed[:-len(TERM)] + if produced == expected: + print("ok prosecuzione: turno aperto = template(add_generation_prompt=False) " + "senza il <|im_end|> finale") + else: + show("prosecuzione", produced, expected) + failures += 1 + if not produced.endswith("La capitale e'"): + print(f"FAIL prosecuzione: il prompt non finisce sull'apertura del client: " + f"{produced[-60:]!r}") + failures += 1 + if produced.rstrip("\n").endswith("<|im_end|>"): + print("FAIL prosecuzione: il turno resta chiuso col terminatore") + failures += 1 + # Controllo negativo: col ramo normale lo stesso scambio DEVE finire sulla cue, o il + # confronto qui sopra non starebbe distinguendo niente. + if not openai_server.render_chat_for_arch( + aperto, enable_thinking=True, reasoning_effort="low").endswith("\n"): + print("FAIL prosecuzione: il ramo normale non emette piu' il prompt di generazione") + failures += 1 + # Il giro completo: rendere una chiamata e rileggerla deve restituire quello # che ci era stato dato. E' la meta' che il confronto col template non copre, # perche' il template sa solo scrivere. diff --git a/c/tests/test_qwen38_idot.c b/c/tests/test_qwen38_idot.c new file mode 100644 index 000000000..3a683b63c --- /dev/null +++ b/c/tests/test_qwen38_idot.c @@ -0,0 +1,94 @@ +/* qwen38's trunk on the CPU as int8 rows met by an int8 activation in the + * integer kernel (idot.h), and the vector FP8 kernel for the routed experts. + * + * Pinned, with no model: + * - the int8 trunk is the default; Q38_TRUNK_CPU_INT8=0 is the only way off; + * - a BF16 matrix quantized as the loader does answers q38_weight_matmul from + * its int8 rows within what an int8 row and an int8 activation allow, for one + * row (decode) and for a batch (prefill), and keeps answering after the BF16 + * copy is released, with the same bytes; + * - the vector FP8 kernel agrees with quant.h's table kernel to float + * summation order on a matrix with a partial last block and an odd number of + * rows, for one row and for a batch, and Q38_FP8_KERNEL=scalar routes the + * dispatch to the table kernel byte for byte. */ +#define _GNU_SOURCE +#define QWEN38_NO_MAIN +#include "../qwen38.c" +#include "../compat.h" /* setenv/unsetenv: MinGW has neither */ + +static int fails; +static void ck(int ok,const char *what){ if(ok){printf(" ok %s\n",what);return;} printf(" FAIL %s\n",what); fails++; } + +static unsigned g_seed=777; +static float rnd(void){ g_seed=g_seed*1103515245u+12345u; return ((g_seed>>8)&0xFFFF)/32768.f-1.f; } +static double rel_gap(const float *a,const float *b,int n){ + double worst=0,scale=1e-6; + for(int i=0;iworst)worst=d; if(fabs((double)b[i])>scale)scale=fabs((double)b[i]); } + return worst/scale; +} +static uint16_t f32_to_bf16(float f){ uint32_t u; memcpy(&u,&f,4); return (uint16_t)(u>>16); } + +int main(void){ + printf("the flag\n"); + unsetenv("Q38_TRUNK_CPU_INT8"); + ck(q38_trunk_cpu_int8_wanted()==1,"int8 trunk on by default"); + setenv("Q38_TRUNK_CPU_INT8","0",1); + ck(q38_trunk_cpu_int8_wanted()==0,"Q38_TRUNK_CPU_INT8=0 keeps BF16"); + setenv("Q38_TRUNK_CPU_INT8","1",1); + ck(q38_trunk_cpu_int8_wanted()==1,"=1 still means on"); + unsetenv("Q38_TRUNK_CPU_INT8"); + + printf("the int8 trunk\n"); + { + enum { I=320, O=96, S=5 }; + Q38Weight w={0}; q38_weight_reserve(&w,Q38_WEIGHT_BF16,O,I); + uint16_t *wb=(uint16_t*)w.data; + for(int64_t k=0;k<(int64_t)O*I;k++)wb[k]=f32_to_bf16(rnd()); + float *x=malloc((size_t)S*I*sizeof(float)),*ref=malloc((size_t)S*O*sizeof(float)); + float *y=malloc((size_t)S*O*sizeof(float)),*y2=malloc((size_t)S*O*sizeof(float)); + for(int64_t k=0;k<(int64_t)S*I;k++)x[k]=rnd()*2.f; + q38_matmul_bf16(ref,x,wb,S,I,O); + q38_weight_matmul(y,x,&w,S,I,O); + ck(!memcmp(y,ref,(size_t)S*O*sizeof(float)),"without int8 rows the BF16 kernel answers, byte for byte"); + q38_trunk_quantize(&w,&w.q8,&w.q8sc); + q38_weight_matmul(y,x,&w,S,I,O); + ck(rel_gap(y,ref,S*O)<3e-2,"int8 rows x int8 activation within 3% of the BF16 reference, 5 rows"); + q38_weight_matmul(y2,x,&w,1,I,O); + ck(rel_gap(y2,ref,O)<3e-2,"same for one row"); + /* the loader releases the BF16 copy once the rows exist */ + free(w.data); w.data=NULL; w.owns_data=0; + q38_weight_matmul(y2,x,&w,S,I,O); + ck(!memcmp(y,y2,(size_t)S*O*sizeof(float)),"after the BF16 copy is released the int8 rows answer with the same bytes"); + free(w.q8); free(w.q8sc); free(x); free(ref); free(y); free(y2); + } + + printf("the vector FP8 kernel\n"); + { + enum { I=200, O=7, S=5 }; /* a partial last block, an odd row count */ + int nbo=(O+127)/128, nbi=(I+127)/128; + uint8_t *q=malloc((size_t)O*I); float *sc=malloc((size_t)nbo*nbi*sizeof(float)); + for(int64_t k=0;k<(int64_t)O*I;k++){ unsigned b=(unsigned)((rnd()+1.f)*127.99f); if((b&0x7F)==0x7F)b&=0x7E; q[k]=(uint8_t)b; } + for(int k=0;k #include "../qwen38.c" +#include "../compat.h" /* setenv: MinGW has none */ #define CHECK(x) do { if(!(x)){ \ fprintf(stderr,"%s:%d: check failed: %s\n",__FILE__,__LINE__,#x);return 1; \ @@ -383,7 +384,9 @@ static int check_parallel_batch(const char *directory){ int result=-1;model.expert_parallel_reads=1; int first_ids[2]={1,0};Slot *first[2]={0}; if(q38_expert_get_batch(&model,0,first_ids,2,first)||model.cache[0].n|| - model.miss||model.hits)goto cleanup; + model.miss||model.hits|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY) + goto cleanup; if(q38_segment_cache_resize(&model,2))goto cleanup; if(!q38_expert_get_batch(&model,0,first_ids,2,first)||!first[0]||!first[1]|| first[0]==first[1]||first[0]->eid!=1||first[1]->eid!=0|| @@ -401,6 +404,74 @@ static int check_parallel_batch(const char *directory){ destroy_fixture_model(&model);return result; } +static int check_parallel_fallback_reasons(const char *directory){ + int ids[Q38_MAX_TOPK+1]={0};Slot *selected[Q38_MAX_TOPK+1]={0}; + Model model; + + if(init_fixture_model(&model,directory,1))return -1; + model.expert_parallel_reads=0; + if(q38_expert_get_batch(&model,0,ids,2,selected)|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_DISABLED){ + destroy_fixture_model(&model);return -1; + } + model.expert_parallel_reads=1; + if(q38_expert_get_batch(&model,0,ids,2,selected)|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_DISABLED){ + destroy_fixture_model(&model);return -1; + } + destroy_fixture_model(&model); + + if(init_fixture_model(&model,directory,1))return -1; + model.expert_parallel_reads=1; + /* a route wider than the per-layer cache (the #1686 shape: no top-k ceiling) */ + if(q38_expert_get_batch(&model,0,ids,2,selected)|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY){ + destroy_fixture_model(&model);return -1; + } + destroy_fixture_model(&model); + + if(init_fixture_model(&model,directory,0)||q38_segment_cache_resize(&model,2)){ + destroy_fixture_model(&model);return -1; + } + model.expert_parallel_reads=1; + if(q38_expert_get_batch(&model,0,ids,2,selected)|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK){ + destroy_fixture_model(&model);return -1; + } + destroy_fixture_model(&model); + + if(init_fixture_model(&model,directory,1)||q38_segment_cache_resize(&model,2)){ + destroy_fixture_model(&model);return -1; + } + model.expert_parallel_reads=1; + if(q38_expert_get_batch(&model,0,ids,2,selected)|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_DUPLICATE){ + destroy_fixture_model(&model);return -1; + } + destroy_fixture_model(&model); + + if(init_fixture_model(&model,directory,1)||q38_segment_cache_resize(&model,2)){ + destroy_fixture_model(&model);return -1; + } + model.expert_parallel_reads=1; + if(!q38_prepare_expert_scale_bank(&model,0)){ + destroy_fixture_model(&model);return -1; + } + char name[320]; + q38_name(&model,name,sizeof name,0,"mlp.experts.1.gate_proj.weight"); + st_tensor *weight=st_find(&model.S,name); + if(!weight){destroy_fixture_model(&model);return -1;} + int original_dtype=weight->dtype;weight->dtype=0; + int layout_result=q38_expert_get_batch(&model,0,(int[]){0,1},2,selected); + weight->dtype=original_dtype; + if(layout_result|| + model.expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_LAYOUT){ + destroy_fixture_model(&model);return -1; + } + destroy_fixture_model(&model); + return 0; +} + static int check_malformed_scale_metadata(const char *directory){ Model model;if(init_fixture_model(&model,directory,1))return -1; char name[320]; @@ -573,6 +644,9 @@ static int check_segment_failure_outputs(void){ } int main(void){ + /* this file pins storage and dispatch against the table kernel byte for + * byte; the vector FP8 kernel has its own tolerance test (test_qwen38_idot) */ + setenv("Q38_FP8_KERNEL","scalar",1); enum { S=2, I=257, O=129 }; Q38Weight fp8={0};q38_weight_reserve(&fp8,Q38_WEIGHT_FP8,O,I); CHECK(fp8.scale_count==6);CHECK(q38_weight_bytes(&fp8)==(uint64_t)O*I+6*sizeof(float)); @@ -635,6 +709,7 @@ int main(void){ CHECK(check_fixture_mode(adjacent_directory,1,1)==0); CHECK(check_fixture_mode(adjacent_directory,0,0)==0); CHECK(check_parallel_batch(adjacent_directory)==0); + CHECK(check_parallel_fallback_reasons(adjacent_directory)==0); CHECK(check_malformed_scale_metadata(adjacent_directory)==0); CHECK(check_moe_prefill_parity(adjacent_directory,1,1)==0); CHECK(check_moe_prefill_parity(adjacent_directory,1,2)==0); diff --git a/c/tests/test_qwen38_prefix.c b/c/tests/test_qwen38_prefix.c index 6baf99270..7d13068ec 100644 --- a/c/tests/test_qwen38_prefix.c +++ b/c/tests/test_qwen38_prefix.c @@ -30,6 +30,7 @@ static int check_state(const Model *m,float value){ static void free_fake(Model *m){ q38_prefix_cache_release(m); + kv_prefix_free(&m->kvp); for(int i=0;ic.layers;i++){ free(m->DN_rec[i]);free(m->DN_conv[i]); free(m->K[i]);free(m->V[i]);free(m->IK[i]); @@ -61,6 +62,7 @@ int main(void){ const int first[]={1,2,3};float first_logits[8]; for(int i=0;i<8;i++)first_logits[i]=(float)(100+i); + kv_prefix_record(&m.kvp,first,0,3); fill_state(&m,7.f);CHECK(q38_prefix_cache_save(&m,first,3,first_logits,0)); fill_state(&m,99.f);m.kv_len=3; const int extension[]={1,2,3,4}; @@ -73,6 +75,7 @@ int main(void){ /* An exact prompt restores the recurrent state and its saved logits. */ fill_state(&m,55.f);CHECK(q38_prefix_restore(&m,first,3)==3);CHECK(check_state(&m,7.f)); fill_state(&m,56.f);CHECK(q38_prefix_restore(&m,extension,4)==3);CHECK(check_state(&m,7.f)); + kv_prefix_record(&m.kvp,extension,0,4); CHECK(q38_prefix_cache_save(&m,extension,4,first_logits,0)); fill_state(&m,88.f);CHECK(q38_prefix_restore(&m,extension,4)==4);CHECK(check_state(&m,7.f)); const float *cached=q38_prefix_cached_logits(&m);CHECK(cached&&cached[0]==100.f&&cached[7]==107.f); @@ -84,11 +87,13 @@ int main(void){ CHECK(q38_prefix_cache_save(&m,first,3,first_logits,1)); CHECK(g_q38_prefix.pinned&&g_q38_prefix.len==3); CHECK(q38_prefix_restore(&m,extension,4)==3); + kv_prefix_record(&m.kvp,extension,0,4); CHECK(q38_prefix_cache_save(&m,extension,4,first_logits,0)); CHECK(!g_q38_prefix.pinned&&g_q38_prefix.len==4); CHECK(q38_prefix_cache_save(&m,extension,4,first_logits,1)); q38_prefix_cache_invalidate(); CHECK(!g_q38_prefix.pinned&&!g_q38_prefix.valid); + kv_prefix_record(&m.kvp,extension,0,4); CHECK(q38_prefix_cache_save(&m,extension,4,first_logits,0)); /* A mismatch is never restored; the serve miss path invalidates then @@ -101,6 +106,14 @@ int main(void){ for(size_t i=0;i #include #include -#include +#if !defined(__HIPCC__) +#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */ +#endif /* pull in the kernel definitions (same idiom as the CPU tests' #include "../colibri.c") */ #include "../backend_cuda.cu" diff --git a/c/tests/test_serve_budget.c b/c/tests/test_serve_budget.c new file mode 100644 index 000000000..5c6ce5116 --- /dev/null +++ b/c/tests/test_serve_budget.c @@ -0,0 +1,29 @@ +/* max_tokens is a ceiling. The remaining serve engines (kimi, inkling, olmoe) + * used to refuse when prompt + budget exceeded the context, so coli chat's + * interactive default (16384) 400'd every Kimi/Inkling turn at the default + * 8192-token window. Only a prompt that does not fit is refused. */ +#include "../serve_budget.h" +#include +#include + +#define CHECK(c) do { if (!(c)) { \ + fprintf(stderr, "%s:%d: %s\n", __FILE__, __LINE__, #c); return 1; \ +} } while (0) + +int main(void) { + CHECK(coli_serve_budget(2, 16384, 8192, 0) == 8190); + CHECK(coli_serve_budget(2, 8192, 8192, 0) == 8190); + CHECK(coli_serve_budget(100, 50, 8192, 0) == 50); + CHECK(coli_serve_budget(8191, 1, 8192, 0) == 1); + CHECK(coli_serve_budget(8192, 1, 8192, 0) == -1); + CHECK(coli_serve_budget(9000, 1, 8192, 0) == -1); + CHECK(coli_serve_budget(0, 1, 8192, 0) == -1); + CHECK(coli_serve_budget(8192, 0, 8192, 1) == 0); + CHECK(coli_serve_budget(8193, 0, 8192, 1) == -1); + CHECK(coli_serve_budget(20, 0, 8192, 1) == 0); + CHECK(coli_serve_budget(8191, 100, 8192, 1) == 1); + CHECK(coli_serve_budget(20, INT_MAX, 8192, 0) == 8172); + CHECK(coli_serve_budget(3500, 1024, 4096, 0) == 596); + puts("serve budget ceiling: ok"); + return 0; +} diff --git a/c/tests/test_shard_kvb_refuse.c b/c/tests/test_shard_kvb_refuse.c new file mode 100644 index 000000000..ff6cadc42 --- /dev/null +++ b/c/tests/test_shard_kvb_refuse.c @@ -0,0 +1,151 @@ +/* layer_cuda_shard_kvb() (colibri.c, COLI_CUDA_ATTN_SHARD) -- the multi-device kv_b + * head-shard uploader. Its rb/weights/scale arithmetic is written for exactly fmt=1/2/3/4 + * (per-row byte strides, per-row or per-group scales); before the guard this file proves, + * an fmt=8 (fp8-e4m3-b128) kv_b was ADMITTED and only failed safe BY ACCIDENT: fmt=8 + * keeps its raw e4m3 bytes in q8 (q4 stays NULL, per the QT struct comment), so the + * function selected a NULL weight pointer with an int2 row stride and + * coli_cuda_tensor_upload_g's !weights check happened to reject the upload before + * anything dereferenced it -- silent, unnamed, and one refactor away from a misread + * (fmt=8's per-128x128-BLOCK scale array would ALSO have been sliced with per-row + * geometry). This probe pins the explicit refusal that replaced the accident: + * (1) an fmt=8 kv_b shard attempt refuses BY NAME on stderr, BEFORE any pointer/stride + * use, and the message says what serves fmt=8 instead (the absorb path on the + * layer home device); + * (2) no shard state is minted (n_kv_b_shard==0, kv_b_shard[] untouched, + * kv_b.cuda_eligible unchanged); + * (3) the notice is bounded (once per process per fmt, never per layer); + * (4) other un-shardable fmts (fmt=6 here) refuse by name too, with their own message; + * (5) an allowlisted fmt (fmt=1) is NOT refused -- with no device context initialized + * its upload fails silently and no shard is minted, but no refusal line appears, + * so the guard is format-targeted, not a blanket gate. + * PROOF-OF-BITE: built against the pre-guard colibri.c, (1)/(3)/(4) fail -- the fmt=8 + * call slid past the format check into the accidental-safe upload rejection with no + * message at all. Needs -DCOLI_CUDA (CUDA=1) to compile the function; a CPU-only build + * SKIPs loudly instead of pretending to cover it. No GPU work is performed: every path + * exercised here returns before any device context exists, so this runs on a CUDA build + * host even without a card. + * Portable stderr capture: freopen/dup2, same seam as tests/test_kvb_notice.c. */ +#define main coli_glm_main_unused +#include "../colibri.c" +#undef main + +#include +#include +#include + +#ifndef COLI_CUDA +int main(void){ + printf("test_shard_kvb_refuse: SKIP (built without COLI_CUDA -- build with CUDA=1 on a CUDA host to exercise layer_cuda_shard_kvb)\n"); + return 0; +} +#else + +#include + +static int fails = 0; +#define CHECK(c) do{ if(!(c)){ printf("FAIL %s:%d: %s\n", __FILE__, __LINE__, #c); fails++; } }while(0) + +static int redirect_stderr(const char *path){ + fflush(stderr); + int saved = dup(fileno(stderr)); + if(saved<0){ printf("FAIL: dup(stderr) failed -- capture seam unusable, aborting\n"); exit(1); } + if(!freopen(path, "w+", stderr)){ + printf("FAIL: freopen(%s) failed -- capture seam unusable, aborting\n", path); + exit(1); + } + return saved; +} +static void restore_stderr(int saved, char *buf, size_t bufsz){ + fflush(stderr); + long n = ftell(stderr); + if(n<0) n=0; + rewind(stderr); + size_t want = (size_t)n < bufsz-1 ? (size_t)n : bufsz-1; + size_t got = fread(buf, 1, want, stderr); + buf[got] = 0; + fflush(stderr); + dup2(saved, fileno(stderr)); + close(saved); +} + +static int count_sub(const char *hay, const char *needle){ + int n=0; const char *p=hay; + while((p=strstr(p,needle))){ n++; p++; } + return n; +} + +/* kv_b-shaped fixture at H=4, Q=8, V=8 -> O=H*(Q+V)=64; I=200 (nblkI=2, partial column + * tail -- the exact scale geometry the shard's per-row slicing would have misread). */ +enum { H=4, Q=8, V=8, O=H*(Q+V), I=200 }; + +static void mk_layer_fmt8(Layer *l, int8_t *q8, float *s){ + memset(l,0,sizeof *l); + for(int64_t i=0;i<(int64_t)O*I;i++) q8[i]=(int8_t)(i*37+11); + int64_t nblk=fp8_nblk(O)*fp8_nblk(I); + for(int64_t i=0;ikv_b.fmt=8; l->kv_b.O=O; l->kv_b.I=I; l->kv_b.gs=0; + l->kv_b.q8=q8; l->kv_b.s=s; /* q4 stays NULL: the fmt=8 convention */ + l->kv_b.cuda_eligible=1; l->kv_b.cuda_device=0; +} + +static void run_shard(Layer *l, char *buf, size_t bufsz){ + int saved = redirect_stderr("tests/tmp_shard_kvb_refuse.stderr"); + layer_cuda_shard_kvb(l,H,Q,V); + restore_stderr(saved, buf, bufsz); + remove("tests/tmp_shard_kvb_refuse.stderr"); +} + +int main(void){ + /* pretend a 2-device dense-CUDA setup so the function's early gate passes; no + * device context is ever initialized, and none is needed (see file header). */ + g_cuda_enabled=1; g_cuda_dense=1; g_cuda_ndev=2; + g_cuda_devices[0]=0; g_cuda_devices[1]=0; + + static int8_t q8[(int64_t)O*I]; + static float s[((O+127)/128)*((I+127)/128)]; + static float srow[O]; /* per-row scales for the allowlisted fmt=1 case */ + char err[4096]; + Layer l; + + /* (1)+(2): fmt=8 refuses by name, names the absorb path, mints no shard state */ + mk_layer_fmt8(&l,q8,s); + run_shard(&l,err,sizeof err); + CHECK(strstr(err,"layer_cuda_shard_kvb")!=NULL); + CHECK(strstr(err,"refus")!=NULL); + CHECK(strstr(err,"fmt=8")!=NULL); + CHECK(strstr(err,"absorb path")!=NULL); /* says what serves fmt=8 instead */ + CHECK(l.n_kv_b_shard==0); + CHECK(l.kv_b_shard[0]==NULL && l.kv_b_shard[1]==NULL); + CHECK(l.kv_b.cuda_eligible==1); /* shard bookkeeping never ran */ + + /* (3): bounded -- a second fmt=8 layer (any of the other 60) adds no second line */ + mk_layer_fmt8(&l,q8,s); + run_shard(&l,err,sizeof err); + CHECK(count_sub(err,"layer_cuda_shard_kvb")==0); + CHECK(l.n_kv_b_shard==0); + + /* (4): fmt=6 (E8/IQ3, single 4-byte scale tag) refuses by name with its own message */ + memset(&l,0,sizeof l); + l.kv_b.fmt=6; l.kv_b.O=O; l.kv_b.I=I; l.kv_b.gs=0; + l.kv_b.q4=(uint8_t*)q8; l.kv_b.s=s; /* non-NULL on purpose: only the guard saves it */ + run_shard(&l,err,sizeof err); + CHECK(strstr(err,"layer_cuda_shard_kvb")!=NULL); + CHECK(strstr(err,"refus")!=NULL); + CHECK(strstr(err,"fmt=6")!=NULL); + CHECK(l.n_kv_b_shard==0); + + /* (5): allowlisted fmt=1 is NOT refused -- upload fails silently (no device + * context), no shard minted, but no refusal line either */ + memset(&l,0,sizeof l); + for(int i=0;iF32 and F16->F32 are lossless IEEE-754 + * widenings: for every one of the 65536 possible input bit patterns -- every + * normal and subnormal exponent, +-0, +-inf, and every NaN payload -- there + * is exactly one correct output, so there is no tolerance to allow: the + * vectorized tiers must produce bit-identical floats to the scalar reference + * (bf16_to_f32/f16_to_f32) for the entire input space, not just a sample. + * + * Runs the full 65536-value sweep (which lands on an exact multiple of both + * the AVX2 8-wide and SSE4.1 4-wide vector width) and a set of odd-length, + * odd-offset slices of the same data so the scalar remainder tail in each + * tier's loop is also exercised, not just the vectorized body. */ +#include "../st.h" +#include +#include +#include + +static int failures; +#define CHECK(cond, ...) do { if (!(cond)) { \ + fprintf(stderr, "FAIL %s:%d: ", __FILE__, __LINE__); \ + fprintf(stderr, __VA_ARGS__); fputc('\n', stderr); failures++; } } while (0) + +static int32_t bits_of(float f) { int32_t u; memcpy(&u, &f, 4); return u; } + +static void check_range(const uint16_t *src, int64_t n, const char *label) { + float *bulk_bf = malloc((size_t)n * sizeof(float)); + float *bulk_f16 = malloc((size_t)n * sizeof(float)); + if (!bulk_bf || !bulk_f16) { fprintf(stderr, "OOM\n"); exit(2); } + bf16_to_f32_bulk(src, bulk_bf, n); + f16_to_f32_bulk(src, bulk_f16, n); + int64_t bf_diffs = 0, f16_diffs = 0; + int64_t bf_first = -1, f16_first = -1; + for (int64_t i = 0; i < n; i++) { + float ref_bf = bf16_to_f32(src[i]); + float ref_f16 = f16_to_f32(src[i]); + if (bits_of(ref_bf) != bits_of(bulk_bf[i])) { bf_diffs++; if (bf_first < 0) bf_first = i; } + if (bits_of(ref_f16) != bits_of(bulk_f16[i])) { f16_diffs++; if (f16_first < 0) f16_first = i; } + } + if (bf_diffs) fprintf(stderr, " bf16 first mismatch idx=%lld h=0x%04x scalar=0x%08x bulk=0x%08x\n", + (long long)bf_first, src[bf_first], bits_of(bf16_to_f32(src[bf_first])), bits_of(bulk_bf[bf_first])); + if (f16_diffs) fprintf(stderr, " f16 first mismatch idx=%lld h=0x%04x scalar=0x%08x bulk=0x%08x\n", + (long long)f16_first, src[f16_first], bits_of(f16_to_f32(src[f16_first])), bits_of(bulk_f16[f16_first])); + CHECK(bf_diffs == 0, "%s: bf16_to_f32_bulk %lld/%lld mismatches", label, (long long)bf_diffs, (long long)n); + CHECK(f16_diffs == 0, "%s: f16_to_f32_bulk %lld/%lld mismatches", label, (long long)f16_diffs, (long long)n); + free(bulk_bf); free(bulk_f16); +} + +int main(void) { + /* every possible 16-bit pattern, once each -- covers +-0, every normal + * and subnormal exponent, +-inf, and every NaN payload for both formats */ + uint16_t *all = malloc(65536 * sizeof(uint16_t)); + if (!all) { fprintf(stderr, "OOM\n"); return 2; } + for (int32_t h = 0; h <= 0xFFFF; h++) all[h] = (uint16_t)h; + + check_range(all, 65536, "full sweep"); + printf("st f16/bf16 simd: exhaustive 65536/65536 patterns exact\n"); + + /* odd offsets/lengths so the scalar remainder tail (n%8 for AVX2, n%4 for + * SSE4.1) actually runs, on real (not synthetic-zero) data */ + static const int64_t offs[] = {0, 1, 2, 3, 5, 7, 9, 13, 15, 16, 17, 100, 12345}; + static const int64_t lens[] = {1, 2, 3, 4, 5, 6, 7, 8, 9, 15, 16, 17, 65423, 65535}; + for (size_t oi = 0; oi < sizeof(offs)/sizeof(offs[0]); oi++) { + for (size_t li = 0; li < sizeof(lens)/sizeof(lens[0]); li++) { + int64_t off = offs[oi], len = lens[li]; + if (off + len > 65536) continue; + char label[64]; snprintf(label, sizeof label, "off=%lld len=%lld", (long long)off, (long long)len); + check_range(all + off, len, label); + } + } + printf("st f16/bf16 simd: tail-remainder slices ok\n"); + + free(all); + if (failures) { fprintf(stderr, "st f16/bf16 simd: %d failure(s)\n", failures); return 1; } + puts("st f16/bf16 simd: ok"); + return 0; +} diff --git a/c/tests/test_st_missing.c b/c/tests/test_st_missing.c index 5d2adc883..98a044b35 100644 --- a/c/tests/test_st_missing.c +++ b/c/tests/test_st_missing.c @@ -14,6 +14,7 @@ int main(void) { puts("test_st_missing: skipped on Windows (fork)"); return 0; } #include #include #include +#include #include "../st.h" #define CHECK(c) do { if (!(c)) { fprintf(stderr, "%s:%d: check failed: %s\n", __FILE__, __LINE__, #c); return 1; } } while (0) @@ -82,6 +83,35 @@ int main(void) { /* and a present tensor still reads fine with the index around */ { shards S; memset(&S, 0, sizeof S); st_init(&S, D); float v[8]; CHECK(st_read_f32(&S, "a", v, 0) == 8); st_destroy(&S); } wipe(D); + + /* 5. an index path that does not fit the buffer is refused, not opened truncated: + * with a 1193-char directory, "/model" is all snprintf would keep of the name. + * The case has to build that directory, so it runs where the file system takes a + * path that long: on macOS PATH_MAX is 1024 and mkdir answers ENAMETOOLONG, and + * there the case steps aside -- the Linux runs cover the guard. Any other mkdir + * failure is a real one and fails the test. */ + { + char dir[1200] = "tmp_st_long", p[1300]; size_t n = strlen(dir); + int made = mkdir(dir, 0755) == 0; + while (made && n < 1193) { + size_t k = 1193 - n - 1 > 200 ? 200 : 1193 - n - 1; + dir[n++] = '/'; memset(dir + n, 'd', k); n += k; dir[n] = 0; + made = mkdir(dir, 0755) == 0; + } + int too_long = !made && errno == ENAMETOOLONG, refused = 1; + if (made) { + snprintf(p, sizeof p, "%s/model", dir); + FILE *f = fopen(p, "wb"); CHECK(f != NULL); fputs("{\"weight_map\":{\"a\":\"x\"}}", f); fclose(f); + st_index ix; memset(&ix, 0, sizeof ix); st_index_load(&ix, dir); + refused = ix.root == NULL && ix.map == NULL; + remove(p); + } + for (char *s; (s = strrchr(dir, '/')) != NULL; *s = 0) rmdir(dir); + rmdir(dir); + CHECK(made || too_long); + CHECK(refused); + if (!made) puts("st_index_load long-path guard: skipped, the file system caps the path"); + } puts("st_die_missing diagnosis tests: ok"); return 0; } diff --git a/c/tests/test_systemone_api.py b/c/tests/test_systemone_api.py new file mode 100644 index 000000000..0e63e10e0 --- /dev/null +++ b/c/tests/test_systemone_api.py @@ -0,0 +1,182 @@ +"""POST /v1/systemone: the request and the reply of TypeSafe's Jev API, served +by the brio channel. + +A client written for Jev sends `state`, `model` and a map of questions typed +noul / choice / score, and reads `answers` keyed by its own ids plus +`usage.input_tokens/output_tokens`. This pins the mapping onto the `questions` +form of /v1/brio with the deterministic scoring engine of test_brio_api: what +each primitive puts in the prompt, what comes back, the confidence formula +their docs give, that any model name is accepted on this route, and that the +state is still photographed once for all the questions. +""" +import json +import math +import threading +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from openai_server import APIServer +if __package__: + from .test_brio_api import ScoringEngine +else: + from test_brio_api import ScoringEngine + +STATE = ("Hi, I've been trying to connect my Stripe account for 3 days and the " + "integration keeps failing. I'm losing sales. Please help ASAP.") + + +def confidence(ps): + n = len(ps) + return (n * max(ps) - 1.0) / (n - 1) + + +class SystemOneApi(unittest.TestCase): + def serve(self, table): + self.engine = ScoringEngine(table) + self.server = APIServer(("127.0.0.1", 0), self.engine, "test-model") + self.addCleanup(self.server.shutdown) + self.addCleanup(self.server.server_close) + threading.Thread(target=self.server.serve_forever, daemon=True).start() + self.base = f"http://127.0.0.1:{self.server.server_port}" + + def post(self, body): + request = Request(self.base + "/v1/systemone", data=json.dumps(body).encode(), + headers={"Content-Type": "application/json"}) + with urlopen(request, timeout=10) as response: + return json.loads(response.read()) + + def post_error(self, body): + try: + self.post(body) + except HTTPError as error: + return error.code, json.loads(error.read()) + self.fail("expected an error") + + def pins(self): + return [c["prompt"] for c in self.engine.calls if c["pin"]] + + def test_noul_is_the_probability_of_yes(self): + self.serve({"yes": -0.1, "no": -2.0}) + out = self.post({"model": "jev-latest", "state": STATE, "questions": { + "urgency": {"type": "noul", "instructions": "Does this message express urgency?"}}}) + answer = out["answers"]["urgency"] + self.assertEqual(answer["type"], "noul") + want = math.exp(-0.1) / (math.exp(-0.1) + math.exp(-2.0)) + self.assertAlmostEqual(answer["noul"], want, places=5) + self.assertNotIn("confidence", answer) # their docs: noul carries none + # the question is asked as a yes/no question, the state as the context + asked = [c["prompt"] for c in self.engine.calls if "Does this message" in c["prompt"]] + self.assertTrue(asked and "Answer yes or no." in asked[0], asked[:1]) + + def test_noul_criteria_go_into_the_question(self): + self.serve({"yes": -0.1, "no": -2.0}) + self.post({"model": "jev-latest", "state": STATE, "questions": { + "q": {"type": "noul", "instructions": "Is the customer at risk of churning?", + "criteria": {"true": "they threaten to leave or mention losses", + "false": "a routine question"}}}}) + pinned = self.pins()[1] + self.assertIn("yes: they threaten to leave", pinned) + self.assertIn("no: a routine question", pinned) + + def test_choice_labels_probabilities_and_confidence(self): + self.serve({"billing": -0.2, "technical": -3.0, "sales": -4.0}) + out = self.post({"model": "jev-latest", "state": STATE, "questions": { + "department": {"type": "choice", "instructions": "Which team should handle this?", + "criteria": {"billing": "payments, invoices, Stripe", + "technical": "bugs and outages", + "sales": "pricing and plans"}}}}) + answer = out["answers"]["department"] + self.assertEqual(answer["type"], "choice") + self.assertEqual(answer["choice"], "billing") + self.assertEqual(list(answer["probabilities"]), ["billing", "technical", "sales"]) + self.assertAlmostEqual(sum(answer["probabilities"].values()), 1.0, places=5) + self.assertAlmostEqual(answer["confidence"], + confidence(list(answer["probabilities"].values())), places=5) + # the descriptions are in the question, the labels are the options scored + pinned = self.pins()[1] + self.assertIn("- billing: payments, invoices, Stripe", pinned) + self.assertIn("- sales: pricing and plans", pinned) + scored = [c["prompt"] for c in self.engine.calls if not c["pin"]] + self.assertTrue(any(p.endswith(" technical") for p in scored), scored[-3:]) + + def test_choice_with_null_descriptions_still_lists_the_labels(self): + self.serve({"a": -0.1, "b": -1.0}) + out = self.post({"model": "jev-latest", "state": STATE, "questions": { + "q": {"type": "choice", "criteria": {"a": None, "b": None}}}}) + self.assertEqual(out["answers"]["q"]["choice"], "a") + self.assertIn("- a\n- b", self.pins()[1]) + + def test_score_expected_value_legend_and_confidence(self): + self.serve({"4": -0.1, "1": -5.0, "2": -5.0, "3": -5.0, "5": -5.0}) + levels = ["not urgent", "low", "moderate", "high", "critical"] + out = self.post({"model": "jev-latest", "state": STATE, "questions": { + "urgency": {"type": "score", "instructions": "How urgent is this?", "criteria": levels}}}) + answer = out["answers"]["urgency"] + self.assertEqual(answer["type"], "score") + self.assertEqual(answer["legend"], {str(i + 1): d for i, d in enumerate(levels)}) + self.assertEqual(list(answer["probabilities"]), ["1", "2", "3", "4", "5"]) + ps = answer["probabilities"] + self.assertAlmostEqual(sum(ps.values()), 1.0, places=5) + self.assertAlmostEqual(answer["score"], sum(int(k) * v for k, v in ps.items()), places=5) + self.assertGreater(answer["score"], 3.9) # the mass sits on level 4 + self.assertAlmostEqual(answer["confidence"], confidence(list(ps.values())), places=5) + self.assertIn("4: high", self.pins()[1]) + + def test_structured_state_and_instructions_are_serialized(self): + self.serve({"yes": -0.1, "no": -2.0}) + out = self.post({"model": "jev-latest", + "state": {"ticket": 4711, "text": "refund not received"}, + "questions": {"q": {"type": "noul", + "instructions": {"ask": "is this about money?"}}}}) + self.assertEqual(out["model"], "test-model") # the served model answers + state_pin = self.pins()[0] + self.assertIn('"ticket": 4711', state_pin) + self.assertIn('"ask": "is this about money?"', self.pins()[1]) + + def test_state_is_photographed_once_for_all_questions(self): + self.serve({"yes": -0.1, "no": -2.0, "a": -0.1, "b": -1.0}) + out = self.post({"model": "jev-latest", "state": STATE, "questions": { + "one": {"type": "noul", "instructions": "q1?"}, + "two": {"type": "choice", "criteria": {"a": "x", "b": "y"}}, + "three": {"type": "score", "criteria": ["bad", "good"]}}}) + self.assertEqual(list(out["answers"]), ["one", "two", "three"]) + pins = self.pins() + self.assertEqual(len(pins), 4, pins) # the state, then each question once + self.assertEqual(pins[0], f"Context:\n{STATE}\n\n") + self.assertEqual(set(out["usage"]), {"input_tokens", "output_tokens"}) + self.assertGreater(out["usage"]["input_tokens"], 0) + self.assertGreater(out["usage"]["output_tokens"], 0) + + def test_validation_errors_are_422(self): + self.serve({}) + code, body = self.post_error({"model": "jev-latest", "questions": {"q": {"type": "noul"}}}) + self.assertEqual(code, 422) + self.assertIn("state", body["error"]["message"]) + code, body = self.post_error({"model": "jev-latest", "state": STATE, + "questions": {"q": {"type": "guess"}}}) + self.assertEqual(code, 422) + self.assertIn("type", body["error"]["message"]) + code, body = self.post_error({"model": "jev-latest", "state": STATE, + "questions": {"q": {"type": "choice", "criteria": ["a", "b"]}}}) + self.assertEqual(code, 422) + code, body = self.post_error({"model": "jev-latest", "state": STATE, + "questions": {"q": {"type": "score", "criteria": ["only one"]}}}) + self.assertEqual(code, 422) + self.assertIn("2 to 10", body["error"]["message"]) + code, body = self.post_error({"model": "jev-latest", "state": STATE, "questions": []}) + self.assertEqual(code, 422) + + def test_brio_route_still_checks_the_model_name(self): + self.serve({"a": -0.1, "b": -1.0}) + request = Request(self.base + "/v1/brio", + data=json.dumps({"model": "jev-latest", "state": STATE, + "options": ["a", "b"]}).encode(), + headers={"Content-Type": "application/json"}) + with self.assertRaises(HTTPError) as caught: + urlopen(request, timeout=10) + self.assertEqual(caught.exception.code, 404) + + +if __name__ == "__main__": + unittest.main() diff --git a/c/tests/test_tok_gpt2.c b/c/tests/test_tok_gpt2.c new file mode 100644 index 000000000..7a27ffe25 --- /dev/null +++ b/c/tests/test_tok_gpt2.c @@ -0,0 +1,65 @@ +/* GPT-2 pre-tokenizer family (bare ByteLevel with use_regex, no Split: OLMoE, + * GPT-NeoX) against HF-tokenizers-generated expectations. tests/tok_gpt2_tiny.json + * is a byte-level BPE trained with `tokenizers` on a few sentences (14 KB, no + * model download) with exactly OLMoE's pre_tokenizer block, and + * tests/tok_gpt2_cases.txt holds the ids HF produced for each case. Guards the + * rules where GPT-2 and cl100k differ, which tok.h used to get wrong for OLMoE: + * " 4" and " 20" as one piece, unbounded digit runs, a non-space prefix NOT glued + * to the word ("(abc"), case-sensitive contractions, whitespace runs before text, + * and added-token boundaries. Round-trips every case. Measured on the real OLMoE + * tokenizer: 1560/1708 cases identical before this family, 1708/1708 after. */ +#define _GNU_SOURCE +#include "../tok.h" + +int main(void) { + Tok T; + tok_load(&T, "tests/tok_gpt2_tiny.json"); + if (!T.gpt2) { fprintf(stderr, "test_tok_gpt2: GPT-2 family (bare ByteLevel) not detected\n"); return 1; } + FILE *f = fopen("tests/tok_gpt2_cases.txt", "rb"); + if (!f) { perror("tests/tok_gpt2_cases.txt"); return 1; } + /* fgets, not getline: MinGW's UCRT lacks getline and this must run on + * the windows job. Case lines are short; 8 KB is generous. */ + char line[8192]; + int pass = 0, tot = 0, dpass = 0; + while (fgets(line, sizeof(line), f)) { + size_t nr = strlen(line); + while (nr > 0 && (line[nr-1] == '\n' || line[nr-1] == '\r')) line[--nr] = 0; + if (nr == 0) continue; + char *tab = strchr(line, '\t'); if (!tab) continue; + *tab = 0; + const char *text = line, *idstr = tab + 1; + char tbuf[4096]; int tn = 0; + for (const char *q = text; *q && tn < 4095; q++) { + if (q[0]=='\\' && q[1]=='n') { tbuf[tn++]='\n'; q++; } + else if (q[0]=='\\' && q[1]=='t') { tbuf[tn++]='\t'; q++; } + else if (q[0]=='\\' && q[1]=='r') { tbuf[tn++]='\r'; q++; } + else if (q[0]=='\\' && q[1]=='\\') { tbuf[tn++]='\\'; q++; } + else tbuf[tn++] = *q; + } + tbuf[tn] = 0; + int exp[512], ne = 0; + for (const char *q = idstr; *q; ) { + while (*q == ',' || *q == ' ') q++; + if (!*q) break; + exp[ne++] = atoi(q); + while (*q && *q != ',') q++; + } + int got[512]; int ng = tok_encode(&T, tbuf, tn, got, 512); + int ok = (ng == ne); + for (int i = 0; i < ng && ok; i++) ok = (got[i] == exp[i]); + tot++; if (ok) pass++; + char dec[8192]; int dn = tok_decode(&T, got, ng, dec, 8191); + int drt = (dn == tn) && !memcmp(dec, tbuf, tn); + if (drt) dpass++; + if (!ok || !drt) { + fprintf(stderr, "MISMATCH text=%s\n exp(%d):", text, ne); + for (int i = 0; i < ne; i++) fprintf(stderr, " %d", exp[i]); + fprintf(stderr, "\n got(%d):", ng); + for (int i = 0; i < ng; i++) fprintf(stderr, " %d", got[i]); + fprintf(stderr, "\n decode_ok=%d\n", drt); + } + } + fclose(f); + printf("test_tok_gpt2: ENCODE %d/%d DECODE %d/%d\n", pass, tot, dpass, tot); + return (pass == tot && dpass == tot) ? 0 : 2; +} diff --git a/c/tests/test_v4_cli.py b/c/tests/test_v4_cli.py index 402fe50f4..2d80b155a 100644 --- a/c/tests/test_v4_cli.py +++ b/c/tests/test_v4_cli.py @@ -106,6 +106,7 @@ def fake_run(command, **kwargs): self.assertNotIn("text", captured) self.assertEqual(captured["env"]["CHAT"], "1") self.assertEqual(captured["env"]["MAX_NEW"], "32") + self.assertEqual(captured["env"]["SNAP"], os.path.abspath(str(root))) finally: directory.cleanup() @@ -117,18 +118,29 @@ def test_v4_engine_environment_forwards_ram_and_context(self): self.assertEqual(env["CTX"], "4096") def test_sister_engines_get_snap_from_the_model_flag(self): - """#1501: `coli run` handed olmoe (and every non-GLM engine) an + """#1501 / #1600: `coli run` handed olmoe (and every non-GLM engine) an environment without SNAP, so the engine exited with "started without - a model" while chat and serve, which set it elsewhere, worked.""" + a model" while chat and serve, which set it elsewhere, worked. + SNAP is the model directory, same as env_for() for glm: --model wins + over a leftover SNAP in the parent environment.""" from family_registry import family_ids for arch in [f for f in family_ids() if f != "glm"]: args = argparse.Namespace(ngen=8, temp=None, ram=0, ctx=None, model="models/demo") env = self.cli.env_for_engine(args, arch) - self.assertEqual(env.get("SNAP"), os.path.abspath("models/demo"), arch) - # an explicit SNAP in the caller's environment still wins + self.assertEqual(env.get("SNAP"), os.path.abspath(args.model), arch) with mock.patch.dict(os.environ, {"SNAP": "/elsewhere"}): args = argparse.Namespace(ngen=8, temp=None, ram=0, ctx=None, model="models/demo") - self.assertEqual(self.cli.env_for_engine(args, "olmoe")["SNAP"], "/elsewhere") + self.assertEqual( + self.cli.env_for_engine(args, "olmoe")["SNAP"], + os.path.abspath(args.model), + ) + + def test_v41_ram_flag_overrides_inherited_budget(self): + args = argparse.Namespace(ngen=8, temp=None, ram=96, ctx=4096) + with mock.patch.dict(os.environ, {"RAM_GB": "32"}): + env = self.cli.env_for_engine(args, "deepseek_v41") + self.assertEqual(env["RAM_GB"], "96") + self.assertEqual(env["CTX"], "4096") def test_kimi_engine_environment_forwards_ram(self): """#855: `--ram` reached the environment for deepseek_v4 only, so on Kimi diff --git a/c/tests/tok_gpt2_cases.txt b/c/tests/tok_gpt2_cases.txt new file mode 100644 index 000000000..c632bf972 --- /dev/null +++ b/c/tests/tok_gpt2_cases.txt @@ -0,0 +1,27 @@ + 4 262 +x = 4 << 20 87,347,262,282,281 +MAX_BODY = 4 << 20 301,62,384,347,262,282,281 +1234567 apples 16,294,383,401 +(abc 7,386 +[def] {ghi} 58,285,60,349,389,92 +'S 6,50 +'s 288 +It's 40,83,288 +don't. 372,289,13 +they're 83,256,88,290 + x 220,278 +x 87,350 + 350 +a\n\nb 64,198,198,65 +a.\nb 64,13,198,65 +Hello, world! 385,11,381,0 +\t\tif x:\n\t\t\treturn 197,197,314,278,25,343,197,197,400 +café 388 +<|endoftext|>x.<|endoftext|> 405,87,13,405 +x.<|endoftext|> 87,13,405 + <|endoftext|> 350,405 + été 42 220,342,83,342,366 +a b c 64,220,276,350,220,66 +end.\n 68,324,13,198 +\n\n 198,198 +x\r\ny 87,201,198,88 diff --git a/c/tests/tok_gpt2_tiny.json b/c/tests/tok_gpt2_tiny.json new file mode 100644 index 000000000..c1c2acccd --- /dev/null +++ b/c/tests/tok_gpt2_tiny.json @@ -0,0 +1,1056 @@ +{ + "version": "1.0", + "truncation": null, + "padding": null, + "added_tokens": [ + { + "id": 405, + "content": "<|endoftext|>", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": false, + "special": true + }, + { + "id": 406, + "content": "<|padding|>", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": false, + "special": true + } + ], + "normalizer": { + "type": "NFC" + }, + "pre_tokenizer": { + "type": "ByteLevel", + "add_prefix_space": false, + "trim_offsets": true, + "use_regex": true + }, + "post_processor": null, + "decoder": { + "type": "ByteLevel", + "add_prefix_space": true, + "trim_offsets": true, + "use_regex": true + }, + "model": { + "type": "BPE", + "dropout": null, + "unk_token": null, + "continuing_subword_prefix": null, + "end_of_word_suffix": null, + "fuse_unk": false, + "byte_fallback": false, + "ignore_merges": false, + "vocab": { + "!": 0, + "\"": 1, + "#": 2, + "$": 3, + "%": 4, + "&": 5, + "'": 6, + "(": 7, + ")": 8, + "*": 9, + "+": 10, + ",": 11, + "-": 12, + ".": 13, + "/": 14, + "0": 15, + "1": 16, + "2": 17, + "3": 18, + "4": 19, + "5": 20, + "6": 21, + "7": 22, + "8": 23, + "9": 24, + ":": 25, + ";": 26, + "<": 27, + "=": 28, + ">": 29, + "?": 30, + "@": 31, + "A": 32, + "B": 33, + "C": 34, + "D": 35, + "E": 36, + "F": 37, + "G": 38, + "H": 39, + "I": 40, + "J": 41, + "K": 42, + "L": 43, + "M": 44, + "N": 45, + "O": 46, + "P": 47, + "Q": 48, + "R": 49, + "S": 50, + "T": 51, + "U": 52, + "V": 53, + "W": 54, + "X": 55, + "Y": 56, + "Z": 57, + "[": 58, + "\\": 59, + "]": 60, + "^": 61, + "_": 62, + "`": 63, + "a": 64, + "b": 65, + "c": 66, + "d": 67, + "e": 68, + "f": 69, + "g": 70, + "h": 71, + "i": 72, + "j": 73, + "k": 74, + "l": 75, + "m": 76, + "n": 77, + "o": 78, + "p": 79, + "q": 80, + "r": 81, + "s": 82, + "t": 83, + "u": 84, + "v": 85, + "w": 86, + "x": 87, + "y": 88, + "z": 89, + "{": 90, + "|": 91, + "}": 92, + "~": 93, + "¡": 94, + "¢": 95, + "£": 96, + "¤": 97, + "¥": 98, + "¦": 99, + "§": 100, + "¨": 101, + "©": 102, + "ª": 103, + "«": 104, + "¬": 105, + "®": 106, + "¯": 107, + "°": 108, + "±": 109, + "²": 110, + "³": 111, + "´": 112, + "µ": 113, + "¶": 114, + "·": 115, + "¸": 116, + "¹": 117, + "º": 118, + "»": 119, + "¼": 120, + "½": 121, + "¾": 122, + "¿": 123, + "À": 124, + "Á": 125, + "Â": 126, + "Ã": 127, + "Ä": 128, + "Å": 129, + "Æ": 130, + "Ç": 131, + "È": 132, + "É": 133, + "Ê": 134, + "Ë": 135, + "Ì": 136, + "Í": 137, + "Î": 138, + "Ï": 139, + "Ð": 140, + "Ñ": 141, + "Ò": 142, + "Ó": 143, + "Ô": 144, + "Õ": 145, + "Ö": 146, + "×": 147, + "Ø": 148, + "Ù": 149, + "Ú": 150, + "Û": 151, + "Ü": 152, + "Ý": 153, + "Þ": 154, + "ß": 155, + "à": 156, + "á": 157, + "â": 158, + "ã": 159, + "ä": 160, + "å": 161, + "æ": 162, + "ç": 163, + "è": 164, + "é": 165, + "ê": 166, + "ë": 167, + "ì": 168, + "í": 169, + "î": 170, + "ï": 171, + "ð": 172, + "ñ": 173, + "ò": 174, + "ó": 175, + "ô": 176, + "õ": 177, + "ö": 178, + "÷": 179, + "ø": 180, + "ù": 181, + "ú": 182, + "û": 183, + "ü": 184, + "ý": 185, + "þ": 186, + "ÿ": 187, + "Ā": 188, + "ā": 189, + "Ă": 190, + "ă": 191, + "Ą": 192, + "ą": 193, + "Ć": 194, + "ć": 195, + "Ĉ": 196, + "ĉ": 197, + "Ċ": 198, + "ċ": 199, + "Č": 200, + "č": 201, + "Ď": 202, + "ď": 203, + "Đ": 204, + "đ": 205, + "Ē": 206, + "ē": 207, + "Ĕ": 208, + "ĕ": 209, + "Ė": 210, + "ė": 211, + "Ę": 212, + "ę": 213, + "Ě": 214, + "ě": 215, + "Ĝ": 216, + "ĝ": 217, + "Ğ": 218, + "ğ": 219, + "Ġ": 220, + "ġ": 221, + "Ģ": 222, + "ģ": 223, + "Ĥ": 224, + "ĥ": 225, + "Ħ": 226, + "ħ": 227, + "Ĩ": 228, + "ĩ": 229, + "Ī": 230, + "ī": 231, + "Ĭ": 232, + "ĭ": 233, + "Į": 234, + "į": 235, + "İ": 236, + "ı": 237, + "IJ": 238, + "ij": 239, + "Ĵ": 240, + "ĵ": 241, + "Ķ": 242, + "ķ": 243, + "ĸ": 244, + "Ĺ": 245, + "ĺ": 246, + "Ļ": 247, + "ļ": 248, + "Ľ": 249, + "ľ": 250, + "Ŀ": 251, + "ŀ": 252, + "Ł": 253, + "ł": 254, + "Ń": 255, + "he": 256, + "Ġt": 257, + "es": 258, + "re": 259, + "ve": 260, + "wo": 261, + "Ġ4": 262, + "ĠI": 263, + "Ġa": 264, + "Ġf": 265, + "20": 266, + "<<": 267, + "de": 268, + "do": 269, + "in": 270, + "la": 271, + "ll": 272, + "ow": 273, + "ove": 274, + "Ġ1": 275, + "Ġb": 276, + "Ġs": 277, + "Ġx": 278, + "Ġhe": 279, + "Ġwo": 280, + "Ġ20": 281, + "Ġ<<": 282, + "Ġthe": 283, + "ĠIt": 284, + "def": 285, + "ine": 286, + "'d": 287, + "'s": 288, + "'t": 289, + "'re": 290, + "'ve": 291, + "'ll": 292, + "):": 293, + "23": 294, + "45": 295, + "67": 296, + "AX": 297, + "BO": 298, + "DY": 299, + "He": 300, + "MAX": 301, + "The": 302, + "ab": 303, + "ac": 304, + "af": 305, + "aÃ": 306, + "bla": 307, + "bove": 308, + "ck": 309, + "caf": 310, + "gh": 311, + "go": 312, + "hre": 313, + "if": 314, + "im": 315, + "is": 316, + "ick": 317, + "ju": 318, + "kn": 319, + "ld": 320, + "les": 321, + "line": 322, + "mp": 323, + "nd": 324, + "nk": 325, + "naÃ": 326, + "op": 327, + "ox": 328, + "pp": 329, + "pac": 330, + "qu": 331, + "rn": 332, + "row": 333, + "rld": 334, + "tu": 335, + "tes": 336, + "top": 337, + "we": 338, + "ytes": 339, + "zy": 340, + "¯ve": 341, + "é": 342, + "Ċĉ": 343, + "Ġ#": 344, + "Ġ(": 345, + "Ġ+": 346, + "Ġ=": 347, + "Ġ[": 348, + "Ġ{": 349, + "ĠĠ": 350, + "Ġdo": 351, + "Ġla": 352, + "Ġove": 353, + "Ġgo": 354, + "Ġis": 355, + "Ġju": 356, + "Ġkn": 357, + "Ġline": 358, + "ĠnaÃ": 359, + "Ġqu": 360, + "Ġwe": 361, + "Ġtwo": 362, + "Ġthre": 363, + "Ġtim": 364, + "retu": 365, + "Ġ42": 366, + "Ġabove": 367, + "Ġand": 368, + "Ġapp": 369, + "Ġfine": 370, + "Ġfox": 371, + "don": 372, + "llo": 373, + "Ġ123": 374, + "Ġbrow": 375, + "Ġbytes": 376, + "Ġspac": 377, + "Ġstop": 378, + "Ġhere": 379, + "Ġwon": 380, + "Ġworld": 381, + "Ġthey": 382, + "4567": 383, + "BODY": 384, + "Hello": 385, + "abc": 386, + "blank": 387, + "café": 388, + "ghi": 389, + "mps": 390, + "Ġdog": 391, + "Ġlazy": 392, + "Ġover": 393, + "Ġjumps": 394, + "Ġknow": 395, + "Ġnaïve": 396, + "Ġquick": 397, + "Ġthree": 398, + "Ġtimes": 399, + "return": 400, + "Ġapples": 401, + "Ġ1234567": 402, + "Ġbrown": 403, + "Ġspaces": 404 + }, + "merges": [ + [ + "h", + "e" + ], + [ + "Ġ", + "t" + ], + [ + "e", + "s" + ], + [ + "r", + "e" + ], + [ + "v", + "e" + ], + [ + "w", + "o" + ], + [ + "Ġ", + "4" + ], + [ + "Ġ", + "I" + ], + [ + "Ġ", + "a" + ], + [ + "Ġ", + "f" + ], + [ + "2", + "0" + ], + [ + "<", + "<" + ], + [ + "d", + "e" + ], + [ + "d", + "o" + ], + [ + "i", + "n" + ], + [ + "l", + "a" + ], + [ + "l", + "l" + ], + [ + "o", + "w" + ], + [ + "o", + "ve" + ], + [ + "Ġ", + "1" + ], + [ + "Ġ", + "b" + ], + [ + "Ġ", + "s" + ], + [ + "Ġ", + "x" + ], + [ + "Ġ", + "he" + ], + [ + "Ġ", + "wo" + ], + [ + "Ġ", + "20" + ], + [ + "Ġ", + "<<" + ], + [ + "Ġt", + "he" + ], + [ + "ĠI", + "t" + ], + [ + "de", + "f" + ], + [ + "in", + "e" + ], + [ + "'", + "d" + ], + [ + "'", + "s" + ], + [ + "'", + "t" + ], + [ + "'", + "re" + ], + [ + "'", + "ve" + ], + [ + "'", + "ll" + ], + [ + ")", + ":" + ], + [ + "2", + "3" + ], + [ + "4", + "5" + ], + [ + "6", + "7" + ], + [ + "A", + "X" + ], + [ + "B", + "O" + ], + [ + "D", + "Y" + ], + [ + "H", + "e" + ], + [ + "M", + "AX" + ], + [ + "T", + "he" + ], + [ + "a", + "b" + ], + [ + "a", + "c" + ], + [ + "a", + "f" + ], + [ + "a", + "Ã" + ], + [ + "b", + "la" + ], + [ + "b", + "ove" + ], + [ + "c", + "k" + ], + [ + "c", + "af" + ], + [ + "g", + "h" + ], + [ + "g", + "o" + ], + [ + "h", + "re" + ], + [ + "i", + "f" + ], + [ + "i", + "m" + ], + [ + "i", + "s" + ], + [ + "i", + "ck" + ], + [ + "j", + "u" + ], + [ + "k", + "n" + ], + [ + "l", + "d" + ], + [ + "l", + "es" + ], + [ + "l", + "ine" + ], + [ + "m", + "p" + ], + [ + "n", + "d" + ], + [ + "n", + "k" + ], + [ + "n", + "aÃ" + ], + [ + "o", + "p" + ], + [ + "o", + "x" + ], + [ + "p", + "p" + ], + [ + "p", + "ac" + ], + [ + "q", + "u" + ], + [ + "r", + "n" + ], + [ + "r", + "ow" + ], + [ + "r", + "ld" + ], + [ + "t", + "u" + ], + [ + "t", + "es" + ], + [ + "t", + "op" + ], + [ + "w", + "e" + ], + [ + "y", + "tes" + ], + [ + "z", + "y" + ], + [ + "¯", + "ve" + ], + [ + "Ã", + "©" + ], + [ + "Ċ", + "ĉ" + ], + [ + "Ġ", + "#" + ], + [ + "Ġ", + "(" + ], + [ + "Ġ", + "+" + ], + [ + "Ġ", + "=" + ], + [ + "Ġ", + "[" + ], + [ + "Ġ", + "{" + ], + [ + "Ġ", + "Ġ" + ], + [ + "Ġ", + "do" + ], + [ + "Ġ", + "la" + ], + [ + "Ġ", + "ove" + ], + [ + "Ġ", + "go" + ], + [ + "Ġ", + "is" + ], + [ + "Ġ", + "ju" + ], + [ + "Ġ", + "kn" + ], + [ + "Ġ", + "line" + ], + [ + "Ġ", + "naÃ" + ], + [ + "Ġ", + "qu" + ], + [ + "Ġ", + "we" + ], + [ + "Ġt", + "wo" + ], + [ + "Ġt", + "hre" + ], + [ + "Ġt", + "im" + ], + [ + "re", + "tu" + ], + [ + "Ġ4", + "2" + ], + [ + "Ġa", + "bove" + ], + [ + "Ġa", + "nd" + ], + [ + "Ġa", + "pp" + ], + [ + "Ġf", + "ine" + ], + [ + "Ġf", + "ox" + ], + [ + "do", + "n" + ], + [ + "ll", + "o" + ], + [ + "Ġ1", + "23" + ], + [ + "Ġb", + "row" + ], + [ + "Ġb", + "ytes" + ], + [ + "Ġs", + "pac" + ], + [ + "Ġs", + "top" + ], + [ + "Ġhe", + "re" + ], + [ + "Ġwo", + "n" + ], + [ + "Ġwo", + "rld" + ], + [ + "Ġthe", + "y" + ], + [ + "45", + "67" + ], + [ + "BO", + "DY" + ], + [ + "He", + "llo" + ], + [ + "ab", + "c" + ], + [ + "bla", + "nk" + ], + [ + "caf", + "é" + ], + [ + "gh", + "i" + ], + [ + "mp", + "s" + ], + [ + "Ġdo", + "g" + ], + [ + "Ġla", + "zy" + ], + [ + "Ġove", + "r" + ], + [ + "Ġju", + "mps" + ], + [ + "Ġkn", + "ow" + ], + [ + "ĠnaÃ", + "¯ve" + ], + [ + "Ġqu", + "ick" + ], + [ + "Ġthre", + "e" + ], + [ + "Ġtim", + "es" + ], + [ + "retu", + "rn" + ], + [ + "Ġapp", + "les" + ], + [ + "Ġ123", + "4567" + ], + [ + "Ġbrow", + "n" + ], + [ + "Ġspac", + "es" + ] + ] + } +} \ No newline at end of file diff --git a/c/tok.h b/c/tok.h index 54c5c6994..83b649df5 100644 --- a/c/tok.h +++ b/c/tok.h @@ -58,6 +58,8 @@ typedef struct { int o200k; /* pre_tokenizer regex family: 0 = cl100k (GLM), 1 = o200k (Inkling) */ int kimi; /* 1 = Kimi (K3) family: o200k rules + a leading \p{Han}-run rule, * Han excluded from the letter classes, no '/' tail in the punct rule */ + int gpt2; /* 1 = GPT-2 family (OLMoE / GPT-NeoX): a bare ByteLevel pre_tokenizer + * with use_regex and no Split, so HF applies the original GPT-2 pattern */ int rankbpe; /* 1 = no merges list (tiktoken-derived vocab): merge the adjacent * pair whose CONCATENATION has the lowest vocab id — exactly * tiktoken's byte_pair_encode, no recovered merges to diverge */ @@ -218,6 +220,13 @@ static void tok_load(Tok *T, const char *path){ * case-category classes (\p{Lu}...) which cl100k does not use */ jval *pt=json_get(root,"pre_tokenizer"); if(pt){ + /* No Split at all, just ByteLevel with use_regex (the default): HF runs + * the original GPT-2 regex, not cl100k. OLMoE's tokenizer.json is this. */ + jval *pty=json_get(pt,"type"); + if(pty && pty->t==J_STR && pty->str && !strcmp(pty->str,"ByteLevel")){ + jval *ur=json_get(pt,"use_regex"); + if(!ur || (ur->t==J_BOOL && ur->boolean)) T->gpt2=1; + } jval *ps=json_get(pt,"pretokenizers"); if(ps&&ps->t==J_ARR) for(int i=0;ilen;i++){ jval *pat=json_get(ps->kids[i],"pattern"); @@ -532,6 +541,52 @@ static void pretok_chunk_kimi(Tok *T, const unsigned char *p, int a, int b, int free(cp); free(off); } +/* ---------- pre-tokenizer GPT-2 (bare ByteLevel with use_regex: OLMoE, GPT-NeoX) ---------- + * The original GPT-2 pattern, which HF applies when the pre_tokenizer is a + * ByteLevel with use_regex=true and no Split in front of it: + * 's|'t|'re|'ve|'m|'ll|'d | ?\p{L}+ | ?\p{N}+ | ?[^\s\p{L}\p{N}]+ | \s+(?!\S) | \s+ + * It differs from cl100k in every rule: the contractions are case-sensitive, + * letters and digits take ONE optional leading space (not any non-letter), + * digit runs are unbounded, the punctuation run has no newline tail, and + * there is no \s*[\r\n]+ rule. Measured on OLMoE against `tokenizers`: + * 1560/1708 cases identical under the cl100k rules, every miss a " 4" / + * " 20"-style number or a non-space prefix glued to a word. */ +static void pretok_chunk_gpt2(Tok *T, const unsigned char *p, int a, int b, int *out, int *no, int max){ + int nb=b-a; if(nb<=0) return; + uint32_t *cp=malloc((nb+1)*sizeof(uint32_t)); int *off=malloc((nb+2)*sizeof(int)); int n=0; + for(int i=a;ii){ int end=(r id (split sugli added token, poi pretok+BPE) ---------- */ static int tok_encode(Tok *T, const char *text, int len, int *out, int max){ const unsigned char *p=(const unsigned char*)text; int no=0; int i=0; @@ -546,7 +601,8 @@ static int tok_encode(Tok *T, const char *text, int len, int *out, int max){ } int chunk_end = (hitpos<0) ? len : hitpos; if(chunk_end>i){ - if(T->kimi) pretok_chunk_kimi(T,p,i,chunk_end,out,&no,max); + if(T->gpt2) pretok_chunk_gpt2(T,p,i,chunk_end,out,&no,max); + else if(T->kimi) pretok_chunk_kimi(T,p,i,chunk_end,out,&no,max); else if(T->o200k) pretok_chunk_o200k(T,p,i,chunk_end,out,&no,max); else pretok_chunk(T,p,i,chunk_end,out,&no,max); } diff --git a/c/tools/bench_fp4_matmul.c b/c/tools/bench_fp4_matmul.c new file mode 100644 index 000000000..47a54c1ce --- /dev/null +++ b/c/tools/bench_fp4_matmul.c @@ -0,0 +1,144 @@ +/* bench_fp4_matmul.c — microbench for coli_fp4_matmul_batch_rows16_order. + * + * Measures the cost of the SIMD arm vs the scalar #else arm of the FP4 expert + * matmul used in DeepSeek V4 CPU prefill (see Makefile.deepseek-v4: the arm is + * selected by __AVX2__ / __ARM_NEON on the build). Build the same unit twice + * — default flags and with EXTRA_CFLAGS=-mno-avx2 (x86) — then compare. + * + * Usage: bench_fp4_matmul [S] [I] [O] [iters] [ref_file] + * S rows (token batch); the kernel contract caps S at 128 + * I/O input/output dims. Real V4-Flash experts: (S,4096,2048) w1/w3, + * (S,2048,4096) w2. Tiny fixture: (S,128,128). + * ref optional path: first run writes a float32 reference output, + * second run (other arm) compares bit-exactly against it. + * + * Data notes: + * - seeded xorshift, deterministic across platforms/builds + * - e8m0 scale bytes limited to 120..127 (2^-8 .. 2^-1) so products can + * never overflow to inf/NaN — full-range random scale bytes poison any + * bit-comparison with NaN (NaN != NaN). + */ +#include +#include +#include +#include +#include + +#if defined(_WIN32) +# include +# define BK_ACCESS(p, m) _access(p, m) +#else +# include +# define BK_ACCESS(p, m) access(p, m) +#endif + +/* Mirrors the kernel's own arm selection (deepseek_v4.c gates on __AVX2__ + * only today): on arm64 the SIMD arm does not exist yet, so the build is the + * scalar one even though NEON is available. */ +/* Mirrors the kernel's own arm selection; scalar when both/neither. */ +#if defined(__AVX2__) +# define BK_ARM_NAME "avx2" +#elif defined(__ARM_NEON) +# define BK_ARM_NAME "neon" +#else +# define BK_ARM_NAME "scalar" +#endif + +/* Link stubs: GPU-tier entry points referenced (never taken) by the CPU-only + * build of this unit. Signatures from deepseek_v4_internal.h. */ +int coli_v4_gpu_matvec_grouped(const void *w, float *output, + const float *input, int groups) { (void)w; (void)output; (void)input; (void)groups; return 1; } +int coli_v4_gpu_fp8_matvec(const void *w, float *output, + const float *input) { (void)w; (void)output; (void)input; return 1; } +int coli_v4_gpu_fp8_matmul_batch(const void *w, float *outputs, + const float *inputs, int batch) { (void)w; (void)outputs; (void)inputs; (void)batch; return 1; } + +void coli_fp4_matmul_batch_rows16_order(float *y, const uint8_t *q4, + const uint8_t *e8s, const float *x, + int S, int I, int O); + +static double now_s(void) { + struct timespec ts; +#if defined(__APPLE__) + clock_gettime(CLOCK_MONOTONIC, &ts); /* works since macOS 10.12 */ +#else + clock_gettime(CLOCK_MONOTONIC, &ts); +#endif + return ts.tv_sec + ts.tv_nsec * 1e-9; +} + +static uint32_t rng_state = 0x1234abcd; +static uint32_t rng(void) { + rng_state ^= rng_state << 13; rng_state ^= rng_state >> 17; + rng_state ^= rng_state << 5; return rng_state; +} + +int main(int argc, char **argv) { + int S = argc > 1 ? atoi(argv[1]) : 128; + int I = argc > 2 ? atoi(argv[2]) : 4096; + int O = argc > 3 ? atoi(argv[3]) : 2048; + int iters = argc > 4 ? atoi(argv[4]) : 7; + const char *ref_path = argc > 5 ? argv[5] : NULL; + + if (S < 1 || I % 2 || I % 32 || O < 1) { + fprintf(stderr, "bad shape: need S>=1, I%%2==0, I%%32==0 (FP4 group " + "packing), O>=1\n"); + return 2; + } + if (S > 128) { + fprintf(stderr, "S=%d exceeds the kernel's batch cap (sums[128]); the " + "sole caller chunks prefill to <=128 rows\n", S); + return 2; + } + + size_t rb = (size_t)I / 2, ng = (size_t)I / 32; + uint8_t *q4 = malloc(rb * O); + uint8_t *e8s = malloc(ng * O); + float *x = malloc((size_t)S * I * sizeof(float)); + float *y = malloc((size_t)S * O * sizeof(float)); + if (!q4 || !e8s || !x || !y) { fprintf(stderr, "oom\n"); return 1; } + for (size_t i = 0; i < rb * O; i++) q4[i] = (uint8_t)rng(); + for (size_t i = 0; i < ng * O; i++) e8s[i] = (uint8_t)(120 + (rng() & 7)); + for (size_t i = 0; i < (size_t)S * I; i++) x[i] = (float)((double)(rng() >> 8) / 8388608.0 - 1.0); + + memset(y, 0, (size_t)S * O * sizeof(float)); + coli_fp4_matmul_batch_rows16_order(y, q4, e8s, x, S, I, O); + + if (ref_path) { + FILE *f = fopen(ref_path, "rb"); + if (f) { + size_t n = (size_t)S * O; + float *ref = malloc(n * sizeof(float)); + if (fread(ref, sizeof(float), n, f) == n) { + size_t bad = 0; + for (size_t i = 0; i < n; i++) if (ref[i] != y[i]) bad++; + printf("BITEXACT vs %s : %s (%zu/%zu differ)\n", ref_path, + bad ? "MISMATCH" : "IDENTICAL", bad, n); + if (bad) return 3; + } else printf("BITEXACT: ref file wrong size\n"); + fclose(f); free(ref); + } else printf("BITEXACT: no ref (this run writes it)\n"); + } + + /* timing: discard first timed call (page faults), median of the rest */ + double best = 1e9, times[64]; int nt = 0; + if (iters > 63) iters = 63; + for (int it = 0; it < iters + 1; it++) { + double t0 = now_s(); + coli_fp4_matmul_batch_rows16_order(y, q4, e8s, x, S, I, O); + double dt = now_s() - t0; + if (it > 0) { times[nt++] = dt; if (dt < best) best = dt; } + } + for (int i = 0; i < nt; i++) for (int j = i + 1; j < nt; j++) + if (times[j] < times[i]) { double t = times[i]; times[i] = times[j]; times[j] = t; } + double med = times[nt / 2]; + double flops = 2.0 * S * I * O; + printf("KERNEL S=%d I=%d O=%d arm=%s iters=%d med=%.4fs best=%.4fs " + "gflops_med=%.2f\n", + S, I, O, BK_ARM_NAME, iters, med, best, flops / med / 1e9); + if (ref_path && BK_ACCESS(ref_path, 0) != 0) { + FILE *f = fopen(ref_path, "wb"); + if (f) { fwrite(y, sizeof(float), (size_t)S * O, f); fclose(f); } + } + return 0; +} diff --git a/c/tools/bench_fp4_matmul.sh b/c/tools/bench_fp4_matmul.sh new file mode 100755 index 000000000..fd7fb8e98 --- /dev/null +++ b/c/tools/bench_fp4_matmul.sh @@ -0,0 +1,57 @@ +#!/usr/bin/env bash +# bench_fp4_matmul.sh — build & A/B the FP4 expert matmul's SIMD vs scalar arm. +# +# Builds c/deepseek_v4.c twice (default flags = SIMD arm active; then with +# EXTRA_CFLAGS=-mno-avx2 on x86 to force the scalar #else arm), links the +# bench_fp4_matmul.c harness against each, verifies the two arms agree +# bit-exactly, and prints timings at the real DeepSeek-V4-Flash expert shapes. +# +# Usage (from c/): bash tools/bench_fp4_matmul.sh [S] [I] [O] [iters] +# Needs: gcc or clang, make. CPU-only; no model files required. +set -euo pipefail +cd "$(dirname "$0")/.." + +S="${1:-128}"; I="${2:-4096}"; O="${3:-2048}"; IT="${4:-7}" + +# Mirror the repo's own unit flags (Makefile.deepseek-v4 / .units). +ARCH=armv8.2-a+dotprod +SCALAR="" +AB=1 +case "$(uname -m)" in + x86_64*|amd64) ARCH=x86-64-v3 ; SCALAR="-mno-avx2" ;; + arm64|aarch64) SCALAR="-march=armv8-a+nosimd" ;; # +nosimd undefines __ARM_NEON: + # the scalar #else arm, against which + # the NEON arm must be bit-exact (#1696) + *) echo "unsupported arch $(uname -m)"; exit 1 ;; +esac + +UNIT="-DCOLI_V4_UNIT_NATIVE_QUANT" +COMMON="-O2 -fopenmp -I." +# Apple clang lacks -fopenmp; prefer a real gcc when present. +CC=gcc +$CC --version 2>/dev/null | grep -qi "gcc" || CC=gcc-15 +command -v $CC >/dev/null || { echo "no gcc with -fopenmp found"; exit 1; } +echo "using $($CC --version | head -1)" + +echo "== building SIMD-arm unit ==" +$CC $COMMON -march=$ARCH $UNIT -c deepseek_v4.c -o /tmp/bk_unit_simd.o +$CC $COMMON -march=$ARCH -c tools/bench_fp4_matmul.c -o /tmp/bk_main_simd.o +$CC /tmp/bk_unit_simd.o /tmp/bk_main_simd.o -o /tmp/bk_simd -lm -fopenmp + +REF=/tmp/bk_ref.f32 +rm -f "$REF" +if [ "$AB" = 1 ]; then + echo "== building scalar-arm unit ($SCALAR) ==" + $CC $COMMON -march=$ARCH $SCALAR $UNIT -c deepseek_v4.c -o /tmp/bk_unit_sca.o + $CC $COMMON -march=$ARCH $SCALAR -c tools/bench_fp4_matmul.c -o /tmp/bk_main_sca.o + $CC /tmp/bk_unit_sca.o /tmp/bk_main_sca.o -o /tmp/bk_sca -lm -fopenmp + echo "== SIMD arm (writes reference) ==" + /tmp/bk_simd "$S" "$I" "$O" "$IT" "$REF" + echo "== scalar arm (compares bit-exactly + times) ==" + /tmp/bk_sca "$S" "$I" "$O" "$IT" "$REF" | tee /tmp/bk_sca.log + grep -q "BITEXACT vs .*: IDENTICAL" /tmp/bk_sca.log || { echo "FAIL: SIMD and scalar arms differ"; exit 1; } + echo "Done. Repeat /tmp/bk_simd ... to re-verify bit-exactness." +else + echo "== no SIMD arm on this arch/kernel yet (#1696); scalar baseline only ==" + /tmp/bk_simd "$S" "$I" "$O" "$IT" +fi diff --git a/c/tools/benchmark_baseline.py b/c/tools/benchmark_baseline.py new file mode 100644 index 000000000..a94909407 --- /dev/null +++ b/c/tools/benchmark_baseline.py @@ -0,0 +1,243 @@ +#!/usr/bin/env python3 +"""Plan, collect and summarize a Colibri/SGLang/vLLM HTTP baseline.""" +import argparse +import datetime +import hashlib +import json +import math +import os +from pathlib import Path +import statistics + +try: + from . import benchmark_http_serving as http +except ImportError: + import benchmark_http_serving as http + +ENGINES = ("colibri", "sglang", "vllm") +SCHEMA = "colibri.serving-baseline/v1" + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def read_json(path): + return json.loads(Path(path).read_text(encoding="utf-8")) + + +def require(condition, message): + if not condition: + raise ValueError(message) + + +def text_fields(obj, names): + require(isinstance(obj, dict), "expected an object") + for name in names: + require(isinstance(obj.get(name), str) and obj[name].strip() + and not obj[name].startswith("REPLACE"), f"fill in {name}") + + +def load_manifest(path): + spec = read_json(path) + require(spec.get("schema") == SCHEMA, "unsupported manifest schema") + require(spec.get("comparison") in ("matched_artifact", "deployment"), + "comparison must be matched_artifact or deployment") + text_fields(spec, ("experiment", "workload", "cache_policy", "reasoning_policy", + "speculation_policy", "quality_protocol", "residency")) + require(spec["residency"] in ("fully_resident", "offload"), "invalid residency") + text_fields(spec.get("hardware"), ("host", "cpu", "ram", "gpu", "storage", "os", "driver")) + text_fields(spec.get("model"), ("source_revision", "tokenizer_sha256", "chat_template_sha256")) + for key in ("tokenizer_sha256", "chat_template_sha256"): + require(valid_hash(spec["model"][key]), f"invalid {key}") + matrix = spec.get("matrix", {}) + require(set(matrix) == {"concurrency", "rounds", "repeats", "warmup_requests", "max_tokens", + "temperature", "timeout", "slo_first_output", "slo_duration"}, + "matrix fields differ from the documented schema") + require(isinstance(matrix["concurrency"], list) and matrix["concurrency"], "empty concurrency") + values = matrix["concurrency"] + require(all(type(n) is int and n > 0 for n in values) and len(set(values)) == len(values), + "concurrency must contain unique positive integers") + for key in ("rounds", "repeats", "max_tokens"): + require(type(matrix[key]) is int and matrix[key] > 0, f"invalid {key}") + require(type(matrix["warmup_requests"]) is int and matrix["warmup_requests"] >= 0, + "invalid warmup_requests") + for key in ("temperature", "timeout", "slo_first_output", "slo_duration"): + value = matrix[key] + require(type(value) in (int, float) and math.isfinite(value) + and (value >= 0 if key == "temperature" else value > 0), f"invalid {key}") + engines = spec.get("engines", {}) + require(set(engines) == set(ENGINES), "provide exactly colibri, sglang and vllm") + for engine in engines.values(): + text_fields(engine, ("base_url", "served_model", "revision", "launch_command", + "artifact_sha256", "weight_format", "quantization", "api_key_env")) + http.endpoint(engine["base_url"]) + require(valid_hash(engine["artifact_sha256"]), "invalid artifact_sha256") + if spec["comparison"] == "matched_artifact": + identities = {(e["artifact_sha256"], e["weight_format"], e["quantization"]) + for e in engines.values()} + require(len(identities) == 1, "matched_artifact requires identical weights/format/quantization") + workload_path = Path(path).resolve().parent / spec["workload"] + workload, workload_hash = http.load_workload(workload_path) + require(len(workload) * matrix["repeats"] >= max(values), + "request count must reach the largest concurrency") + return spec, workload, workload_hash + + +def valid_hash(value): + return (isinstance(value, str) and len(value) == 64 + and all(c in "0123456789abcdef" for c in value) and len(set(value)) > 1) + + +def plan(spec): + for number in range(1, spec["matrix"]["rounds"] + 1): + shift = (number - 1) % len(ENGINES) + for engine in ENGINES[shift:] + ENGINES[:shift]: + yield {"round": number, "engine": engine, + "concurrency": spec["matrix"]["concurrency"]} + + +def collect(spec, workload, workload_hash, engine, number, output): + require(engine in ENGINES, "unknown engine") + require(1 <= number <= spec["matrix"]["rounds"], "round outside matrix") + output = Path(output) + output.mkdir(parents=True, exist_ok=True) + paths = [output / f"{engine}-r{number}-c{c}.json" for c in spec["matrix"]["concurrency"]] + require(not any(p.exists() for p in paths), "round already has results; use a new output directory") + config = spec["engines"][engine] + key = os.environ.get(config["api_key_env"], "") + matrix = spec["matrix"] + failed = False + for concurrency, path in zip(matrix["concurrency"], paths): + kwargs = dict(url=http.endpoint(config["base_url"]), model=config["served_model"], + concurrency=concurrency, max_tokens=matrix["max_tokens"], + temperature=matrix["temperature"], key=key, timeout=matrix["timeout"]) + started = datetime.datetime.now(datetime.timezone.utc).isoformat() + warmup = None + if matrix["warmup_requests"]: + prompts = [workload[i % len(workload)] for i in range(matrix["warmup_requests"])] + rows, summary = http.run(workload=prompts, repeats=1, **kwargs) + warmup = {"requests": rows, "summary": summary} + rows, summary = [], None + if warmup is None or warmup["summary"]["failed"] == 0: + rows, summary = http.run(workload=workload, repeats=matrix["repeats"], + slo_first_output=matrix["slo_first_output"], + slo_duration=matrix["slo_duration"], **kwargs) + report = {"schema": SCHEMA, "manifest": spec, "engine": engine, "round": number, + "concurrency": concurrency, "workload_sha256": workload_hash, + "harness_sha256": digest(http.__file__), "collector_sha256": digest(__file__), + "started_at": started, "status": "warmup_failed" if summary is None else "measured", + "warmup": warmup, "requests": rows, "summary": summary} + with path.open("x", encoding="utf-8") as stream: + stream.write(json.dumps(report, indent=2, allow_nan=False) + "\n") + failed |= summary is None or summary["failed"] > 0 + print(path) + if summary is None: + break + return int(failed) + + +def spread(values): + present = [v for v in values if v is not None] + if len(present) != len(values): + return {"count": len(present), "min": None, "median": None, "max": None} + return {"count": len(values), "min": min(values), + "median": statistics.median(values), "max": max(values)} + + +def compare(spec, workload, workload_hash, directory): + expected = {(e, r, c) for e in ENGINES for r in range(1, spec["matrix"]["rounds"] + 1) + for c in spec["matrix"]["concurrency"]} + reports, issues, fingerprints = {}, [], set() + files = sorted(Path(directory).glob("*-r*-c*.json")) + for path in files: + report = read_json(path) + require(report.get("schema") == SCHEMA, f"unsupported report: {path}") + require(report.get("manifest") == spec, f"manifest mismatch: {path}") + require(report.get("workload_sha256") == workload_hash, f"workload mismatch: {path}") + identity = (report["engine"], report["round"], report["concurrency"]) + require(identity in expected and identity not in reports, f"unexpected/duplicate cell: {path}") + fingerprints.add((report["harness_sha256"], report["collector_sha256"])) + reports[identity] = report + summary = report["summary"] + if summary is None: + issues.append(f"{path.name}: measurement absent") + continue + count = len(workload) * spec["matrix"]["repeats"] + rows = report["requests"] + require(len(rows) == count and [r["index"] for r in rows] == list(range(count)), + f"request set mismatch: {path}") + require(type(summary["wall_seconds"]) in (int, float) + and math.isfinite(summary["wall_seconds"]) and summary["wall_seconds"] > 0, + f"invalid wall time: {path}") + recalculated = http.summarize(rows, summary["wall_seconds"], + spec["matrix"]["slo_first_output"], spec["matrix"]["slo_duration"]) + require(summary == recalculated, f"summary differs from raw requests: {path}") + if summary["failed"]: + issues.append(f"{path.name}: failed requests") + if any(r["completion_tokens"] is None for r in rows if r["success"]): + issues.append(f"{path.name}: missing token usage") + if any(r["success"] and (r["first_output_seconds"] is None or r["completion_tokens"] == 0 + or r["finish_reason"] == "content_filter") for r in rows): + issues.append(f"{path.name}: empty or filtered output") + require(len(fingerprints) <= 1, "mixed benchmark implementations") + for e, r, c in sorted(expected - reports.keys()): + issues.append(f"missing {e}-r{r}-c{c}") + table = [] + for engine in ENGINES: + for concurrency in spec["matrix"]["concurrency"]: + cells = [reports.get((engine, r, concurrency)) + for r in range(1, spec["matrix"]["rounds"] + 1)] + summaries = [cell["summary"] if cell else None for cell in cells] + def metric(field, percentile=None): + values = [s[field] if s else None for s in summaries] + if percentile: + values = [v[percentile] if v else None for v in values] + return spread(values) + measured = [cell for cell in cells if cell and cell["summary"]] + lengths = [row["completion_tokens"] for cell in measured for row in cell["requests"] + if row["success"] and row["completion_tokens"] is not None] + table.append({"engine": engine, "concurrency": concurrency, + "rounds_measured": len(measured), + "aggregate_completion_tokens_per_second": metric("successful_completion_tokens_per_second"), + "first_output_p50_seconds": metric("successful_first_output_seconds", "p50"), + "first_output_p95_seconds": metric("successful_first_output_seconds", "p95"), + "first_output_p99_seconds": metric("successful_first_output_seconds", "p99"), + "duration_p95_seconds": metric("successful_duration_seconds", "p95"), + "failure_rate": metric("failure_rate"), + "slo_goodput_requests_per_second": metric("latency_slo", "goodput_requests_per_second"), + "reported_output_tokens": http.distribution(lengths)}) + return {"schema": SCHEMA, "manifest": spec, "workload_sha256": workload_hash, + "sources": [{"file": p.name, "sha256": digest(p)} for p in files], + "comparison": spec["comparison"], + "status": "incomplete_or_failed" if issues else "protocol_complete", + "quality": "not_assessed", "hardware_and_server_config": "operator_declared", + "token_latency": "not_measured", "memory_and_power": "external_telemetry_required", + "issues": issues, "cells": table} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("action", choices=("plan", "run", "compare")) + parser.add_argument("--manifest", required=True) + parser.add_argument("--engine", choices=ENGINES) + parser.add_argument("--round", type=int) + parser.add_argument("--results", default="baseline-results") + args = parser.parse_args() + try: + spec, workload, workload_hash = load_manifest(args.manifest) + if args.action == "plan": + print(json.dumps(list(plan(spec)), indent=2)) + return 0 + if args.action == "run": + require(args.engine is not None and args.round is not None, "run needs --engine and --round") + return collect(spec, workload, workload_hash, args.engine, args.round, args.results) + result = compare(spec, workload, workload_hash, args.results) + print(json.dumps(result, indent=2, allow_nan=False)) + return int(bool(result["issues"])) + except (ValueError, OSError, KeyError, TypeError) as exc: + parser.error(str(exc)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/c/tools/benchmark_http_serving.py b/c/tools/benchmark_http_serving.py new file mode 100644 index 000000000..cd21f88e0 --- /dev/null +++ b/c/tools/benchmark_http_serving.py @@ -0,0 +1,349 @@ +#!/usr/bin/env python3 +"""OpenAI chat streaming benchmark with bounded concurrency (stdlib only).""" +import argparse +import concurrent.futures +import hashlib +import json +import math +import os +import random +from pathlib import Path +import time +import urllib.error +import urllib.parse +import urllib.request + + +class StreamError(ValueError): + pass + + +class NoRedirect(urllib.request.HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def endpoint(base_url): + parsed = urllib.parse.urlsplit(base_url) + if (parsed.scheme not in ("http", "https") or not parsed.hostname + or parsed.username is not None or parsed.password is not None + or parsed.query or parsed.fragment): + raise ValueError("base URL must be HTTP(S), without credentials, query or fragment") + return base_url.rstrip("/") + "/chat/completions" + + +def load_workload(path): + raw = Path(path).read_bytes() + workload = [] + for line in raw.decode("utf-8").splitlines(): + if not line.strip(): + continue + item = json.loads(line) + if not isinstance(item, dict) or set(item) != {"messages"}: + raise ValueError("each JSONL row must contain only 'messages'") + messages = item["messages"] + if not isinstance(messages, list) or not messages: + raise ValueError("messages must be a nonempty list") + for message in messages: + if (not isinstance(message, dict) or set(message) != {"role", "content"} + or message["role"] not in ("system", "user", "assistant") + or not isinstance(message["content"], str)): + raise ValueError("messages require role and string content") + workload.append(item) + if not workload: + raise ValueError("workload is empty") + return workload, hashlib.sha256(raw).hexdigest() + + +def sse_events(response): + """Dispatch only complete SSE events; EOF is not a successful terminator.""" + data = [] + for raw in response: + line = raw.decode("utf-8").rstrip("\r\n") + if not line: + if data: + yield "\n".join(data) + data = [] + continue + field, _, value = line.partition(":") + if field == "data": + data.append(value[1:] if value.startswith(" ") else value) + + +def nonempty_text(value): + if value is not None and not isinstance(value, str): + raise StreamError("invalid_output_text") + return bool(value) + + +def has_output(delta): + if not isinstance(delta, dict): + raise StreamError("invalid_delta") + output = False + for field in ("content", "reasoning_content", "reasoning"): + output |= nonempty_text(delta.get(field)) + calls = delta.get("tool_calls") + if calls is None: + calls = [] + if not isinstance(calls, list): + raise StreamError("invalid_tool_calls") + legacy = delta.get("function_call") + if legacy is not None: + calls = calls + [{"function": legacy}] + for tool in calls: + if not isinstance(tool, dict): + raise StreamError("invalid_tool_call") + function = tool.get("function") + if function is None: + continue + if not isinstance(function, dict): + raise StreamError("invalid_tool_function") + for field in ("name", "arguments"): + output |= nonempty_text(function.get(field)) + return output + + +def request_one(url, payload, key, timeout, index, origin): + start = time.perf_counter() + result = {"index": index, "start_seconds": start - origin, + "success": False, "http_status": None, "error": None, + "first_output_seconds": None, "completion_tokens": None, + "finish_reason": None} + headers = {"Content-Type": "application/json", "Accept": "text/event-stream"} + if key: + headers["Authorization"] = "Bearer " + key + request = urllib.request.Request(url, data=json.dumps(payload).encode(), headers=headers) + try: + opener = urllib.request.build_opener(NoRedirect) + with opener.open(request, timeout=timeout) as response: + result["http_status"] = response.status + if response.headers.get_content_type() != "text/event-stream": + raise StreamError("unexpected_content_type") + done = False + for event in sse_events(response): + if event == "[DONE]": + done = True + break + chunk = json.loads(event) + if not isinstance(chunk, dict): + raise StreamError("invalid_chunk") + if "error" in chunk: + raise StreamError("stream_error") + choices = chunk.get("choices", []) + if not isinstance(choices, list) or len(choices) > 1: + raise StreamError("invalid_choices") + if choices and result["finish_reason"] is not None: + raise StreamError("choice_after_finish") + for choice in choices: + if (not isinstance(choice, dict) or type(choice.get("index")) is not int + or choice["index"] != 0): + raise StreamError("unexpected_choice") + delta = choice.get("delta") + if has_output({} if delta is None else delta): + if result["first_output_seconds"] is None: + result["first_output_seconds"] = time.perf_counter() - start + finish = choice.get("finish_reason") + if finish is not None: + if finish not in ("stop", "length", "tool_calls", "function_call", "content_filter"): + raise StreamError("unsuccessful_finish_reason") + result["finish_reason"] = finish + usage = chunk.get("usage") + if usage is not None: + if not isinstance(usage, dict): + raise StreamError("invalid_usage") + tokens = usage.get("completion_tokens") + if type(tokens) is not int or tokens < 0: + raise StreamError("invalid_completion_tokens") + result["completion_tokens"] = tokens + if not done or result["finish_reason"] is None: + raise StreamError("incomplete_stream") + result["success"] = True + except StreamError as exc: + result["error"] = str(exc) + except urllib.error.HTTPError as exc: + result["http_status"] = exc.code + result["error"] = "HTTPError" + exc.close() + except Exception as exc: + # Do not copy server bodies, request contents or credentials into reports. + result["error"] = type(exc).__name__ + result["duration_seconds"] = time.perf_counter() - start + return result + + +def distribution(values): + if not values: + return {"count": 0, "mean": None, "p50": None, "p95": None, "p99": None} + ordered = sorted(values) + result = {"count": len(values), "mean": sum(values) / len(values)} + for percentile in (50, 95, 99): + result[f"p{percentile}"] = ordered[math.ceil(len(ordered) * percentile / 100) - 1] + return result + + +def summarize(results, elapsed, slo_first_output=None, slo_duration=None): + successful = [row for row in results if row["success"]] + counted = [row for row in successful if row["completion_tokens"] is not None] + tokens = sum(row["completion_tokens"] for row in counted) + complete = bool(successful) and len(counted) == len(successful) + summary = {"requests": len(results), "succeeded": len(successful), + "failed": len(results) - len(successful), + "failure_rate": (len(results) - len(successful)) / len(results), + "wall_seconds": elapsed, + "successful_requests_per_second": len(successful) / elapsed, + "successful_requests_with_usage": len(counted), + "reported_successful_completion_tokens": tokens, + "successful_completion_tokens_per_second": tokens / elapsed if complete else None, + "successful_duration_seconds": distribution([r["duration_seconds"] for r in successful]), + "successful_first_output_seconds": distribution([ + r["first_output_seconds"] for r in successful if r["first_output_seconds"] is not None])} + + paced = all(row.get("scheduled_seconds") is not None for row in results) + summary["arrival_timing"] = None + if paced: + summary["arrival_timing"] = { + "dispatch_delay_seconds": distribution([r["dispatch_delay_seconds"] for r in results]), + "successful_duration_seconds": distribution([r["arrival_duration_seconds"] for r in successful]), + "successful_first_output_seconds": distribution([ + r["arrival_first_output_seconds"] for r in successful + if r["arrival_first_output_seconds"] is not None])} + prefix = "arrival_" if paced else "" + summary["latency_slo"] = None + if slo_first_output is not None or slo_duration is not None: + met = sum( + (slo_duration is None or row[prefix + "duration_seconds"] <= slo_duration) + and (slo_first_output is None or ( + row[prefix + "first_output_seconds"] is not None + and row[prefix + "first_output_seconds"] <= slo_first_output)) + for row in successful) + summary["latency_slo"] = { + "timing_basis": "scheduled_arrival" if paced else "request_start", + "first_output_seconds": slo_first_output, "duration_seconds": slo_duration, + "requests_met": met, "fraction_of_attempts": met / len(results), + "goodput_requests_per_second": met / elapsed} + return summary + + +def run(url, workload, model, concurrency, repeats, max_tokens, temperature, key, timeout, + slo_first_output=None, slo_duration=None, request_rate=None, + arrival_distribution="periodic", seed=0): + count = len(workload) * repeats + scheduled = [0.0] * count + if request_rate is not None: + rng = random.Random(seed) + for index in range(1, count): + scheduled[index] = (index / request_rate if arrival_distribution == "periodic" + else scheduled[index - 1] + rng.expovariate(request_rate)) + origin = time.perf_counter() + # Without a rate, workers immediately take the next request. With a rate, + # absolute arrival deadlines keep submission independent of response time. + with concurrent.futures.ThreadPoolExecutor(max_workers=concurrency) as pool: + futures = [] + for index in range(count): + if request_rate is not None: + delay = origin + scheduled[index] - time.perf_counter() + if delay > 0: + time.sleep(delay) + payload = dict(workload[index % len(workload)], model=model, stream=True, + stream_options={"include_usage": True}, max_tokens=max_tokens, + temperature=temperature, n=1) + futures.append(pool.submit(request_one, url, payload, key, timeout, index, origin)) + results = [future.result() for future in futures] + if request_rate is not None: + for row in results: + row["scheduled_seconds"] = scheduled[row["index"]] + wait = max(0.0, row["start_seconds"] - row["scheduled_seconds"]) + row["dispatch_delay_seconds"] = wait + row["arrival_duration_seconds"] = wait + row["duration_seconds"] + first = row["first_output_seconds"] + row["arrival_first_output_seconds"] = None if first is None else wait + first + return results, summarize(results, time.perf_counter() - origin, slo_first_output, slo_duration) + + +def positive_int(value): + number = int(value) + if number <= 0: + raise argparse.ArgumentTypeError("must be positive") + return number + + +def nonnegative_int(value): + number = int(value) + if number < 0: + raise argparse.ArgumentTypeError("must be nonnegative") + return number + + +def positive_float(value): + number = float(value) + if not math.isfinite(number) or number <= 0: + raise argparse.ArgumentTypeError("must be finite and positive") + return number + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--base-url", required=True, help="API root including /v1") + parser.add_argument("--model", required=True) + parser.add_argument("--workload", required=True, help="JSONL rows containing messages") + parser.add_argument("--output", required=True, help="JSON report path") + parser.add_argument("--concurrency", type=positive_int, default=1) + parser.add_argument("--repeats", type=positive_int, default=1) + parser.add_argument("--warmup-requests", type=nonnegative_int, default=0, + help="unmeasured requests before the timed phase (default: 0)") + parser.add_argument("--request-rate", type=positive_float, help="mean scheduled arrivals per second") + parser.add_argument("--arrival-distribution", choices=("periodic", "poisson"), + default="periodic", help="arrival intervals when --request-rate is set") + parser.add_argument("--seed", type=int, default=0, help="Poisson arrival seed (default: 0)") + parser.add_argument("--max-tokens", type=positive_int, default=128) + parser.add_argument("--temperature", type=float, default=0) + parser.add_argument("--timeout", type=positive_float, default=60, help="socket operation timeout in seconds") + parser.add_argument("--api-key-env", default="OPENAI_API_KEY") + parser.add_argument("--slo-first-output", type=positive_float, help="first output latency target, seconds") + parser.add_argument("--slo-duration", type=positive_float, help="completed request latency target, seconds") + args = parser.parse_args() + if args.arrival_distribution == "poisson" and args.request_rate is None: + parser.error("--arrival-distribution poisson requires --request-rate") + try: + url = endpoint(args.base_url) + workload, digest = load_workload(args.workload) + if not math.isfinite(args.temperature) or args.temperature < 0: + raise ValueError("temperature must be finite and nonnegative") + if Path(args.output).resolve() == Path(args.workload).resolve(): + raise ValueError("output must differ from workload") + except (ValueError, OSError) as exc: + parser.error(str(exc)) + key = os.environ.get(args.api_key_env, "") + warmup = None + if args.warmup_requests: + prompts = [workload[i % len(workload)] for i in range(args.warmup_requests)] + rows, stats = run(url, prompts, args.model, args.concurrency, 1, + args.max_tokens, args.temperature, key, args.timeout) + warmup = {"summary": stats, "requests": rows} + results, summary = [], None + if warmup is None or warmup["summary"]["failed"] == 0: + results, summary = run(url, workload, args.model, args.concurrency, args.repeats, + args.max_tokens, args.temperature, key, args.timeout, + args.slo_first_output, args.slo_duration, args.request_rate, + args.arrival_distribution, args.seed) + report = {"schema_version": 1, "config": { + "endpoint": url, "model": args.model, "workload_sha256": digest, + "workload_rows": len(workload), "concurrency": args.concurrency, + "load_model": ("closed_loop" if args.request_rate is None else + "poisson" if args.arrival_distribution == "poisson" else "fixed_rate"), + "arrival_distribution": args.arrival_distribution if args.request_rate is not None else None, + "arrival_seed": args.seed if args.arrival_distribution == "poisson" else None, + "request_rate": args.request_rate, "warmup_requests": args.warmup_requests, + "repeats": args.repeats, "max_tokens": args.max_tokens, + "temperature": args.temperature, "socket_timeout_seconds": args.timeout, + "slo_first_output_seconds": args.slo_first_output, "slo_duration_seconds": args.slo_duration, + "stream": True, "include_usage": True, "n": 1}, + "status": "warmup_failed" if summary is None else "measured", + "warmup": warmup, "summary": summary, "requests": results} + Path(args.output).write_text(json.dumps(report, indent=2, allow_nan=False) + "\n", encoding="utf-8") + print(json.dumps(summary if summary is not None else warmup, indent=2)) + return 1 if summary is None or summary["failed"] else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/c/tools/clean.py b/c/tools/clean.py index c309212a4..f610f763a 100644 --- a/c/tools/clean.py +++ b/c/tools/clean.py @@ -27,12 +27,13 @@ "iobench", "iobench.exe", "backend_cuda.o", "backend_loader.o", "backend_cuda_test", "backend_cuda_test.exe", + "mxfp4_expert_cuda_test", "mxfp4_expert_cuda_test.exe", "backend_cuda_bench", "backend_cuda_bench.exe", "backend_metal.o", "backend_metal_test", "coli_cuda.dll", "coli_cuda.lib", "coli_cuda.exp", # hipcc emits an import library, export file and PDB alongside the DLL. "coli_hip.dll", "coli_hip.lib", "coli_hip.exp", "coli_hip.pdb", - "deepseek_v4", "deepseek_v4.exe", + "deepseek_v4", "deepseek_v4.exe", "deepseek_v4.cflags", "deepseek_v4.cudaflags", "deepseek_v41", "deepseek_v41.exe", "native_quant.o", "native_quant_parallel.o", "native_quant_dual.o", "native_quant_batch_avx512.o", "native_quant_fp4_rows16.o", diff --git a/c/tools/convert_gguf_to_olmoe.py b/c/tools/convert_gguf_to_olmoe.py new file mode 100644 index 000000000..aec91239f --- /dev/null +++ b/c/tools/convert_gguf_to_olmoe.py @@ -0,0 +1,392 @@ +#!/usr/bin/env python3 +"""Convert a GGUF OLMoE checkpoint into a colibri container for c/olmoe.c. + +Flow (one user-facing entry point): + wrapper (this file) -> gguf_reader (structure) -> gguf_dequant (numbers) + -> gguf_olmoe_profile (OLMoE mapping/layout/quant) + -> safetensors shards + config.json + tokenizer.json + +Output, per tensor, matches what c/olmoe.c loads: + * dense tensors dequantized to F16 (1-D norms kept F32), names remapped; + * routed experts dequantized and re-quantized to int8 row-wise, merged into + model.layers.N.mlp.experts.E.merged_weight (I8) + .qs (F32); + * config.json reconstructed from the GGUF olmoe.* metadata. GGUF does not + carry norm_topk_prob; the default is false (matching the official OLMoE + checkpoints) and --norm-topk-prob overrides it. q/k are taken as stored: + the official OLMoE GGUF is already in the HuggingFace layout. --qk-permuted + undoes llama.cpp's RoPE permutation for a GGUF that carries it; + * tokenizer.json rebuilt from tokenizer.ggml.* (byte-level GPT-2), validated + token-for-token against the official OLMoE tokenizer, so the container is + usable for chat/serve, not only loadable for validation. + +Safety: + * refuses a non-OLMoE GGUF, unknown tensor names, and missing required tensors; + * refuses to touch an --output that already contains a container unless the user + passes --overwrite (rebuild from scratch); it never silently skips work; + * writes shards atomically (tmp + rename), so a crash leaves no half shard; + * --remove-source-file deletes the source GGUF ONLY after a fully successful + conversion (never on a partial run); + * --dry-run prints the plan without writing anything or creating the output + directory. + +Usage: + python tools/convert_gguf_to_olmoe.py \\ + --input /path/to/model.gguf # or a dir holding exactly one .gguf + --output /path/to/olmoe-colibri # created if missing +""" + +import argparse +import json +import os +import sys +import time +from pathlib import Path + +import numpy as np + +from gguf_reader import GGUFReader +from gguf_dequant import dequantize +import gguf_olmoe_profile as profile + +try: + from safetensors.numpy import save_file +except ImportError as exc: + sys.exit("Missing dependency: %s. Install: pip install numpy safetensors" % exc) + + +class ConversionError(Exception): + pass + + +def resolve_sources(model): + path = Path(model).expanduser() + if path.is_dir(): + files = sorted(p for p in path.iterdir() if p.suffix == ".gguf") + if not files: + raise ConversionError("no .gguf files in %s" % path) + if len(files) > 1: + raise ConversionError( + "%s holds %d .gguf files; OLMoE is converted from a single file " + "for now, pass one explicitly" % (path, len(files))) + return [files[0]] + if not path.is_file(): + raise ConversionError("model not found: %s" % path) + return [path] + + +def required_names(config): + names = { + "model.embed_tokens.weight", + "lm_head.weight", + "model.norm.weight", + } + for layer in range(config["num_hidden_layers"]): + base = "model.layers.%d" % layer + names.update({ + base + ".input_layernorm.weight", + base + ".post_attention_layernorm.weight", + base + ".self_attn.q_proj.weight", + base + ".self_attn.k_proj.weight", + base + ".self_attn.v_proj.weight", + base + ".self_attn.o_proj.weight", + base + ".self_attn.q_norm.weight", + base + ".self_attn.k_norm.weight", + base + ".mlp.gate.weight", + }) + return names + + +def plan_tensors(reader): + """Return (dense, experts, unmapped) planned from the GGUF tensor list.""" + dense = [] + experts = {} + unmapped = [] + for tensor in reader.tensors: + name = tensor["name"] + parsed = profile.parse_expert(name) + if parsed is not None: + layer, kind = parsed + experts.setdefault(layer, {})[kind] = tensor + continue + target = profile.dense_target(name) + if target is None: + unmapped.append(name) + continue + dense.append((target, tensor)) + return dense, experts, unmapped + + +class OutputWriter: + def __init__(self, out_dir, flush_every): + self.out_dir = out_dir + self.flush_every = flush_every + self.buf = {} + self.shard_idx = 0 + + def add_many(self, items): + """Buffer related tensors and flush once the buffer reaches flush_every. + + Contract: every tensor of one logical unit -- e.g. an expert's int8 + merged_weight AND its f32 .qs -- is passed in a single call, so a flush + never splits the pair across shards during a normal run. + """ + for name, array in items: + self.buf[name] = np.ascontiguousarray(array) + if len(self.buf) >= self.flush_every: + self.flush() + + def flush(self): + if not self.buf: + return + destination = self.out_dir / ("model-%05d.safetensors" % self.shard_idx) + temporary = Path(str(destination) + ".tmp") + temporary.unlink(missing_ok=True) + try: + save_file(self.buf, str(temporary)) + os.replace(temporary, destination) + finally: + temporary.unlink(missing_ok=True) + print(" wrote %s (%d tensors)" % (destination.name, len(self.buf)), flush=True) + self.buf = {} + self.shard_idx += 1 + + +def _write_json_atomic(path, payload): + temporary = Path(str(path) + ".tmp") + temporary.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8") + os.replace(temporary, path) + + +def _dense_to_storage(logical): + if logical.ndim == 1: + return logical.astype(np.float32) + return logical.astype(np.float16) + + +def convert_dense(reader, dense, writer, config, qk_permuted): + for target, tensor in dense: + raw = reader.read_tensor(tensor) + flat = dequantize(raw, tensor["ggml_type"], tensor["numel"]) + logical = profile.to_logical(flat, tensor["dims"]) + if qk_permuted: + # llama.cpp's dense OLMo converter (OlmoModel, not OlmoeModel) + # RoPE-permutes q/k; c/olmoe.c uses the plain HuggingFace layout, so + # undo it. The official OLMoE GGUF is not permuted. + name = tensor["name"] + if name.endswith("attn_q.weight"): + logical = profile.restore_rope_layout(logical, config["num_attention_heads"], + config["num_attention_heads"]) + elif name.endswith("attn_k.weight"): + logical = profile.restore_rope_layout(logical, config["num_attention_heads"], + config["num_key_value_heads"]) + writer.add_many(((target, _dense_to_storage(logical)),)) + + +def convert_experts(reader, experts, writer, config): + layer_count = config["num_hidden_layers"] + expert_count = config["num_experts"] + hidden = config["hidden_size"] + inter = config["intermediate_size"] + want_w = inter * hidden + inter * hidden + hidden * inter + want_s = inter + inter + hidden + total = layer_count * expert_count + done_experts = 0 + + for layer in range(layer_count): + kinds = experts[layer] + raw_bytes, expert_layout = {}, {} + for kind in ("gate", "up", "down"): + tensor = kinds[kind] + raw_bytes[kind] = reader.read_tensor(tensor) + expert_layout[kind] = (profile.expert_matrix_shape(tensor["dims"]), + tensor["ggml_type"], + profile.expert_bytes_per_slice(tensor["dims"], tensor["ggml_type"])) + for expert in range(expert_count): + merged_name = "model.layers.%d.mlp.experts.%d.merged_weight" % (layer, expert) + qs_name = "model.layers.%d.mlp.experts.%d.qs" % (layer, expert) + per_kind = {} + for kind in ("gate", "up", "down"): + expert_shape, ggml_type, chunk = expert_layout[kind] + start = expert * chunk + flat = dequantize(raw_bytes[kind][start:start + chunk], ggml_type, + expert_shape[0] * expert_shape[1]) + per_kind[kind] = flat.reshape(expert_shape) + merged, scales = profile.merge_expert(per_kind["gate"], per_kind["up"], per_kind["down"]) + if merged.size != want_w or scales.size != want_s: + raise ConversionError( + "expert [%d,%d] produced %d bytes / %d scales, expected %d / %d" + % (layer, expert, merged.size, scales.size, want_w, want_s)) + # The int8 weights and their f32 scales are ONE logical unit: one + # add_many call keeps them in the same shard (see OutputWriter). + writer.add_many(((merged_name, merged), (qs_name, scales))) + done_experts += 1 + print(" layer %d/%d experts done (%d/%d)" + % (layer + 1, layer_count, done_experts, total), flush=True) + + +def _format_duration(seconds): + seconds = int(round(seconds)) + if seconds < 60: + return "%ds" % seconds + minutes, secs = divmod(seconds, 60) + if minutes < 60: + return "%dm %02ds" % (minutes, secs) + hours, minutes = divmod(minutes, 60) + return "%dh %02dm %02ds" % (hours, minutes, secs) + + +def run(args): + started = time.perf_counter() + sources = resolve_sources(args.input) + source = sources[0] + out_dir = Path(args.output).expanduser() + + print("colibri: converting a GGUF OLMoE checkpoint into a colibri container", flush=True) + print(" flow: %s" % source.name, flush=True) + print(" -> gguf_reader (header/metadata/tensor index)", flush=True) + print(" -> gguf_dequant + gguf_olmoe_profile (dequant, layout, int8 experts)", flush=True) + print(" -> safetensors shards + config.json + tokenizer.json", flush=True) + print(" input: %s" % source, flush=True) + print(" output: %s" % out_dir, flush=True) + + reader = GGUFReader(str(source)) + try: + arch = reader.metadata.get("general.architecture") + if arch != "olmoe": + raise ConversionError("general.architecture is %r, expected 'olmoe'" % arch) + config = profile.build_config(reader.metadata, + norm_topk_prob=args.norm_topk_prob) + dense, experts, unmapped = plan_tensors(reader) + if unmapped: + raise ConversionError("unrecognized tensors: %s" % ", ".join(sorted(unmapped)[:8])) + + present = {target for target, _ in dense} + for layer, kinds in experts.items(): + for expert in range(config["num_experts"]): + present.add("model.layers.%d.mlp.experts.%d.merged_weight" % (layer, expert)) + missing = sorted(required_names(config) - present) + if missing: + raise ConversionError("missing required tensors: %s" % ", ".join(missing[:8])) + for layer in range(config["num_hidden_layers"]): + kinds = experts.get(layer) + if not kinds or set(kinds) != {"gate", "up", "down"}: + raise ConversionError("layer %d: incomplete expert trio: %s" + % (layer, sorted(kinds) if kinds else None)) + + experts_total = config["num_hidden_layers"] * config["num_experts"] + if args.dry_run: + print("dry-run: %s" % source) + print(" architecture: olmoe, %d layers, %d experts (top-%d), hidden %d, inter %d, vocab %d" + % (config["num_hidden_layers"], config["num_experts"], + config["num_experts_per_tok"], config["hidden_size"], + config["intermediate_size"], config["vocab_size"])) + print(" dense tensors: %d" % len(dense)) + print(" expert tensors: %d (%d merged_weight + %d qs)" + % (experts_total * 2, experts_total, experts_total)) + if out_dir.exists() and any(out_dir.glob("model-*.safetensors")): + print(" note: output already contains a container; --overwrite rebuilds it") + return 0 + + existing_shards = sorted(out_dir.glob("model-*.safetensors")) + if existing_shards and not args.overwrite: + raise ConversionError( + "output already contains a converted container (%d safetensors shard(s)):\n" + " %s\n" + "refusing to overwrite it. Pass --overwrite to rebuild it from scratch." + % (len(existing_shards), out_dir)) + + out_dir.mkdir(parents=True, exist_ok=True) + if args.overwrite: + for path in list(out_dir.glob("model-*.safetensors")): + path.unlink() + for name in ("config.json", "tokenizer.json"): + sidecar = out_dir / name + if sidecar.is_file(): + sidecar.unlink() + print(" --overwrite: cleared previous container outputs", flush=True) + writer = OutputWriter(out_dir, args.flush_every) + _write_json_atomic(out_dir / "config.json", config) + print(" wrote config.json", flush=True) + tokenizer = profile.build_tokenizer(reader.metadata) + _write_json_atomic(out_dir / "tokenizer.json", tokenizer) + print(" wrote tokenizer.json (%d vocab, %d merges, %d added)" + % (len(tokenizer["model"]["vocab"]), + len(tokenizer["model"].get("merges", [])), + len(tokenizer["added_tokens"])), flush=True) + convert_dense(reader, dense, writer, config, args.qk_permuted) + convert_experts(reader, experts, writer, config) + writer.flush() + finally: + reader.close() + + if args.remove_source_file: + os.remove(source) + print(" removed source %s" % source, flush=True) + print("Done:") + print("- path: %s" % out_dir) + print("- time: %s" % _format_duration(time.perf_counter() - started)) + return 0 + + +def main(): + parser = argparse.ArgumentParser( + prog="convert_gguf_to_olmoe.py", + description="Convert an OLMoE GGUF checkpoint into a colibri container " + "readable by the c/olmoe.c engine (safetensors shards + " + "config.json + tokenizer.json).", + epilog=( + "examples:\n" + " convert_gguf_to_olmoe.py --input model.gguf --output ./olmoe-colibri\n" + " convert_gguf_to_olmoe.py --input ./gguf-dir --output ~/llm/olmoe --dry-run\n" + " convert_gguf_to_olmoe.py --input model.gguf --output ./olmoe --remove-source-file\n" + "\n" + "notes:\n" + " --input may be a single .gguf file or a directory holding exactly one\n" + " .gguf (more than one is refused: OLMoE is converted from one file).\n" + " --output is created (with any missing parent) if it does not exist.\n" + " If --output already holds a container, the run REFUSES with a message;\n" + " pass --overwrite to rebuild it from scratch.\n" + " --remove-source-file deletes the input .gguf only after the whole\n" + " conversion finished successfully; it never removes on a partial run.\n" + " --dry-run prints the plan, writes nothing and creates no directory.\n" + " The engine can load the result for validation; chat/serve work because\n" + " tokenizer.json is reconstructed from the GGUF tokenizer metadata.\n"), + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + parser.add_argument("--input", required=True, metavar="PATH", + help="source GGUF file, or a directory holding exactly one .gguf") + parser.add_argument("--output", required=True, metavar="DIR", + help="destination directory for the colibri container " + "(created, including missing parents, if absent)") + parser.add_argument("--dry-run", action="store_true", + help="print the conversion plan (layers, experts, output dir) " + "and exit without writing anything or creating --output") + parser.add_argument("--remove-source-file", action="store_true", + help="after a fully successful conversion, delete the source " + ".gguf (never on a partial or failed run)") + parser.add_argument("--overwrite", action="store_true", + help="rebuild an existing container in --output from scratch. " + "Without it, the run refuses when --output already holds " + "model-*.safetensors") + parser.add_argument("--norm-topk-prob", action="store_true", + help="set config.json norm_topk_prob=true (renormalize top-k router " + "weights). GGUF does not carry this field; the default is false, " + "which matches the official OLMoE checkpoints") + parser.add_argument("--qk-permuted", action="store_true", + help="undo llama.cpp's dense-OLMo RoPE permutation of q_proj/k_proj. " + "Off by default because the official OLMoE GGUF is already in " + "the HuggingFace layout; use it only for GGUFs that were permuted") + parser.add_argument("--flush-every", type=int, default=512, metavar="N", + help="flush an output safetensors shard every N tensors " + "(default: 512; smaller = more, smaller shards)") + args = parser.parse_args() + try: + return run(args) + except ConversionError as exc: + sys.stderr.write("ERROR: %s\n" % exc) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/c/tools/convert_olmoe_merged.py b/c/tools/convert_olmoe_merged.py index 36ac01d1f..e1be21eca 100644 --- a/c/tools/convert_olmoe_merged.py +++ b/c/tools/convert_olmoe_merged.py @@ -268,7 +268,7 @@ def main(): src = ap.add_mutually_exclusive_group(required=True) src.add_argument("--repo", help="HuggingFace repo ID (streams+deletes one shard at a time)") src.add_argument("--model", help="Local HF checkpoint directory (already fully downloaded)") - ap.add_argument("--out", required=True, help="Output directory for merged model") + ap.add_argument("--out", "--outdir", required=True, dest="out", help="Output directory for merged model") ap.add_argument("--flush-every", type=int, default=512, help="flush an output shard every N converted tensors") ap.add_argument("--min-free-gb", type=float, default=5.0, diff --git a/c/tools/convert_qwen36.py b/c/tools/convert_qwen36.py index 7ec413f94..d97d79f9b 100644 --- a/c/tools/convert_qwen36.py +++ b/c/tools/convert_qwen36.py @@ -17,6 +17,12 @@ pack_int4(). The qs layout (per-row f32 scales) is identical either way. This matches what c/qwen36.c's load_expert_merged expects: it detects int4 by ON-DISK SIZE (N/2 bytes) and unpacks in-place to int8, so the rest of the MoE path is unchanged. + * --down-bits 8 (with --ebits <= 4) is the MIXED layout: gate and up stay packed int4, + down_proj is int8 (--down-gs groups along its input, 0 = per row). One merged_weight per + expert still: uint8 [gate int4 packed | up int4 packed | down int8], 2*inter*hidden bytes + (int4: 1.5x, int8: 3x), qs = [gate scales | up scales | down scales]. Written to meta as + expert_down_bits / expert_down_gs. Q4_K_M keeps down/output in Q6_K; this is the + experiment #1370 asked for, with the engine reading each matrix in its own format. * Attention + router + shared-expert + norms stay f16 (they are tiny vs experts). * Real Qwen3.6 is a vision-language checkpoint: config dims live under `text_config`, and weight keys are prefixed `model.language_model.`. Both are handled transparently. @@ -37,9 +43,12 @@ Usage (local tiny model for end-to-end testing): python tools/convert_qwen36.py --model ../qwen36_tiny --out ../qwen36_tiny_i4 --ebits 8 + +Usage (mixed: int4 gs64 gate/up, int8 down -- the #1370 experiment): + python tools/convert_qwen36.py --model --out ./qwen36_i4_gs64_d8 --ebits 4 --gs 64 --down-bits 8 """ -import argparse, json, math, os, struct, sys +import argparse, json, math, os, struct, sys, tempfile from pathlib import Path # Windows: force UTF-8 output @@ -54,7 +63,7 @@ import torch from safetensors.numpy import safe_open as safe_open_np from safetensors.torch import safe_open as safe_open_pt - from safetensors.torch import save_file + from safetensors.torch import save_file, load_file except ImportError as exc: sys.exit(f"Missing dependencies: {exc}. Install: pip install torch safetensors") @@ -106,7 +115,7 @@ def quantize_row_grouped(w: "torch.Tensor", bits: int, gs: int): return q, scales.view(O, ng) -def make_merged(gate, up, down, ebits, gs=0): +def make_merged(gate, up, down, ebits, gs=0, down_bits=0, down_gs=0): gsz = gs # local `gs` is rebound to the gate scales below -- keep the group size safe """gate/up: [inter, H]; down: [H, inter] (torch, any fp). -> (merged_weight, qs f32 1D). @@ -118,20 +127,31 @@ def make_merged(gate, up, down, ebits, gs=0): The engine (c/qwen36.c) currently reads ebits>=5 (int8); the int4 path is WIP. qs (per-row f32 scales) is identical in both cases. """ - def q(t): - if gsz: - qt, s = quantize_row_grouped(t, ebits, gsz) # [O, ng] scales + def q(t, bits, group): + if group: + qt, s = quantize_row_grouped(t, bits, group) # [O, ng] scales return qt, s.reshape(-1) - qt, s = quantize_row(t.reshape(t.shape[0], -1), ebits) # rows along dim0 + qt, s = quantize_row(t.reshape(t.shape[0], -1), bits) # rows along dim0 return qt, s - gq, gs = q(gate) - uq, us = q(up) - dq, ds = q(down) - mw_i8 = torch.cat([gq.flatten(), uq.flatten(), dq.flatten()]).contiguous() - if ebits <= 4: - mw = pack_int4(mw_i8) # uint8, 2x4-bit per byte -> HALF size (true int4) + gq, gs = q(gate, ebits, gsz) + uq, us = q(up, ebits, gsz) + mixed = bool(down_bits) and down_bits != ebits + if mixed: + if not (ebits <= 4 and down_bits >= 5): + raise ValueError(f"mixed layout needs --ebits <= 4 and --down-bits >= 5 (got {ebits}/{down_bits})") + # gate|up packed int4, down int8 in the same uint8 slab: 2*inter*hidden bytes, + # which is neither the int4 (1.5x) nor the int8 (3x) size -- that is how the + # engine tells the three layouts apart, from the bytes, not from meta. + dq, ds = q(down, down_bits, down_gs) + gu = pack_int4(torch.cat([gq.flatten(), uq.flatten()]).contiguous()) + mw = torch.cat([gu, dq.flatten().to(torch.int8).view(torch.uint8)]).contiguous() else: - mw = mw_i8.to(torch.int8) # int8 storage (ebits 5..8) + dq, ds = q(down, ebits, gsz) + mw_i8 = torch.cat([gq.flatten(), uq.flatten(), dq.flatten()]).contiguous() + if ebits <= 4: + mw = pack_int4(mw_i8) # uint8, 2x4-bit per byte -> HALF size (true int4) + else: + mw = mw_i8.to(torch.int8) # int8 storage (ebits 5..8) qs = torch.cat([gs, us, ds]).contiguous().float() return mw, qs @@ -149,7 +169,6 @@ def _unpack_int4(packed: "torch.Tensor") -> "torch.Tensor": def _selftest(): """Round-trip check for the true-int4 packing used by --ebits<=4.""" - import random print("=== int4 pack/unpack selftest ===") ok = True for ebits in (2, 3, 4): @@ -160,9 +179,7 @@ def _selftest(): d = torch.randn(H, inter) * 3 mw, qs = make_merged(g, u, d, ebits) # serialize to safetensors + reload (exercises the real dtype path) - import tempfile, os td = tempfile.mkdtemp() - from safetensors.torch import save_file, load_file save_file({"merged_weight": mw, "qs": qs}, os.path.join(td, "e0.safetensors")) back = load_file(os.path.join(td, "e0.safetensors")) mw_r = back["merged_weight"] @@ -189,6 +206,24 @@ def qref(t): mw8, _ = make_merged(g, u, d, 8) print(f" ebits=8: dtype={mw8.dtype} (expected int8), bytes={mw8.numel()}") ok = ok and (mw8.dtype == torch.int8) + # mixed layout: gate/up int4 (gs 8 here), down int8 per row -- the bytes must be + # exactly the int4 packing of gate|up followed by the int8 rows of down, the size + # 2*inter*hidden, and the scales [gate ng | up ng | down per row] + inter, H, gsz = 8, 16, 8 + g = torch.randn(inter, H) * 3; u = torch.randn(inter, H) * 3; d = torch.randn(H, inter) * 3 + mwm, qsm = make_merged(g, u, d, 4, gs=gsz, down_bits=8, down_gs=0) + gq, _ = quantize_row_grouped(g, 4, gsz); uq, _ = quantize_row_grouped(u, 4, gsz) + dq, dsc = quantize_row(d, 8) + gu_ref = pack_int4(torch.cat([gq.flatten(), uq.flatten()]).contiguous()) + n_gu = gu_ref.numel() + same_gu = bool(torch.equal(mwm[:n_gu], gu_ref)) + same_d = bool(torch.equal(mwm[n_gu:].view(torch.int8), dq.flatten().to(torch.int8))) + size_ok = mwm.numel() == 2 * inter * H + ng = H // gsz + scales_ok = qsm.numel() == 2 * inter * ng + H and bool(torch.allclose(qsm[-H:], dsc.float())) + print(f" mixed int4gs{gsz}+down int8: dtype={mwm.dtype} bytes={mwm.numel()} (expected {2*inter*H}) " + f"gate/up={same_gu} down={same_d} scales={scales_ok}") + ok = ok and mwm.dtype == torch.uint8 and same_gu and same_d and size_ok and scales_ok print("SELFTEST", "PASS" if ok else "FAIL") sys.exit(0 if ok else 1) @@ -207,6 +242,11 @@ def main(): src.add_argument("--model", help="Local HF checkpoint directory") ap.add_argument("--out", required=False, help="Output container directory") ap.add_argument("--ebits", type=int, default=4, help="Expert quant bits (2..8, default 4)") + ap.add_argument("--down-bits", type=int, default=0, + help="bits for down_proj only (5..8; needs --ebits <= 4): the mixed layout, " + "int4 gate/up + int8 down in one slab. 0 = same as --ebits") + ap.add_argument("--down-gs", type=int, default=0, + help="scale group size along down_proj's input for --down-bits (0 = per row)") ap.add_argument("--gs", type=int, default=0, help="Group size for expert scales (e.g. 64). 0 = per-row (default). " "Group-scaled containers need engine support (expert_gs in meta).") @@ -234,6 +274,11 @@ def main(): if not 2 <= args.ebits <= 8: sys.exit(f"--ebits must be 2..8 (got {args.ebits})") + if args.down_bits and args.down_bits != args.ebits: + if not (args.ebits <= 4 and 5 <= args.down_bits <= 8): + sys.exit(f"--down-bits {args.down_bits} needs --ebits <= 4 and 5..8 (got --ebits {args.ebits})") + if args.down_gs < 0: + sys.exit("--down-gs must be >= 0") token = args.hf_token or os.environ.get("HF_TOKEN") if args.repo: @@ -487,7 +532,7 @@ def upload_local(path: Path): dk = k.replace("gate_up_proj", "down_proj") down = get_tensor(dk).float() # [E, H, inter] for e in range(E): - mw, qs = make_merged(gate[e], up[e], down[e], args.ebits, gs=args.gs) + mw, qs = make_merged(gate[e], up[e], down[e], args.ebits, gs=args.gs, down_bits=args.down_bits, down_gs=args.down_gs) tens[f"model.layers.{a}.mlp.experts.{e}.merged_weight"] = mw tens[f"model.layers.{a}.mlp.experts.{e}.qs"] = qs continue @@ -499,7 +544,7 @@ def upload_local(path: Path): for e in sorted(sep): d = sep[e] if "gate_proj" in d and "up_proj" in d and "down_proj" in d: - mw, qs = make_merged(d["gate_proj"], d["up_proj"], d["down_proj"], args.ebits, gs=args.gs) + mw, qs = make_merged(d["gate_proj"], d["up_proj"], d["down_proj"], args.ebits, gs=args.gs, down_bits=args.down_bits, down_gs=args.down_gs) tens[f"model.layers.{a}.mlp.experts.{e}.merged_weight"] = mw tens[f"model.layers.{a}.mlp.experts.{e}.qs"] = qs else: @@ -571,6 +616,10 @@ def shp_src(nm): if qn is not None: meta["qk_rope_head_dim"] = qn[0] meta["expert_gs"] = args.gs # 0 = per-row scales; >0 = group size along input dim + if args.down_bits and args.down_bits != args.ebits: + # mixed layout (docstring): down_proj in its own format, gate/up as ebits/gs + meta["expert_down_bits"] = args.down_bits + meta["expert_down_gs"] = args.down_gs meta["head_dim"] = meta.get("k_head_dim", meta.get("q_head_dim", 256)) meta["rope_dim"] = meta.get("qk_rope_head_dim", meta["head_dim"] // 4) # ---- DeltaNet (linear_attention) dims. From config (authoritative for dn; unlike the diff --git a/c/tools/engine_evidence.py b/c/tools/engine_evidence.py new file mode 100644 index 000000000..c0f4fa729 --- /dev/null +++ b/c/tools/engine_evidence.py @@ -0,0 +1,140 @@ +"""Helpers for reading what the engine wrote at startup. + +Recognizes the two typed lines the engine prints at startup -- the +"== GLM C engine ..." banner and the following "loaded in ..." record -- +and returns their fields as typed values, used by the evidence checkers +that read raw engine stdout. A line that merely looks like one of these +preambles but fails a field check is a bug worth surfacing loudly, so +parsing raises rather than silently skipping. +""" + +import math +import re + + +class PreambleError(ValueError): + """A line resembles an owned engine preamble but is not source-valid.""" + + +_INT32_MAX = 2**31 - 1 +_UINT_TEXT = r"(?:0|[1-9][0-9]*)" +_FIXED2_TEXT = r"(?:0|[1-9][0-9]*)\.[0-9]{2}" +IDOT_KERNELS = ( + "avx512-vnni", "avx-vnni", "avx2", "neon-i8mm", "neon", "vsx", + "scalar", +) + +_BANNER_RE = re.compile( + r"^== GLM C engine \(glm_moe_dsa\), cache=(?P" + _UINT_TEXT + + r") experts/layer \| compute experts@(?P" + _UINT_TEXT + + r")-bit dense@(?P" + _UINT_TEXT + + r")-bit \| idot: (?P" + "|".join(IDOT_KERNELS) + r") ==$") +_LOADED_RE = re.compile( + r"^loaded in (?P" + _FIXED2_TEXT + + r")s \| resident dense: (?P" + _FIXED2_TEXT + + r") MB \| layers=(?P" + _UINT_TEXT + + r") experts=(?P" + _UINT_TEXT + + r") \| MTP (?PACTIVE|absent|DISABLED \(multiplexed serve\)) " + r"\(draft=(?P" + _UINT_TEXT + r")\)$") + + +def parse_engine_banner(line): + """Return typed fields for the exact production "== GLM C engine" banner.""" + if not isinstance(line, str): + raise PreambleError(f"engine banner is not text: {line!r}") + match = _BANNER_RE.fullmatch(line) + if not match: + raise PreambleError(f"not an exact engine banner: {line!r}") + try: + cap, expert_bits, dense_bits = map( + int, match.group("cap", "expert_bits", "dense_bits")) + except ValueError as exc: + # _UINT_TEXT has no digit-count cap of its own, so a numeric + # field beyond Python's int-string conversion limit (4300 + # digits by default) reaches here as a bare ValueError, not + # this module's own PreambleError -- every caller of this + # module refuses with a named error, never a crash. + raise PreambleError(f"engine banner field is not a valid integer: " + f"{line!r}") from exc + if not 1 <= cap <= _INT32_MAX: + raise PreambleError(f"engine cache outside [1,{_INT32_MAX}]: {cap}") + if not 1 <= expert_bits <= 16 or not 1 <= dense_bits <= 16: + raise PreambleError( + f"engine compute bits outside [1,16]: {expert_bits}/{dense_bits}") + return { + "kind": "BANNER", "cap": cap, "expert_bits": expert_bits, + "dense_bits": dense_bits, "kernel": match.group("kernel"), + } + + +def parse_engine_loaded(line): + """Return typed fields for the exact "loaded in ..." record that follows the banner.""" + if not isinstance(line, str): + raise PreambleError(f"engine load record is not text: {line!r}") + match = _LOADED_RE.fullmatch(line) + if not match: + raise PreambleError(f"not an exact engine load record: {line!r}") + try: + load_s = float(match.group("load_s")) + resident_mb = float(match.group("resident_mb")) + layers, experts, draft = map( + int, match.group("layers", "experts", "draft")) + except ValueError as exc: + # Same rationale as parse_engine_banner: _FIXED2_TEXT/_UINT_TEXT + # have no digit-count cap, so a field beyond Python's + # int-string conversion limit reaches here as a bare + # ValueError, not this module's own PreambleError. + raise PreambleError(f"engine load record field is not a valid " + f"number: {line!r}") from exc + mtp = match.group("mtp") + if not math.isfinite(load_s) or not math.isfinite(resident_mb): + raise PreambleError("engine load metrics must be finite") + if load_s < 0 or resident_mb < 0: + raise PreambleError("engine load metrics must be nonnegative") + if not 1 <= layers <= 128: + raise PreambleError(f"engine layers outside [1,128]: {layers}") + if not 1 <= experts <= 4096: + raise PreambleError(f"engine experts outside [1,4096]: {experts}") + if not 0 <= draft <= 63: + raise PreambleError(f"engine draft outside [0,63]: {draft}") + if mtp == "DISABLED (multiplexed serve)" and draft != 0: + raise PreambleError("disabled multiplexed MTP requires draft=0") + return { + "kind": "LOADED", "load_s": load_s, "resident_mb": resident_mb, + "layers": layers, "experts": experts, "mtp": mtp, "draft": draft, + } + + +def parse_engine_preamble(line): + """Dispatch to the banner/loaded parser by prefix, or return None. + + Accepts only the two exact lines the engine actually prints at + startup -- the "== GLM C engine ..." banner and the "loaded in ..." + record that follows it -- each matched and range-checked field for + field. Every other line is refused: an ordinary line that shares + neither literal prefix returns None (see below); a line that DOES + share one of the two prefixes but does not go on to match that + parser's exact grammar raises PreambleError rather than being + treated as unowned. + + The prefix test itself is mechanical, not semantic, which makes the + dispatch slightly broader than the two records it is named for: + "loaded index ..." also starts with "loaded in" purely because + "index" itself starts with "in", and would be routed to + parse_engine_loaded and refused there -- fail-loud, deliberately, + the same as any other line sharing the prefix without matching the + grammar. This is a documented characteristic, not a live concern: + no engine anywhere in this tree emits "loaded index" or any other + line that collides with either prefix today (confirmed against + every "loaded"-prefixed printf in the C sources), so there is + nothing to actually tolerate -- if a future engine change ever adds + one, this note is why the refusal is expected rather than a + surprise. + """ + if not isinstance(line, str): + raise PreambleError(f"engine preamble is not text: {line!r}") + if line.startswith("== GLM C engine"): + return parse_engine_banner(line) + if line.startswith("loaded in"): + return parse_engine_loaded(line) + return None diff --git a/c/tools/eval_glm.py b/c/tools/eval_glm.py index 76097cd6e..efc7a96e1 100644 --- a/c/tools/eval_glm.py +++ b/c/tools/eval_glm.py @@ -22,7 +22,25 @@ # leve di ricerca: passate al motore via env TOPP=0.9 python3 tools/eval_glm.py --snap /path/to/glm52_i4 --data ./bench --tasks mmlu --ram 15 """ -import os, sys, subprocess, argparse, random, json, tempfile, time, threading +import argparse +import hashlib +import json +import math +import os +import random +import re +import signal +import subprocess +import sys +import tempfile +import threading +import time + +_TOOLS_DIR = os.path.dirname(os.path.abspath(__file__)) +if _TOOLS_DIR not in sys.path: + sys.path.insert(0, _TOOLS_DIR) +from engine_evidence import (PreambleError, parse_engine_banner, + parse_engine_loaded, parse_engine_preamble) # mini-set OFFLINE per testare la meccanica (NON misura qualita': domande banali) SMOKE = [ @@ -38,6 +56,276 @@ "arc_challenge": {"GLM-5.2 (pubbl.)": None}, } + +class EvidenceError(ValueError): + """A SCORE stream cannot support a complete, finite evidence result.""" + + +class ChildTerminateRequested(BaseException): + """A termination signal (SIGTERM) arrived while the engine child was + running. Raised from a signal handler installed only for the + duration of that child's run, so it unwinds through the same + ``finally`` cleanup as any other mid-run exception (including + Python's own SIGINT-to-KeyboardInterrupt) and the child is never + left running as a zombie/orphan. + """ + + +_INT32_MAX = 2**31 - 1 +_ENGINE_TEXT_MAX_BYTES = 256 << 20 +_UINT_TEXT = r"(?:0|[1-9][0-9]*)" +_C17G_TEXT = (r"-?(?:0|[1-9][0-9]*)(?:\.[0-9]+)?" + r"(?:e[+-](?:0[0-9]|[1-9][0-9]{1,2}))?") +_SCORE_RE = re.compile( + rf"^({_C17G_TEXT}) ({_UINT_TEXT}) ([01])$") + + +def _checked_engine_text_size(length, label): + if type(length) is not int or not 0 <= length <= _ENGINE_TEXT_MAX_BYTES: + raise EvidenceError( + f"{label} exceeds the inclusive 256 MiB engine limit") + return length + + +def parse_c17g(text): + """Parse the canonical C-locale numeric token the engine actually emits. + + The engine has shipped two mutually exclusive SCORE spellings across + its history -- the byte-compatible ``printf("%.6f")`` form dev still + emits today, and the newer opt-in ``printf("%.17g")`` evidence form. + Both are exact, finite, round-trippable spellings of the same + C-locale numeric domain, so both are accepted here (the function name + is kept for the newer form this module's identity checks depend on); + a text that is neither exact spelling -- including any non-canonical + variant such as a non-canonical ``%.17g`` corpus (nan/inf/-inf never + survive: neither spelling is finite-preserving for them) -- is + refused. + """ + # fullmatch (not match/search) here is UNREACHABLE via the real + # production path: this function's only caller, parse_score_result, + # only ever passes the substring _SCORE_RE captured for this exact + # sub-pattern (_SCORE_RE embeds _C17G_TEXT verbatim as its first + # group), and a regex capture group's matched text always already + # satisfies the sub-pattern it was captured with -- fullmatch, + # match and search are identical for that input by construction, + # not merely by absence of a counterexample. For a DIRECT caller + # (this function is called directly elsewhere in this module's own + # tests) fullmatch vs. match/search is still equivalent in + # practice: RAN mutation testing (both variants, full suite) found + # every constructible counterexample (trailing garbage, a bare + # trailing ".", a single-digit exponent) independently rejected by + # the round-trip format comparison a few lines below regardless, + # since format(value, ...) can never reproduce a non-canonical + # tail. See test_parse_c17g_rejects_* for the direct-caller pins. + if not isinstance(text, str) or not re.fullmatch(_C17G_TEXT, text): + raise EvidenceError(f"not a canonical %.6f/%.17g token: {text!r}") + try: + value = float(text) + except ValueError as exc: + raise EvidenceError(f"malformed %.6f/%.17g token: {text!r}") from exc + if (not math.isfinite(value) or + (format(value, ".17g") != text and format(value, ".6f") != text)): + raise EvidenceError( + f"not an exact finite %.6f/%.17g spelling: {text!r}") + return value + + +def is_score_preamble(line): + """Validate one of the two exact stdout records emitted before SCORE.""" + try: + return parse_engine_preamble(line) is not None + except (PreambleError, ValueError): + # engine_evidence's own int() conversions can raise a bare + # ValueError (not its PreambleError) for a numeric field beyond + # Python's 4300-digit int-string conversion limit; a query + # function is not the place to let that escape uncaught. + return False + + +def parse_score_result(line): + """Return (exact_text, value, contlen, greedy) for one complete SCORE line. + + fullmatch (not match/search) is UNREACHABLE-via-the-classifier for + the one input shape that could distinguish them: a `$` anchor (built + into _SCORE_RE's own pattern text) is zero-width and matches just + before a trailing "\\n" as well as at true end-of-string, so + match()/search() would accept a newline-terminated record that + fullmatch correctly rejects (this is exactly the M11-class gap + engine_evidence.py's own fullmatch calls had). classify_score_stdout + and ScoreStdoutClassifier.classify -- the only production callers -- + both strip exactly one trailing newline before calling this function + and both refuse any OTHER embedded "\\n" first (`raw_line.count("\\n") + != 1`), so `line` here is guaranteed newline-free through that path: + the exploit cannot be entered. For a caller reaching this function + directly (as this module's own tests do), the newline-terminated + input is still refused today -- test_direct_call_rejects_trailing_ + newline_record pins it so a future match()/search() weakening does + not go unnoticed on the direct-call path either. + """ + match = _SCORE_RE.fullmatch(line) + if not match: + raise EvidenceError(f"not an exact SCORE record: {line!r}") + exact, contlen_text, greedy_text = match.groups() + try: + value = parse_c17g(exact) + contlen = int(contlen_text) + greedy = int(greedy_text) + except ValueError as exc: + raise EvidenceError(f"malformed SCORE fields: {line!r}") from exc + if not math.isfinite(value) or value > 0.0: + raise EvidenceError(f"SCORE logprob is not finite/non-positive: {exact}") + if not 1 <= contlen <= _INT32_MAX or greedy not in (0, 1): + raise EvidenceError(f"invalid SCORE metadata: {line!r}") + return exact, value, contlen, greedy + + +def classify_score_stdout(raw_line): + """Accept one exact newline-terminated production stdout record.""" + if (not raw_line.endswith("\n") or raw_line.count("\n") != 1 or + "\r" in raw_line): + raise EvidenceError(f"unterminated/non-canonical stdout record: {raw_line!r}") + line = raw_line[:-1] + if not line: + raise EvidenceError("blank SCORE stdout record") + try: + preamble = parse_engine_preamble(line) + except (PreambleError, ValueError) as exc: + # See ScoreStdoutClassifier.classify's identical clause: a + # numeric preamble field beyond Python's int-string conversion + # limit raises a bare ValueError from engine_evidence, not its + # own PreambleError -- refuse it with a named error here too + # rather than let it escape uncaught. + raise EvidenceError(str(exc)) from exc + if preamble is not None: + return None + return parse_score_result(line) + + +class ScoreStdoutClassifier: + """Own exactly one banner then one load record before SCORE results. + + Beyond the banner/load preamble, no other multiplexed-serve global + record (``PROF``, ``HITS``, ``EMAP``, ...) or stderr-only banner + (``[prefill]``, ``[PIN]``, ``[USAGE]``, ...) can ever reach this + classifier: ``run_score`` never prints them to stdout (verified + against the engine source), so any such line arriving here is + refused by name like any other unrecognized record, not specially + recognized or passed through. + """ + + def __init__(self): + self._state = 0 + + def classify(self, raw_line): + if (not raw_line.endswith("\n") or raw_line.count("\n") != 1 or + "\r" in raw_line): + raise EvidenceError( + f"unterminated/non-canonical stdout record: {raw_line!r}") + line = raw_line[:-1] + if not line: + raise EvidenceError("blank SCORE stdout record") + try: + if self._state == 0: + parse_engine_banner(line) + self._state = 1 + return None + if self._state == 1: + parse_engine_loaded(line) + self._state = 2 + return None + preamble = parse_engine_preamble(line) + except (PreambleError, ValueError) as exc: + # A numeric preamble field beyond Python's 4300-digit + # int-string conversion limit raises a bare ValueError from + # engine_evidence's own int()/float() calls, not its + # PreambleError -- this module's contract is to refuse every + # non-canonical record with a named error, never to let one + # escape uncaught. + raise EvidenceError(str(exc)) from exc + if preamble is not None: + raise EvidenceError(f"duplicate/out-of-order SCORE preamble: {line!r}") + return parse_score_result(line) + + def finish(self): + if self._state != 2: + missing = "engine banner" if self._state == 0 else "engine load record" + raise EvidenceError(f"missing {missing} before SCORE EOF") + + +def completion_error(returncode, completed, expected, continuation_tokens, + stream_error=None): + """Return the reason this run has NOTHING trustworthy to report, or + None if it can report a table (even a partial one). + + Matches dev's own exit-code contract exactly, since callers such as + ``coli bench`` (``sys.exit(subprocess.call(cmd, ...))``) and + ``diag_harness.py`` (which parses this tool's own accuracy table from + a subprocess call) depend on it: dev exits nonzero ONLY when the + engine itself produced nothing at all (nonzero exit AND zero + requests scored); a partial run -- some but not all requests scored, + or the engine exiting nonzero after scoring at least one request -- + still prints the accuracy table over whatever landed and exits 0. + Nothing here checks ``expected``/``continuation_tokens`` against a + denominator: by the time this runs, ``expected`` (the request count) + is always positive (an empty selection is refused before the engine + ever launches), and a positive ``completed`` count always carries a + positive token count by construction. + + ``stream_error`` is the one condition dev's own contract has no + analog for: a genuinely corrupted or self-inconsistent SCORE stream, + which this module's evidence layer can detect and dev's plain + per-line filter cannot. Concretely, as of this module's current + stdout classifier: a foreign, malformed, or duplicated/out-of-order + banner-or-load preamble line; a SCORE line whose numeric token is + not an exact finite %.6f/%.17g spelling (a bare "nan"/"inf" included + -- neither survives either spelling); a completed request's contlen + not matching what that request actually asked for; more SCORE + result lines than requests were sent; or the stream ending before + both preamble records were ever seen. That failure stays fatal + regardless of how many requests completed, because once any one + line fails to parse as a complete, well-formed record, nothing + guarantees the corruption did not already touch earlier-looking-fine + rows too -- unlike the DATA/logprobs serving channel elsewhere in + this project, where a non-finite value on a masked row is expected + and reported, not fatal, this SCORE/benchmark channel has no such + legitimate non-finite case: a nan/inf SCORE value cannot be compared + against a benchmark's gold answer, and reporting an accuracy table + built partly from corrupted numbers is a wrong PASS, which this + module's whole design treats as worse than refusing to report at + all. (This docstring previously cited identity-bound-evidence + conditions -- mixed identity-bound/legacy records, a replayed + digest, an out-of-vocabulary token -- that no longer exist: the + identity-binding mode itself was removed as unsatisfiable by any + current engine build. The reasoning above is unchanged; only the + concrete list of what can still reach this path is corrected.) + """ + if stream_error: + return str(stream_error) + if returncode != 0 and completed == 0: + return f"engine exited {returncode} with zero requests scored" + return None + + +def write_result_row(out_f, req_idx, meta_row, exact_logprob, greedy): + """Write the exact engine token, never a rounded float reconstruction.""" + task, qi, oi, clen, cchars, gold = meta_row + out_f.write(f"{req_idx},{task},{qi},{oi},{clen},{cchars},{gold}," + f"{exact_logprob},{greedy}\n") + + +def prelaunch_incomplete(out_path, reason): + """Refuse vacuous evidence before Popen and durably mark writable output.""" + message = f"EVIDENCE INCOMPLETE before engine launch: {reason}" + print(message, file=sys.stderr) + if out_path: + try: + with open(out_path, "a") as out_f: + out_f.write(f"# INCOMPLETE: 0/0; error={reason}\n") + except OSError as exc: + print(f"cannot mark output {out_path!r} INCOMPLETE: {exc}", + file=sys.stderr) + return 1 + def load_docs(task, data_dir, limit, seed): if task == "smoke": return SMOKE[:limit] if limit else SMOKE @@ -72,14 +360,96 @@ def build_requests(tk, docs_by_task, prefix=""): cl = len(ctx_ids) while cl > 0 and (cl > len(full) or full[:cl] != ctx_ids[:cl]): cl -= 1 cont_ids = full[cl:] - if not cont_ids: # boundary degenere: forza split esplicito - full = ctx_ids + tk.encode(cont).ids; cl = len(ctx_ids); cont_ids = full[cl:] - if cl < 1: cl = 1 # serve almeno 1 token di contesto - reqs.append(f"{cl} {len(full)-cl} " + " ".join(map(str, full))) - meta.append((t, qi, oi, len(full) - cl, max(1, len(cont)), gold)) + if cl < 1 or not cont_ids: # boundary degenere: forza split esplicito + choice_ids = tk.encode(cont).ids + full = ctx_ids + choice_ids + cl = len(ctx_ids) + cont_ids = choice_ids + if cl < 1 or not cont_ids: + raise EvidenceError( + f"{t} question {qi} choice {oi} has no positive " + "context/continuation token denominator") + reqs.append(f"{cl} {len(cont_ids)} " + " ".join(map(str, full))) + meta.append((t, qi, oi, len(cont_ids), max(1, len(cont)), gold)) perq.setdefault((t, qi), []).append(len(meta) - 1) return reqs, meta, perq + +def score_snapshot_vocab(snap): + """Read the engine's independently loaded vocabulary bound from config.""" + path = os.path.join(snap, "config.json") + try: + with open(path, "rb") as source: + source.seek(0, os.SEEK_END) + length = source.tell() + _checked_engine_text_size(length, "SCORE config.json") + source.seek(0) + raw = source.read(_ENGINE_TEXT_MAX_BYTES + 1) + _checked_engine_text_size(len(raw), "SCORE config.json") + config = json.loads(raw.decode("utf-8")) + vocab = config["vocab_size"] + except (OSError, UnicodeDecodeError, ValueError, + KeyError, TypeError) as exc: + # ValueError (not just its json.JSONDecodeError subclass): a + # numeric literal in the JSON beyond Python's 4300-digit + # int-string conversion limit makes json.loads itself raise a + # bare ValueError, not JSONDecodeError -- same error-contract + # class as F3's engine_evidence fix. main()'s only caller + # catches EvidenceError to mark the run INCOMPLETE before ever + # launching the engine; a bare ValueError escaping here instead + # crashes main() outright with no INCOMPLETE marker written. + raise EvidenceError(f"cannot derive SCORE vocabulary from {path}: {exc}") from exc + if type(vocab) is not int or not 1 <= vocab <= 1 << 24: + raise EvidenceError(f"invalid SCORE vocabulary in {path}: {vocab!r}") + return vocab + + +def score_request_wire(requests, vocab): + """Return strict ASCII/LF records, joined bytes, and per-record SHA-256. + + The C SCORE evidence mode hashes the ``getline`` byte span, including LF; + this helper owns the identical byte domain before the temporary file exists. + """ + if type(vocab) is not int or not 1 <= vocab <= 1 << 24: + raise EvidenceError(f"invalid SCORE vocabulary: {vocab!r}") + lines = [] + continuation_tokens = 0 + image_bytes = 0 + for request in requests: + if not isinstance(request, str) or not request or "\n" in request or "\r" in request: + raise EvidenceError(f"request is not one canonical line: {request!r}") + fields = request.split(" ") + if (" ".join(fields) != request or len(fields) < 4 or + any(not re.fullmatch(_UINT_TEXT, field) for field in fields)): + raise EvidenceError(f"request is not canonical integer grammar: {request!r}") + values = [int(field) for field in fields] + ctxlen, contlen = values[:2] + if (not 1 <= ctxlen <= _INT32_MAX or + not 1 <= contlen <= _INT32_MAX - ctxlen): + raise EvidenceError(f"request lengths are invalid: {request!r}") + total = ctxlen + contlen + tokens = values[2:] + if len(tokens) != total: + raise EvidenceError(f"request token count is invalid: {request!r}") + if any(token >= vocab for token in tokens): + raise EvidenceError(f"request token is outside vocabulary: {request!r}") + if continuation_tokens > (1 << 63) - 1 - contlen: + raise EvidenceError("SCORE continuation denominator exceeds int64") + continuation_tokens += contlen + try: + line = (request + "\n").encode("ascii") + except UnicodeEncodeError as exc: + raise EvidenceError( + f"request is not canonical ASCII: {request!r}") from exc + image_bytes = _checked_engine_text_size( + image_bytes + len(line), "SCORE request image") + lines.append(line) + if not lines or continuation_tokens <= 0 or len(lines) > _INT32_MAX: + raise EvidenceError("SCORE request image has no positive denominator") + frozen = tuple(lines) + return (frozen, b"".join(frozen), + tuple(hashlib.sha256(line).hexdigest() for line in frozen)) + def score_accuracy(tasks, meta, perq, lp): print(f"\n{'task':<18} {'n':>4} {'acc':>7} {'acc_norm':>9}") overall = [] @@ -123,23 +493,44 @@ def main(): score_accuracy(["t"], meta, perq, lp) print("selftest OK" if True else ""); return + tasks = [t.strip() for t in a.tasks.split(",") if t.strip()] + if not tasks: + return prelaunch_incomplete(a.out,"no benchmark tasks selected") + from tokenizers import Tokenizer tk = Tokenizer.from_file(os.path.join(a.snap, "tokenizer.json")) - tasks = [t.strip() for t in a.tasks.split(",") if t.strip()] docs_by_task = {t: load_docs(t, a.data, a.limit, a.seed) for t in tasks} for t, d in docs_by_task.items(): print(f"[{t}] {len(d)} questions", file=sys.stderr) - reqs, meta, perq = build_requests(tk, docs_by_task, detect_prefix(a.snap)) + try: + reqs, meta, perq = build_requests( + tk, docs_by_task, detect_prefix(a.snap)) + except EvidenceError as exc: + return prelaunch_incomplete(a.out, str(exc)) print(f"total requests: {len(reqs)} (answer options)", file=sys.stderr) + if not reqs: + return prelaunch_incomplete(a.out,"selected tasks produced zero SCORE requests") if a.dry: + # Matches dev exactly: --dry stops right after request + # construction, before the vocabulary lookup below -- it never + # needed config.json's vocab_size (a plumbing check has no + # engine, and therefore no vocabulary, to bind requests against). for r in reqs[:3]: print(" example request:", r[:80], "...", file=sys.stderr) print("DRY: request construction and tokenization passed. Engine was not run.", file=sys.stderr); return + try: + score_vocab = score_snapshot_vocab(a.snap) + _, request_payload, _ = score_request_wire( + reqs, score_vocab) + except EvidenceError as exc: + return prelaunch_incomplete(a.out, str(exc)) # mkstemp (non mktemp): crea il file atomicamente con permessi 0600, niente # race TOCTOU/symlink su una tmp dir condivisa (CWE-377). fd, req_path = tempfile.mkstemp(suffix=".txt") - with os.fdopen(fd, "w") as f: - f.write("\n".join(reqs) + "\n") + with os.fdopen(fd, "wb") as f: + written = f.write(request_payload) + if written != len(request_payload): + raise EvidenceError("short write while freezing SCORE requests") env = dict(os.environ, SNAP=a.snap, SCORE=req_path) if a.ram: env["RAM_GB"] = str(a.ram) cmd = [a.glm, str(a.cap)] + a.bits.split() @@ -155,55 +546,131 @@ def main(): out_f.write("req_idx,task,qi,oi,contlen,contchars,gold,logprob,greedy\n") out_f.flush() t0 = time.time() - proc = subprocess.Popen(cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE, - text=True, bufsize=1) # line-buffered - lp = [None] * len(reqs) - n_done = 0 - # Drain stderr (engine progress lines) to console live on a background thread - # so the [score N req] heartbeat is visible while stdout is consumed below. - def _drain_stderr(): - for line in proc.stderr: - print(f" [engine] {line.rstrip()}", file=sys.stderr) - threading.Thread(target=_drain_stderr, daemon=True).start() - for line in proc.stdout: - line = line.strip() - if not line or line[0] not in "-0123456789": continue - parts = line.split() - if n_done >= len(reqs): break - try: logprob = float(parts[0]) - except (ValueError, IndexError): continue - lp[n_done] = logprob - greedy = parts[2] if len(parts) > 2 else "?" - t, qi, oi, clen, cchars, gold = meta[n_done] + proc = None + previous_sigterm = None + + def _on_sigterm(signum, frame): + raise ChildTerminateRequested(f"received signal {signum}") + + try: + # A SIGTERM (or Ctrl+C's SIGINT, which Python already converts to + # KeyboardInterrupt on its own) must not leave the engine child + # running as an orphan/zombie -- the handler below converts + # SIGTERM into the same exception path, so both unwind through + # the identical `finally` cleanup that terminates the child. + previous_sigterm = signal.signal(signal.SIGTERM, _on_sigterm) + proc = subprocess.Popen(cmd, env=env, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, bufsize=1) # line-buffered + lp = [None] * len(reqs) + n_done = 0 + continuation_tokens = 0 + stream_error = None + stdout_classifier = ScoreStdoutClassifier() + # Drain stderr (engine progress lines) to console live on a background thread + # so the [score N req] heartbeat is visible while stdout is consumed below. + def _drain_stderr(): + for line in proc.stderr: + print(f" [engine] {line.rstrip()}", file=sys.stderr) + threading.Thread(target=_drain_stderr, daemon=True).start() + for raw_line in proc.stdout: + if stream_error: + continue # drain fully so the child cannot block + try: + result = stdout_classifier.classify(raw_line) + except EvidenceError as exc: + stream_error = exc + continue + if result is None: + continue + if n_done >= len(reqs): + stream_error = EvidenceError("engine emitted extra SCORE result lines") + continue + try: + exact, logprob, contlen, greedy = result + if contlen != meta[n_done][3]: + raise EvidenceError( + f"request {n_done} contlen {contlen} != expected {meta[n_done][3]}") + except EvidenceError as exc: + stream_error = exc + continue + lp[n_done] = logprob + continuation_tokens += contlen + t, qi, oi, clen, cchars, gold = meta[n_done] + if out_f: + write_result_row(out_f,n_done,meta[n_done],exact,greedy) + out_f.flush() + n_done += 1 + if n_done % 5 == 0 or n_done == len(reqs): + elapsed = time.time() - t0 + rate = n_done / elapsed if elapsed > 0 else 0 + eta = (len(reqs) - n_done) / rate if rate > 0 else 0 + print(f"[progress] {n_done}/{len(reqs)} requests scored | {elapsed:.0f}s elapsed | " + f"{rate:.2f} req/s | ETA {eta:.0f}s | last: {t} q{qi} opt{oi} lp={logprob:.3f}", + file=sys.stderr) + if not stream_error: + try: + stdout_classifier.finish() + except EvidenceError as exc: + stream_error = exc + proc.wait() + elapsed = time.time() - t0 + # Fatal only in the two cases dev's own contract (and this + # module's own evidence layer) recognize -- see completion_error's + # docstring. Everything else, including a partial request count, + # still reports the accuracy table and exits 0, matching dev. + fatal = completion_error( + proc.returncode,n_done,len(reqs),continuation_tokens,stream_error) + partial = n_done != len(reqs) if out_f: - out_f.write(f"{n_done},{t},{qi},{oi},{clen},{cchars},{gold},{logprob:.6f},{greedy}\n") - out_f.flush() - n_done += 1 - if n_done % 5 == 0 or n_done == len(reqs): - elapsed = time.time() - t0 - rate = n_done / elapsed if elapsed > 0 else 0 - eta = (len(reqs) - n_done) / rate if rate > 0 else 0 - print(f"[progress] {n_done}/{len(reqs)} requests scored | {elapsed:.0f}s elapsed | " - f"{rate:.2f} req/s | ETA {eta:.0f}s | last: {t} q{qi} opt{oi} lp={logprob:.3f}", - file=sys.stderr) - proc.wait() - elapsed = time.time() - t0 - if out_f: - out_f.write(f"# finished: {n_done}/{len(reqs)} in {elapsed:.0f}s, exit={proc.returncode}\n") - out_f.close() - if proc.returncode != 0 and n_done == 0: - print(f"ENGINE ERROR (exit {proc.returncode})", file=sys.stderr); sys.exit(1) - if n_done != len(reqs): - print(f"WARNING: only {n_done}/{len(reqs)} requests scored (engine exited {proc.returncode}); " - f"scoring partial results.", file=sys.stderr) - # Fill any unscored slots with -inf so argmax never picks them - for i in range(len(lp)): - if lp[i] is None: lp[i] = float("-inf") - print(f"(engine: {elapsed:.0f}s, {n_done}/{len(reqs)} scored, exit {proc.returncode})", file=sys.stderr) - score_accuracy(tasks, meta, perq, lp) - print("\nNOTE: compare acc_norm with GLM-5.2's PUBLISHED model-card score. A close result" - "\n indicates that int4 quantization preserved quality. (Fill REFERENCE in tools/eval_glm.py.)") - os.remove(req_path) + if fatal: + out_f.write(f"# INCOMPLETE: {n_done}/{len(reqs)} in {elapsed:.0f}s, " + f"tokens={continuation_tokens}, exit={proc.returncode}; " + f"error={fatal}\n") + else: + out_f.write(f"# finished: {n_done}/{len(reqs)} in {elapsed:.0f}s, " + f"tokens={continuation_tokens}, exit={proc.returncode}\n") + if partial: + # Additive: the run still finished (exit 0, full + # table below) -- this line only ANNOUNCES that fewer + # than the full request count landed, it does not + # replace the "# finished" line or change the exit + # code dev's own consumers depend on. + out_f.write(f"# INCOMPLETE: {n_done}/{len(reqs)} requests " + f"scored; engine exit={proc.returncode}\n") + out_f.close(); out_f=None + if fatal: + print(f"EVIDENCE INCOMPLETE: {fatal}", file=sys.stderr) + return 1 + if partial: + # Same wording and the same exit-0 contract dev used: a + # partial run is a WARNING, not a failure. + print(f"WARNING: only {n_done}/{len(reqs)} requests scored " + f"(engine exited {proc.returncode}); scoring partial " + "results.", file=sys.stderr) + # Fill any unscored slots with -inf so argmax never picks them + # (dev's own fallback for a partial run). + for i in range(len(lp)): + if lp[i] is None: lp[i] = float("-inf") + print(f"(engine: {elapsed:.0f}s, {n_done}/{len(reqs)} scored, " + f"{continuation_tokens} continuation tokens, " + f"exit {proc.returncode})", file=sys.stderr) + score_accuracy(tasks, meta, perq, lp) + print("\nNOTE: compare acc_norm with GLM-5.2's PUBLISHED model-card score. A close result" + "\n indicates that int4 quantization preserved quality. (Fill REFERENCE in tools/eval_glm.py.)") + return 0 + finally: + if previous_sigterm is not None: + signal.signal(signal.SIGTERM, previous_sigterm) + if proc is not None and proc.poll() is None: + # A mid-run exception or termination signal must never leave + # the engine child running: no zombie, no orphan. + proc.terminate() + proc.wait() + if out_f: + out_f.write("# INCOMPLETE: evaluator terminated before a complete denominator\n") + out_f.close() + try: os.remove(req_path) + except FileNotFoundError: pass if __name__ == "__main__": - main() + sys.exit(main() or 0) diff --git a/c/tools/gguf_dequant.py b/c/tools/gguf_dequant.py new file mode 100644 index 000000000..d71f84ac5 --- /dev/null +++ b/c/tools/gguf_dequant.py @@ -0,0 +1,275 @@ +#!/usr/bin/env python3 +"""Dequantization of GGUF/ggml tensor blocks to float32, pure numpy. + +Ported block-for-block from ggml-org/ggml src/ggml-quants.c (dequantize_row_*), +using the block layouts in src/ggml-common.h. This module owns the numerical +decode used by the converter profiles; gguf_reader.py owns structure only. + +Only the types listed in _DEQUANTIZERS are implemented. dequantize() raises +GGUFError for any other type, so a profile never silently mis-decodes bytes. +Correctness is pinned by tests/test_gguf_dequant.py, which compares against +llama.cpp's gguf-py reference on random blocks and against committed vectors. +""" + +import numpy as np + +from gguf_reader import ( + GGML_TYPE_BF16, + GGML_TYPE_F16, + GGML_TYPE_F32, + GGML_TYPE_Q2_K, + GGML_TYPE_Q3_K, + GGML_TYPE_Q4_K, + GGML_TYPE_Q5_K, + GGML_TYPE_Q6_K, + GGML_TYPE_Q8_0, + GGUFError, + ggml_block_spec, + ggml_type_name, +) + +QK_K = 256 +_MASK_KMASK1 = 0x03030303 +_MASK_KMASK2 = 0x0F0F0F0F + + +def _f16_to_f32(pair_bytes): + """(n, 2) uint8 or (n,) uint16 -> float32.""" + u16 = pair_bytes.view("> 6) << 4) + minimum[:, 4:8] = (sc_bytes[:, 8:12] >> 4) | ((sc_bytes[:, 4:8] >> 6) << 4) + return scale, minimum + + +def dequantize_q4_K(blocks): + # Q4_K (ggml-common.h block_q4_K, 144 bytes/256 elems): + # d/dmin (fp16), scales[12] (sixteen 6-bit scale/min pairs), + # qs[128] 4-bit packed quants. 8 sub-blocks of 32 values each. + num_blocks = blocks.shape[0] + packed_q = blocks[:, 16:144] + d = _f16_to_f32(blocks[:, 0:2]) + dmin = _f16_to_f32(blocks[:, 2:4]) + scale6, min6 = _unpack_scale_min_k4(blocks[:, 4:16]) + out = np.empty((num_blocks, QK_K), dtype=np.float32) + for group in range(4): + d1 = d * scale6[:, 2 * group] + m1 = dmin * min6[:, 2 * group] + d2 = d * scale6[:, 2 * group + 1] + m2 = dmin * min6[:, 2 * group + 1] + col = 32 * group + np.arange(32) + lo = 64 * group + out[:, lo:lo + 32] = d1[:, None] * (packed_q[:, col] & 0xF).astype(np.float32) - m1[:, None] + out[:, lo + 32:lo + 64] = d2[:, None] * (packed_q[:, col] >> 4).astype(np.float32) - m2[:, None] + return out + + +def dequantize_q5_K(blocks): + # Q5_K (ggml-common.h block_q5_K, 176 bytes/256 elems): + # d/dmin (fp16), scales[12] (6-bit pairs), qh[32] high bits, qs[128] 4-bit. + # Same 8x32 structure as Q4_K plus a per-plane high bit that adds 16. + num_blocks = blocks.shape[0] + ql = blocks[:, 48:176] + qh = blocks[:, 16:48] + d = _f16_to_f32(blocks[:, 0:2]) + dmin = _f16_to_f32(blocks[:, 2:4]) + scale6, min6 = _unpack_scale_min_k4(blocks[:, 4:16]) + out = np.empty((num_blocks, QK_K), dtype=np.float32) + for group in range(4): + d1 = d * scale6[:, 2 * group] + m1 = dmin * min6[:, 2 * group] + d2 = d * scale6[:, 2 * group + 1] + m2 = dmin * min6[:, 2 * group + 1] + col = 32 * group + np.arange(32) + hi_low = ((qh >> (2 * group)) & 1).astype(np.float32) + hi_high = ((qh >> (2 * group + 1)) & 1).astype(np.float32) + low = (ql[:, col] & 0xF).astype(np.float32) + high = (ql[:, col] >> 4).astype(np.float32) + lo = 64 * group + out[:, lo:lo + 32] = d1[:, None] * (low + 16.0 * hi_low) - m1[:, None] + out[:, lo + 32:lo + 64] = d2[:, None] * (high + 16.0 * hi_high) - m2[:, None] + return out + + +def dequantize_q8_0(blocks): + # Q8_0 (ggml-common.h block_q8_0, 34 bytes/32 elems): + # fp16 scale, then 32 signed int8 values; every value is q * d. + d = _f16_to_f32(blocks[:, 0:2]) + packed_q = blocks[:, 2:34].view(np.int8) + return d[:, None] * packed_q.astype(np.float32) + + +def dequantize_q2_K(blocks): + # Q2_K (ggml-common.h block_q2_K, 84 bytes/256 elems): + # scales[16] (per-16 low nibble=scale, high nibble=min, 4-bit), + # qs[64] (2-bit packed quants), then fp16 block scale (d) and min (dmin). + num_blocks = blocks.shape[0] + scales = blocks[:, 0:16].astype(np.int32) + packed_q = blocks[:, 16:80] + block_scale = _f16_to_f32(blocks[:, 80:82]) + block_min = _f16_to_f32(blocks[:, 82:84]) + out = np.empty((num_blocks, QK_K), dtype=np.float32) + is_ = 0 + for half in range(2): + base = half * 32 + for j in range(4): + shift = j * 2 + for sub in range(2): + sc = scales[:, is_] + is_ += 1 + group_scale = block_scale * (sc & 0xF) + group_min = block_min * (sc >> 4) + cols = base + sub * 16 + np.arange(16) + qv = ((packed_q[:, cols] >> shift) & 3).astype(np.float32) + lo = half * 128 + j * 32 + sub * 16 + out[:, lo:lo + 16] = group_scale[:, None] * qv - group_min[:, None] + return out + + +def dequantize_q3_K(blocks): + # Q3_K (ggml-common.h block_q3_K, 110 bytes/256 elems): + # hmask[32] high bits, qs[64] low 2 bits, scales[12] (sixteen 6-bit scales + # packed into 12 bytes, reassembled by the kmask shuffle below), fp16 scale. + num_blocks = blocks.shape[0] + hmask = blocks[:, 0:32] + packed_q = blocks[:, 32:96] + scale_bytes = blocks[:, 96:108] + block_scale = _f16_to_f32(blocks[:, 108:110]) + + # Sixteen 6-bit scales live packed in 12 bytes: three 32-bit words, then a + # bit-shuffle reassembles them (kmask1/kmask2 from ggml-quants.c). + a0 = (scale_bytes[:, 0].astype(np.uint32) | (scale_bytes[:, 1].astype(np.uint32) << 8) + | (scale_bytes[:, 2].astype(np.uint32) << 16) | (scale_bytes[:, 3].astype(np.uint32) << 24)) + a1 = (scale_bytes[:, 4].astype(np.uint32) | (scale_bytes[:, 5].astype(np.uint32) << 8) + | (scale_bytes[:, 6].astype(np.uint32) << 16) | (scale_bytes[:, 7].astype(np.uint32) << 24)) + a2 = (scale_bytes[:, 8].astype(np.uint32) | (scale_bytes[:, 9].astype(np.uint32) << 8) + | (scale_bytes[:, 10].astype(np.uint32) << 16) | (scale_bytes[:, 11].astype(np.uint32) << 24)) + tmp = a2 + a2n = ((a0 >> 4) & _MASK_KMASK2) | (((tmp >> 4) & _MASK_KMASK1) << 4) + a3n = ((a1 >> 4) & _MASK_KMASK2) | (((tmp >> 6) & _MASK_KMASK1) << 4) + a0n = (a0 & _MASK_KMASK2) | (((tmp >> 0) & _MASK_KMASK1) << 4) + a1n = (a1 & _MASK_KMASK2) | (((tmp >> 2) & _MASK_KMASK1) << 4) + + row_scales = np.empty((num_blocks, 16), dtype=np.int8) + for i, word in enumerate((a0n, a1n, a2n, a3n)): + row_scales[:, i * 4:(i + 1) * 4] = word.astype("> shift) & 3).astype(np.float32) + high_bit = ((hmask[:, hcols] & m) != 0).astype(np.float32) + # low 2-bit code, minus 4 when the high bit of the mask is clear + centered_q = qv - 4.0 * (1.0 - high_bit) + lo = half * 128 + j * 32 + sub * 16 + out[:, lo:lo + 16] = group_scale[:, None] * centered_q + return out + + +def dequantize_q6_K(blocks): + # Q6_K (ggml-common.h block_q6_K, 210 bytes/256 elems): + # ql[128] low 4 bits, qh[64] upper 2 bits, scales[16] int8 per-16 scales, + # fp16 block scale. Two 128-elem halves; codes are (low | high<<4) - 32. + num_blocks = blocks.shape[0] + packed_lo = blocks[:, 0:128] + packed_hi = blocks[:, 128:192] + row_scales = blocks[:, 192:208].view(np.int8) + block_scale = _f16_to_f32(blocks[:, 208:210]) + out = np.empty((num_blocks, QK_K), dtype=np.float32) + for half in range(2): + lo_base = half * 64 + hi_base = half * 32 + scale_base = half * 8 + l = np.arange(32) + group = l // 16 + code0 = ((packed_lo[:, lo_base + l] & 0xF).astype(np.int32) + | (((packed_hi[:, hi_base + l] >> 0) & 3).astype(np.int32) << 4)) - 32 + code1 = ((packed_lo[:, lo_base + l + 32] & 0xF).astype(np.int32) + | (((packed_hi[:, hi_base + l] >> 2) & 3).astype(np.int32) << 4)) - 32 + code2 = ((packed_lo[:, lo_base + l] >> 4).astype(np.int32) + | (((packed_hi[:, hi_base + l] >> 4) & 3).astype(np.int32) << 4)) - 32 + code3 = ((packed_lo[:, lo_base + l + 32] >> 4).astype(np.int32) + | (((packed_hi[:, hi_base + l] >> 6) & 3).astype(np.int32) << 4)) - 32 + scale0 = row_scales[:, scale_base + group + 0].astype(np.float32) + scale1 = row_scales[:, scale_base + group + 2].astype(np.float32) + scale2 = row_scales[:, scale_base + group + 4].astype(np.float32) + scale3 = row_scales[:, scale_base + group + 6].astype(np.float32) + base = half * 128 + out[:, base + l] = block_scale[:, None] * scale0 * code0.astype(np.float32) + out[:, base + l + 32] = block_scale[:, None] * scale1 * code1.astype(np.float32) + out[:, base + l + 64] = block_scale[:, None] * scale2 * code2.astype(np.float32) + out[:, base + l + 96] = block_scale[:, None] * scale3 * code3.astype(np.float32) + return out + + +_DEQUANTIZERS = { + GGML_TYPE_Q2_K: dequantize_q2_K, + GGML_TYPE_Q3_K: dequantize_q3_K, + GGML_TYPE_Q4_K: dequantize_q4_K, + GGML_TYPE_Q5_K: dequantize_q5_K, + GGML_TYPE_Q6_K: dequantize_q6_K, + GGML_TYPE_Q8_0: dequantize_q8_0, +} + + +def dequantize(raw, ggml_type, numel): + """Decode a tensor's stored bytes to a flat float32 array (GGUF element order).""" + if ggml_type in _DEQUANTIZERS: + block_elems, block_bytes = ggml_block_spec(ggml_type) + if numel % block_elems != 0: + raise GGUFError( + "dequant %s: numel %d is not a multiple of block %d" + % (ggml_type_name(ggml_type), numel, block_elems)) + nblocks = numel // block_elems + data = np.frombuffer(raw, dtype=np.uint8) + if data.size != nblocks * block_bytes: + raise GGUFError( + "dequant %s: %d bytes for %d blocks of %d, expected %d" + % (ggml_type_name(ggml_type), data.size, nblocks, block_bytes, + nblocks * block_bytes)) + blocks = data.reshape(nblocks, block_bytes) + return _DEQUANTIZERS[ggml_type](blocks).reshape(-1) + + element_size = {GGML_TYPE_F32: 4, GGML_TYPE_F16: 2, GGML_TYPE_BF16: 2}.get(ggml_type) + if element_size is not None: + expected = numel * element_size + if len(raw) != expected: + raise GGUFError( + "dequant %s: %d bytes for %d elements, expected %d" + % (ggml_type_name(ggml_type), len(raw), numel, expected)) + if ggml_type == GGML_TYPE_F32: + return np.frombuffer(raw, dtype=" colibri container layout. + +This module owns everything OLMoE-specific about the conversion and nothing +about file I/O or CLI: + + * name mapping from GGUF (llama.cpp) tensor names to the names c/olmoe.c + loads: model.embed_tokens.weight, model.layers.N.self_attn.q_proj.weight, + and per-expert model.layers.N.mlp.experts.E.{merged_weight,qs}; + * dequantized GGUF order -> logical HuggingFace order (GGUF stores dims + reversed); + * row-wise int8 quantization identical to olmoe.c's decoder contract + W[o,i] ~= q[o,i]*scale[o] (same math as tools/convert_olmoe_merged.py); + * per-expert gate|up|down merge into one merged_weight + one .qs; + * config.json reconstruction from GGUF olmoe.* metadata. + +It is imported by convert_gguf_to_olmoe.py (CLI) and unit-tested directly. +""" + +import re + +import numpy as np + +EXPERT_RE = re.compile(r"^blk\.(\d+)\.ffn_(gate|up|down)_exps\.weight$") + +DENSE_SIMPLE = { + "token_embd.weight": "model.embed_tokens.weight", + "output.weight": "lm_head.weight", + "output_norm.weight": "model.norm.weight", +} + +DENSE_LAYERED_RE = re.compile( + r"^blk\.(\d+)\.(attn_norm|ffn_norm|attn_q|attn_k|attn_v|attn_output|" + r"attn_q_norm|attn_k_norm|ffn_gate_inp)\.weight$") + +LAYER_SUFFIX = { + "attn_norm": "input_layernorm.weight", + "ffn_norm": "post_attention_layernorm.weight", + "attn_q": "self_attn.q_proj.weight", + "attn_k": "self_attn.k_proj.weight", + "attn_v": "self_attn.v_proj.weight", + "attn_output": "self_attn.o_proj.weight", + "attn_q_norm": "self_attn.q_norm.weight", + "attn_k_norm": "self_attn.k_norm.weight", + "ffn_gate_inp": "mlp.gate.weight", +} + +CONFIG_KEYS = { + "olmoe.embedding_length": "hidden_size", + "olmoe.block_count": "num_hidden_layers", + "olmoe.attention.head_count": "num_attention_heads", + "olmoe.attention.head_count_kv": "num_key_value_heads", + "olmoe.expert_count": "num_experts", + "olmoe.expert_used_count": "num_experts_per_tok", + "olmoe.feed_forward_length": "intermediate_size", + "olmoe.rope.freq_base": "rope_theta", + "olmoe.attention.layer_norm_rms_epsilon": "rms_norm_eps", +} + + +def parse_expert(name): + """(layer, kind) for an expert tensor name, or None.""" + match = EXPERT_RE.match(name) + if not match: + return None + return int(match.group(1)), match.group(2) + + +def dense_target(name): + """colibri name for a GGUF dense tensor, or None if not a known dense tensor.""" + if name in DENSE_SIMPLE: + return DENSE_SIMPLE[name] + match = DENSE_LAYERED_RE.match(name) + if match: + layer, kind = int(match.group(1)), match.group(2) + return "model.layers.%d.%s" % (layer, LAYER_SUFFIX[kind]) + return None + + +def to_logical(flat, gguf_dims): + """Reshape GGUF-order flat values into the logical (HuggingFace) shape. + + GGUF stores the source tensor's bytes unchanged and only reverses the dims + labels, so the flat data is already C-order of the original shape: reshape + to reversed(gguf_dims) reproduces it. No transpose is involved. + """ + return np.ascontiguousarray(flat.reshape(tuple(reversed(gguf_dims)))) + + +def restore_rope_layout(weights, n_head, n_kv_head=None): + """Undo llama.cpp's RoPE permutation of q_proj/k_proj weights. + + Only llama.cpp's *dense* OLMo converter does this: conversion/olmo.py's + OlmoModel (OlmoForCausalLM) permutes q_proj and k_proj exactly like Llama: + forward: w.reshape(n, 2, out // n // 2, *rest).swapaxes(1, 2).reshape(w.shape) + with n = n_kv_head when it differs from n_head (k_proj), else n_head. That + is NOT an involution unless out // n // 2 == 2, so the inverse is the axis + transpose of the (m, 2) decomposition: + inverse: w.reshape(n, m, 2, *rest).swapaxes(1, 2).reshape(w.shape) + with m = out // n // 2. + + The OLMoE converter is a different class, OlmoeModel (OlmoeForCausalLM), and + does NOT permute: llama.cpp runs LLM_ARCH_OLMOE with NEOX RoPE -- pairs of + head values offset by head_dim/2 -- the same layout as HuggingFace's + rotate_half and c/olmoe.c's rope_head. The official OLMoE GGUF is therefore + already in the HF layout; this function is only for a GGUF that a permuting + converter produced. + """ + n = n_kv_head if (n_kv_head and n_head != n_kv_head) else n_head + w = np.asarray(weights) + if w.ndim < 1 or n < 1 or w.shape[0] % n != 0 or (w.shape[0] // n) % 2 != 0: + raise ValueError("cannot restore RoPE layout for shape %r with %d heads" + % (w.shape, n)) + m = w.shape[0] // n // 2 + rest = w.shape[1:] + return np.ascontiguousarray( + w.reshape((n, m, 2) + rest).swapaxes(1, 2).reshape(w.shape)) + + +def quantize_row(weights): + """Row-wise int8 quantization, identical math to c/olmoe.c's decoder. + + olmoe.c computes y[o] = scale[o] * sum_i x[i]*q[o,i], so the stored bytes + must satisfy W[o,i] ~= q[o,i]*scale[o] with scale[o] = row_absmax/127. + Rounding is numpy's rint (round-half-to-even), matching convert_olmoe_merged.py. + """ + w = np.asarray(weights, dtype=np.float32) + if w.ndim != 2: + raise ValueError("quantize_row expects a 2-D weight, got shape %r" % (w.shape,)) + row_max = np.maximum(np.max(np.abs(w), axis=1, keepdims=True), np.float32(1e-12)) + scales = (row_max / np.float32(127.0)).astype(np.float32) + quantized = np.clip(np.rint(w / scales), -128.0, 127.0).astype(np.int8) + return quantized, scales.reshape(-1) + + +def merge_expert(gate, up, down): + """Quantize and merge one expert's gate/up/down into colibri's layout. + + merged_weight = int8(gate | up | down) flattened; + qs = f32(gate rows | up rows | down rows). + gate/up: [inter, hidden]; down: [hidden, inter]. + """ + q_gate, s_gate = quantize_row(gate) + q_up, s_up = quantize_row(up) + q_down, s_down = quantize_row(down) + merged = np.concatenate((q_gate.reshape(-1), q_up.reshape(-1), q_down.reshape(-1))) + scales = np.concatenate((s_gate, s_up, s_down)).astype(np.float32) + return merged, scales + + +def expert_matrix_shape(gguf_dims): + """HF shape of one expert's matrix inside a fused [d0, d1, E] expert tensor.""" + return (gguf_dims[1], gguf_dims[0]) + + +def expert_bytes_per_slice(gguf_dims, ggml_type): + """Bytes of one expert inside a fused expert tensor (contiguous slice).""" + from gguf_reader import ggml_block_spec + block_elems, block_bytes = ggml_block_spec(ggml_type) + if gguf_dims[0] % block_elems != 0: + raise ValueError( + "expert tensor dims[0]=%d is not a multiple of the block %d" + % (gguf_dims[0], block_elems)) + row_blocks = gguf_dims[0] // block_elems + return row_blocks * block_bytes * gguf_dims[1] + + +def _token_type_of(meta, index): + entries = meta.get("tokenizer.ggml.token_type") + if not entries or index >= len(entries): + return 1 + return int(entries[index]) + + +def build_tokenizer(meta): + """Reconstruct an HF-style tokenizer.json from GGUF tokenizer metadata. + + Feed shape matches what c/tok.h loads for GPT-2/byte-level BPE: + model.vocab (token string -> id), model.merges (space-separated "left + right" entries), added_tokens (id/content/special), pre_tokenizer that + does not select the o200k/Kimi maybe-flags, plus bos/eos/pad and the chat + template. token_type 3=CONTROL (special) and 4=USER_DEFINED become added + tokens; 5=UNUSED and 6=BYTE stay plain vocab entries. + """ + tokens = meta.get("tokenizer.ggml.tokens") + if not tokens: + raise ValueError("GGUF metadata missing tokenizer.ggml.tokens") + merges = meta.get("tokenizer.ggml.merges") + + vocab = {} + added = [] + # ggml token_type: 1=NORMAL stays plain BPE vocab; 3=CONTROL and + # 4=USER_DEFINED become added tokens (atomic, literal); 5=UNUSED and 6=BYTE + # stay plain vocab entries (not atomic specials). + for index, token in enumerate(tokens): + vocab[token] = index + kind = _token_type_of(meta, index) + if kind in (3, 4): + added.append({"id": index, "content": token, "special": kind == 3}) + + model = {"type": "BPE", "vocab": vocab} + if merges: + model["merges"] = list(merges) + + tokenizer = { + "add_bos_token": bool(meta.get("tokenizer.ggml.add_bos_token")), + "add_eos_token": bool(meta.get("tokenizer.ggml.add_eos_token")), + "model": model, + "pre_tokenizer": {"type": "ByteLevel", "add_prefix_space": False}, + "added_tokens": added, + } + for gguf_key, hf_key in ( + ("tokenizer.ggml.bos_token_id", "bos_token"), + ("tokenizer.ggml.eos_token_id", "eos_token"), + ("tokenizer.ggml.padding_token_id", "pad_token")): + token_id = meta.get(gguf_key) + if token_id is not None and int(token_id) < len(tokens): + tokenizer[hf_key] = tokens[int(token_id)] + if "tokenizer.chat_template" in meta: + tokenizer["chat_template"] = meta["tokenizer.chat_template"] + return tokenizer + + +def build_config(meta, norm_topk_prob=False, extra=None): + """Reconstruct the config.json c/olmoe.c loads from GGUF metadata. + + norm_topk_prob is NOT in GGUF metadata; the official OLMoE checkpoints use + false (transformers' OlmoeConfig default), so that is the default here. Use + the CLI flag to override for a checkpoint that trained with it on. + """ + config = {} + for gguf_key, cfg_key in CONFIG_KEYS.items(): + if gguf_key not in meta: + raise ValueError("GGUF metadata missing %s" % gguf_key) + config[cfg_key] = meta[gguf_key] + tokens = meta.get("tokenizer.ggml.tokens") + if not tokens: + raise ValueError("GGUF metadata missing tokenizer.ggml.tokens (needed for vocab_size)") + config["vocab_size"] = len(tokens) + config["norm_topk_prob"] = bool(norm_topk_prob) + config["model_type"] = "olmoe" + config["architectures"] = ["OlmoeForCausalLM"] + if extra: + config.update(extra) + return config diff --git a/c/tools/gguf_reader.py b/c/tools/gguf_reader.py new file mode 100644 index 000000000..ccf4953dd --- /dev/null +++ b/c/tools/gguf_reader.py @@ -0,0 +1,374 @@ +#!/usr/bin/env python3 +"""GGUF reader, pure stdlib. + +Parses the binary header, the metadata key/value store and the tensor index of +a GGUF file (https://github.com/ggml-org/ggml/blob/master/docs/gguf.md), +computes each tensor's absolute byte range and reads raw tensor bytes on +demand. No third-party dependency: converter profiles import this module and +own the per-ggml-type decoding. + +Tensor shapes are reported exactly as stored on disk: GGUF writers store the +PyTorch/HuggingFace shape reversed (the fastest-varying dimension first), so a +checkpoint weight logged as [out, in] appears here as dims [in, out]. + +Only the ggml types listed in GGML_TYPE_RAW and GGML_TYPE_BLOCK are understood. +A tensor of any other type makes parsing refuse loudly rather than mis-size +it; converter profiles extend coverage by adding entries to those tables. +""" + +import os +import struct + +GGUF_MAGIC = b"GGUF" +DEFAULT_ALIGNMENT = 32 +SUPPORTED_VERSIONS = (1, 2, 3) + +MAX_METADATA_ITEMS = 1 << 22 +MAX_METADATA_DEPTH = 64 +MAX_TENSORS = 1 << 24 + +GGML_TYPE_F32 = 0 +GGML_TYPE_F16 = 1 +GGML_TYPE_Q4_0 = 2 +GGML_TYPE_Q4_1 = 3 +GGML_TYPE_Q5_0 = 6 +GGML_TYPE_Q5_1 = 7 +GGML_TYPE_Q8_0 = 8 +GGML_TYPE_Q8_1 = 9 +GGML_TYPE_Q2_K = 10 +GGML_TYPE_Q3_K = 11 +GGML_TYPE_Q4_K = 12 +GGML_TYPE_Q5_K = 13 +GGML_TYPE_Q6_K = 14 +GGML_TYPE_Q8_K = 15 +GGML_TYPE_IQ2_XXS = 16 +GGML_TYPE_IQ2_XS = 17 +GGML_TYPE_IQ3_XXS = 18 +GGML_TYPE_IQ1_S = 19 +GGML_TYPE_IQ4_NL = 20 +GGML_TYPE_IQ3_S = 21 +GGML_TYPE_IQ2_S = 22 +GGML_TYPE_IQ4_XS = 23 +GGML_TYPE_I8 = 24 +GGML_TYPE_I16 = 25 +GGML_TYPE_I32 = 26 +GGML_TYPE_I64 = 27 +GGML_TYPE_F64 = 28 +GGML_TYPE_IQ1_M = 29 +GGML_TYPE_BF16 = 30 +GGML_TYPE_TQ1_0 = 34 +GGML_TYPE_TQ2_0 = 35 +GGML_TYPE_MXFP4 = 39 + +GGML_TYPE_RAW = { + GGML_TYPE_F32: ("F32", 4), + GGML_TYPE_F16: ("F16", 2), + GGML_TYPE_I8: ("I8", 1), + GGML_TYPE_I16: ("I16", 2), + GGML_TYPE_I32: ("I32", 4), + GGML_TYPE_I64: ("I64", 8), + GGML_TYPE_F64: ("F64", 8), + GGML_TYPE_BF16: ("BF16", 2), +} + +# Block sizes and bytes-per-block verified against ggml-org/ggml +# src/ggml-common.h (block struct static_asserts) and src/ggml.c +# (ggml_type_traits.type_size/blck_size), QK_K = 256. +GGML_TYPE_BLOCK = { + GGML_TYPE_Q4_0: ("Q4_0", 32, 18), + GGML_TYPE_Q4_1: ("Q4_1", 32, 20), + GGML_TYPE_Q5_0: ("Q5_0", 32, 22), + GGML_TYPE_Q5_1: ("Q5_1", 32, 24), + GGML_TYPE_Q8_0: ("Q8_0", 32, 34), + GGML_TYPE_Q8_1: ("Q8_1", 32, 36), + GGML_TYPE_Q2_K: ("Q2_K", 256, 84), + GGML_TYPE_Q3_K: ("Q3_K", 256, 110), + GGML_TYPE_Q4_K: ("Q4_K", 256, 144), + GGML_TYPE_Q5_K: ("Q5_K", 256, 176), + GGML_TYPE_Q6_K: ("Q6_K", 256, 210), + GGML_TYPE_Q8_K: ("Q8_K", 256, 292), + GGML_TYPE_IQ2_XXS: ("IQ2_XXS", 256, 66), + GGML_TYPE_IQ2_XS: ("IQ2_XS", 256, 74), + GGML_TYPE_IQ3_XXS: ("IQ3_XXS", 256, 98), + GGML_TYPE_IQ1_S: ("IQ1_S", 256, 50), + GGML_TYPE_IQ4_NL: ("IQ4_NL", 32, 18), + GGML_TYPE_IQ3_S: ("IQ3_S", 256, 110), + GGML_TYPE_IQ2_S: ("IQ2_S", 256, 82), + GGML_TYPE_IQ4_XS: ("IQ4_XS", 256, 136), + GGML_TYPE_IQ1_M: ("IQ1_M", 256, 56), + GGML_TYPE_TQ1_0: ("TQ1_0", 256, 54), + GGML_TYPE_TQ2_0: ("TQ2_0", 256, 66), + GGML_TYPE_MXFP4: ("MXFP4", 32, 17), +} + +_GGML_TYPE_NAMES = {} +for _id, (_name, _bytes) in GGML_TYPE_RAW.items(): + _GGML_TYPE_NAMES[_id] = _name +for _id, (_name, _block, _bytes) in GGML_TYPE_BLOCK.items(): + _GGML_TYPE_NAMES[_id] = _name + +GGML_TYPE_NAMES = dict(_GGML_TYPE_NAMES) + +METADATA_UINT8 = 0 +METADATA_INT8 = 1 +METADATA_UINT16 = 2 +METADATA_INT16 = 3 +METADATA_UINT32 = 4 +METADATA_INT32 = 5 +METADATA_FLOAT32 = 6 +METADATA_BOOL = 7 +METADATA_STRING = 8 +METADATA_ARRAY = 9 +METADATA_UINT64 = 10 +METADATA_INT64 = 11 +METADATA_FLOAT64 = 12 + +_SCALAR_FORMATS = { + METADATA_UINT8: ("B", int), + METADATA_INT8: ("b", int), + METADATA_UINT16: ("H", int), + METADATA_INT16: ("h", int), + METADATA_UINT32: ("I", int), + METADATA_INT32: ("i", int), + METADATA_FLOAT32: ("f", float), + METADATA_UINT64: ("Q", int), + METADATA_INT64: ("q", int), + METADATA_FLOAT64: ("d", float), +} + + +class GGUFError(ValueError): + pass + + +def ggml_type_name(ggml_type): + return GGML_TYPE_NAMES.get(ggml_type, "T%d" % ggml_type) + + +def ggml_block_spec(ggml_type): + if ggml_type in GGML_TYPE_RAW: + return 1, GGML_TYPE_RAW[ggml_type][1] + if ggml_type in GGML_TYPE_BLOCK: + name, block, bytes_per_block = GGML_TYPE_BLOCK[ggml_type] + return block, bytes_per_block + raise GGUFError("unsupported ggml type %d (%s)" % (ggml_type, ggml_type_name(ggml_type))) + + +def _aligned(value, alignment): + return (value + alignment - 1) // alignment * alignment + + +def _tensor_bytes(dims, ggml_type): + numel = 1 + for dim in dims: + numel *= dim + block_elems, block_bytes = ggml_block_spec(ggml_type) + row = dims[0] if dims else 1 + if block_elems > 1: + if row % block_elems != 0: + raise GGUFError( + "ggml type %s requires dims[0] (%d) to be a multiple of %d" + % (ggml_type_name(ggml_type), row, block_elems)) + blocks = row // block_elems + if len(dims) > 1: + row_bytes = blocks * block_bytes + rest = 1 + for dim in dims[1:]: + rest *= dim + return row_bytes * rest, numel + return blocks * block_bytes, numel + return numel * block_bytes, numel + + +class GGUFReader: + def __init__(self, path): + self.path = path + self.f = open(path, "rb") + self.file_size = os.fstat(self.f.fileno()).st_size + self.version = None + self.alignment = DEFAULT_ALIGNMENT + self.metadata = {} + self.tensors = [] + self._tensor_index = {} + self._data_base = None + try: + self._parse() + except GGUFError: + self.f.close() + raise + except (OSError, struct.error) as exc: + self.f.close() + raise GGUFError("%s: %s" % (path, exc)) from exc + + def close(self): + self.f.close() + + def __enter__(self): + return self + + def __exit__(self, exc_type, exc, tb): + self.close() + return False + + def _read_exact(self, count, context): + if count < 0 or self.f.tell() + count > self.file_size: + raise GGUFError( + "%s: truncated at offset %d reading %s (%d bytes, file is %d)" + % (self.path, self.f.tell(), context, count, self.file_size)) + data = self.f.read(count) + if len(data) != count: + raise GGUFError( + "%s: short read at offset %d reading %s" + % (self.path, self.f.tell() - len(data), context)) + return data + + def _read_u64(self, context): + return struct.unpack(" (1 << 31): + raise GGUFError("%s: unreasonable %s length %d" % (self.path, context, length)) + raw = self._read_exact(length, context) + try: + return raw.decode("utf-8") + except UnicodeDecodeError as exc: + raise GGUFError("%s: %s is not valid UTF-8: %s" % (self.path, context, exc)) from exc + + def _read_scalar(self, value_type): + fmt, cast = _SCALAR_FORMATS[value_type] + size = struct.calcsize(fmt) + return cast(struct.unpack("<" + fmt, self._read_exact(size, "metadata value"))[0]) + + def _read_metadata_value(self, value_type, depth=0): + if value_type == METADATA_STRING: + return self._read_string("metadata string") + if value_type == METADATA_BOOL: + raw = self._read_exact(1, "metadata bool")[0] + if raw not in (0, 1): + raise GGUFError( + "%s: invalid bool metadata value %d" % (self.path, raw)) + return bool(raw) + if value_type == METADATA_ARRAY: + # Arrays may nest (the spec allows it); bound the recursion so a + # hostile file cannot drive Python into RecursionError, which would + # escape the GGUFError contract every other failure keeps. + if depth >= MAX_METADATA_DEPTH: + raise GGUFError( + "%s: metadata array nesting exceeds %d levels" + % (self.path, MAX_METADATA_DEPTH)) + element_type = self._read_u32("array element type") + count = self._read_u64("array length") + if count > MAX_METADATA_ITEMS: + raise GGUFError( + "%s: metadata array of %d items exceeds cap %d" + % (self.path, count, MAX_METADATA_ITEMS)) + if element_type not in _SCALAR_FORMATS and element_type not in ( + METADATA_STRING, METADATA_BOOL, METADATA_ARRAY): + raise GGUFError( + "%s: unknown metadata array element type %d" + % (self.path, element_type)) + return [self._read_metadata_value(element_type, depth + 1) + for _ in range(count)] + if value_type in _SCALAR_FORMATS: + return self._read_scalar(value_type) + raise GGUFError("%s: unknown metadata value type %d" % (self.path, value_type)) + + def _parse(self): + magic = self._read_exact(4, "magic") + if magic != GGUF_MAGIC: + raise GGUFError( + "%s: bad magic %r (expected %r): not a GGUF file" + % (self.path, magic, GGUF_MAGIC)) + self.version = self._read_u32("version") + if self.version not in SUPPORTED_VERSIONS: + raise GGUFError( + "%s: unsupported GGUF version %d (supported: %s)" + % (self.path, self.version, SUPPORTED_VERSIONS)) + tensor_count = self._read_u64("tensor count") + kv_count = self._read_u64("metadata count") + if kv_count > MAX_METADATA_ITEMS: + raise GGUFError( + "%s: metadata count %d exceeds cap %d" + % (self.path, kv_count, MAX_METADATA_ITEMS)) + if tensor_count > MAX_TENSORS: + raise GGUFError( + "%s: tensor count %d exceeds cap %d" + % (self.path, tensor_count, MAX_TENSORS)) + for _ in range(kv_count): + key = self._read_string("metadata key") + value_type = self._read_u32("metadata value type") + if value_type not in _SCALAR_FORMATS and value_type not in ( + METADATA_STRING, METADATA_BOOL, METADATA_ARRAY): + raise GGUFError( + "%s: unknown metadata value type %d for key %r" + % (self.path, value_type, key)) + self.metadata[key] = self._read_metadata_value(value_type) + alignment = self.metadata.get("general.alignment") + if alignment is not None: + if not isinstance(alignment, int) or alignment <= 0 or alignment % 8 != 0: + raise GGUFError( + "%s: general.alignment %r must be a positive multiple of 8" + % (self.path, alignment)) + self.alignment = int(alignment) + for _ in range(tensor_count): + name = self._read_string("tensor name") + n_dims = self._read_u32("tensor rank") + if n_dims > 8: + raise GGUFError( + "%s: tensor %r has %d dims (max 8)" % (self.path, name, n_dims)) + dims = [self._read_u64("tensor dim") for _ in range(n_dims)] + ggml_type = self._read_u32("tensor type") + offset = self._read_u64("tensor offset") + if name in self._tensor_index: + raise GGUFError( + "%s: duplicate tensor name %r" % (self.path, name)) + tensor = { + "name": name, + "rank": n_dims, + "dims": dims, + "ggml_type": ggml_type, + "type_name": ggml_type_name(ggml_type), + "offset": offset, + } + self.tensors.append(tensor) + self._tensor_index[name] = tensor + self._data_base = _aligned(self.f.tell(), self.alignment) + for tensor in self.tensors: + if tensor["offset"] % self.alignment != 0: + raise GGUFError( + "%s: tensor %r offset %d is not aligned to %d" + % (self.path, tensor["name"], tensor["offset"], self.alignment)) + try: + nbytes, numel = _tensor_bytes(tensor["dims"], tensor["ggml_type"]) + except GGUFError as exc: + raise GGUFError( + "%s: tensor %r: %s" % (self.path, tensor["name"], exc)) from exc + absolute = self._data_base + tensor["offset"] + if absolute + nbytes > self.file_size: + raise GGUFError( + "%s: tensor %r range [%d, %d) exceeds file size %d" + % (self.path, tensor["name"], absolute, + absolute + nbytes, self.file_size)) + tensor["numel"] = numel + tensor["nbytes"] = nbytes + tensor["abs_offset"] = absolute + + def tensor(self, name): + tensor = self._tensor_index.get(name) + if tensor is None: + raise GGUFError( + "%s: missing tensor %r" % (self.path, name)) + return tensor + + def read_tensor(self, tensor): + self.f.seek(tensor["abs_offset"]) + data = self._read_exact(tensor["nbytes"], "tensor %r data" % tensor["name"]) + return data + + +def load(path): + return GGUFReader(path) diff --git a/c/tools/make_glm_oracle.py b/c/tools/make_glm_oracle.py index b8c05b506..97c075685 100644 --- a/c/tools/make_glm_oracle.py +++ b/c/tools/make_glm_oracle.py @@ -30,7 +30,7 @@ Regeneration (run from c/): python3 tools/make_glm_oracle.py --fmt6 # -> glm_tiny_fmt6/ (model.safetensors, config.json, ref_glm.json) python3 tools/make_glm_oracle.py --fmt4 # -> glm_tiny_fmt4/ - # verify: SNAP=./glm_tiny_fmt6 REF=./glm_tiny_fmt6/ref_glm.json TF=1 COLI_TEMP=0 ./colibri 64 16 16""" + # verify: SNAP=./glm_tiny_fmt6 REF=./glm_tiny_fmt6/ref_glm.json TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16""" import json, sys, argparse from pathlib import Path diff --git a/c/tools/repack_fp8_passthrough.py b/c/tools/repack_fp8_passthrough.py index 97c061566..71d097287 100644 --- a/c/tools/repack_fp8_passthrough.py +++ b/c/tools/repack_fp8_passthrough.py @@ -112,28 +112,21 @@ read side. `main()` also copies `config.json` (mandatory, colibri.c's load_cfg) and `generation_config.json` (best-effort) from --indir into --outdir. Together with kv_b_proj emission below, this makes the tool's -output a standalone-loadable model directory on its own, with one caveat: the -engine's fmt=8 absorb support (branch `f8/absorb-fmt8`) must land before a -kv_b_proj-carrying container is safe to decode through the batched serving -path -- see below. +output a standalone-loadable model directory on its own: both halves of the +engine-side absorb support (CPU and CUDA) exist -- see below. kv_b_proj (kind "kvb") is now emitted the same way as o_proj/attn: byte- preserved fp8 weight + renamed `.qs` scale sidecar, stamped (it clears the collision shape the same way o_proj does -- M1 audit, GLM-5.2's kv_b_proj is -never at the ambiguous [O,I] this tool's `_check_geometry` guards). THIS TOOL -CHANGE IS INDEPENDENT of, and does not wait on, the engine-side absorb work: -colibri.c's MLA-absorption CPU path (qt_addrow/qt_matvec_rows, called only on -l->kv_b) and the CUDA absorb kernels have no fmt==8 case as of this tool's own -HEAD -- that support is being built in parallel on `f8/absorb-fmt8` and is -NOT part of this diff. A container minted with this tool BEFORE that engine -branch lands will load clean (qt_from_disk resolves the stamp exactly like -o_proj's) but crash -- loudly, not silently -- on any decode that reaches the -batched (`kvs`-nonNULL) serving path, because `ABSORB=0` cannot bypass absorb -there. Sequencing that correctly (engine work before this container is used -for decode) is the gate's job, not this tool's: repacking kv_b_proj here is -the tool-side half of a two-sided integration, done now because engine and -tool work can (and per the gate's assembly manifest, should) proceed in -parallel. +never at the ambiguous [O,I] this tool's `_check_geometry` guards). The +engine side of this two-sided integration now exists on both backends: +colibri.c's MLA-absorption path (qt_addrow/qt_matvec_rows, called only on +l->kv_b) carries explicit fmt==8 decode arms, and the CUDA absorb kernels +decode fmt=8 through weight_at/absorb_scale (admitted by +coli_cuda_weight_at_supported, uploads gated on the published e4m3 LUT). +A container minted here loads clean (qt_from_disk resolves the stamp +exactly like o_proj's) and decodes correctly through the batched +(`kvs`-nonNULL) serving path, where absorb cannot be bypassed. METADATA STAMP (reference implementation of the FORMATS-registry FR -- see docs/FORMATS.md): every output shard's safetensors `__metadata__` carries a @@ -657,7 +650,7 @@ def _check_stamp_budget(total_stamped): # produce ONE container, but each keeps its own progress manifest (a resume # manifest has to be per-selection -- see --mtp's comment in main()). The stamp # budget is not per-selection: st_fmt_stamp_ingest accumulates S->fmt_n over every -# shard it discovers in the directory (c/st.h:372), so ST_FMT_STAMP_MAX is a +# shard it discovers in the directory (c/st.h, the stamp scan), so ST_FMT_STAMP_MAX is a # CONTAINER-wide bound. Counted per-manifest, two passes could each stay under the # cap and still hand the engine a container over it -- the writer guarantee would # be 2x the reader's bound. So the budget check sums this pass's running total with @@ -933,17 +926,19 @@ def _print_inventory_summary(all_inv, dry_run): # the minted directory does not have). # # The EXTERNAL (non-tensor) files the loader/server actually open from a -# model dir at runtime, verified against this worktree's HEAD: -# - config.json cfg_root, colibri.c:1361 -- fopen(...); if(!f){ +# model dir at runtime. Cited by SYMBOL, not line number: the previous table +# carried line anchors that rotted twice, and a stale number is worse than no +# number because it reads as a verification that was not performed. +# - config.json colibri.c, cfg_root() -- fopen(...); if(!f){ # perror(p); exit(1); } -- MANDATORY, the run aborts without it. Also -# read by openai_server.py's Engine.__init__ (:1773) for arch detection. -# - generation_config.json colibri.c:1404-1405 -- fopen, comment "assente +# read by openai_server.py's Engine.__init__ for arch detection. +# - generation_config.json colibri.c, cfg_root() -- fopen, comment "assente # = nessun problema: e' opzionale" -- best-effort; HF's authority for # generation defaults (extra EOS stop ids) when present. -# - tokenizer.json c/tok.h:101 (tk_read_file, called from tok_load) +# - tokenizer.json c/tok.h, tk_read_file() (called from tok_load) # -- fopen(...); if(!f){ perror(path); exit(1); } -- MANDATORY. Called # from every serve/generate entry point that needs a tokenizer -# (colibri.c:7294 run_text, :7952 run_serve_mux, :8134 main serve loop). +# (colibri.c: run_text, run_serve_mux, and the main serve loop). # Before this fix main() never copied it into --outdir, so a minted # directory was not standalone-loadable for THIS reason alone (confirmed # by V2's end-to-end smoke test, 2026-08-18: load only succeeded via a @@ -954,8 +949,8 @@ def _print_inventory_summary(all_inv, dry_run): # fopen/Path().open against a model dir) and therefore excluded: # - tokenizer_config.json, chat_template.jinja -- the GLM-5.2 chat template # is reimplemented natively in code, not read from the .jinja file -# (colibri.c:8211 comment: "template UFFICIALE GLM-5.2 (chat_template -# .jinja): niente \n dopo i ruoli..."; openai_server.py:1016: "AUTHORITATIVE +# (colibri.c, comment: "template UFFICIALE GLM-5.2 (chat_template +# .jinja): niente \n dopo i ruoli..."; openai_server.py: "AUTHORITATIVE # GLM-5.2 tool-declaration block (byte-matches chat_template.jinja)" -- # both hardcoded to match the file's behavior, not sourced from it). # - README.md, LICENSE, .gitattributes -- pure repo/documentation metadata, diff --git a/c/tools/requirements-gguf.txt b/c/tools/requirements-gguf.txt new file mode 100644 index 000000000..16d5e9e98 --- /dev/null +++ b/c/tools/requirements-gguf.txt @@ -0,0 +1,14 @@ +# Dependencies for the GGUF reader / dequant / OLMoE-profile / converter tests +# (c/tests/test_gguf_reader.py, test_gguf_dequant.py, +# test_gguf_olmoe_profile.py, test_convert_gguf_to_olmoe.py). Installed only by +# the ci.yml `gguf` job; the engine-build and generic Python jobs stay lean. +# +# gguf is llama.cpp's gguf-py reference: test_gguf_dequant.py cross-checks our +# dequantization against gguf.quants.dequantize. The committed golden fixture is +# the frozen pin, so this is a minimum version that lets upstream drift surface. +# torch is deliberately absent: the quantizer is pinned by the committed golden +# fixture (tests/fixtures/olmoe_quantize_row_golden.npz) and the torch-based +# cross-checks stay optional for local runs. +numpy>=1.26 +safetensors>=0.4 +gguf>=0.19 diff --git a/c/version.py b/c/version.py index 58ab54982..db24fa2d8 100644 --- a/c/version.py +++ b/c/version.py @@ -1,3 +1,3 @@ """Single source of truth for the colibri version number.""" -__version__ = "1.12.0" +__version__ = "1.12.1" diff --git a/docs/CACHE_ROUTE.md b/docs/CACHE_ROUTE.md index 78f47e9c0..304409886 100644 --- a/docs/CACHE_ROUTE.md +++ b/docs/CACHE_ROUTE.md @@ -9,6 +9,24 @@ still rank inside top-`M`. This is **routing-side** (can change which experts run). Complementary to **PILOT** (next-layer *prefetch* of weights; does not change expert IDs). +## Engines and residency levels + +| engine | "resident" means | levels | +|---|---|---| +| `colibri` (GLM-5.2) | in the pin set or the RAM LRU cache | one | +| `qwen36` | in the VRAM expert tier (CUDA / Vulkan), else in the RAM LRU cache | two: VRAM first, then RAM | + +In `qwen36` a slot past the sacred top-`J` is filled, inside the top-`M` +window, by the highest-ranked VRAM-resident expert, then the highest-ranked +RAM-resident one, then the plain ranking. With no tier enabled the VRAM level +is empty and the lever degrades to the single-level GLM behaviour. The knobs +and meters are the same in both engines; the `qwen36` footer additionally +splits the swap count into `N to VRAM`. + +The only substitutable slots are ranks `J..K-1`. On a model that routes +top-2, the default `ROUTE_J=2` leaves nothing to substitute and the footer +reports `swap 0.0%`; lower `ROUTE_J` to trade agreement for hits. + ## Flags | Env | Default | Meaning | @@ -42,6 +60,7 @@ Footer / serve `STAT` when enabled: - `route_agree` — |chosen ∩ true top-K| / K - `route_kl` — mass KL (true top-K vs chosen) - `hit N%` — expert cache hit (disk residency) +- `N to VRAM` (`qwen36` only) — of the swaps, how many landed on a tier-resident expert ## A/B vs PILOT @@ -61,5 +80,10 @@ CACHE_ROUTE=1 PILOT=1 ... **Routing-only** + telemetry + this note. Does **not** require CUDA/fuse/device-tier patches — CPU streaming + pin/LRU is enough to A/B the lever against PILOT / #119. +The `qwen36` port keeps that shape: `route_select()` is a pure function of the +softmax row, the group mask and a residency callback, pinned by +`tests/test_qwen36_cache_route.c` without a model. With the lever unset the +router runs the original top-K loop and emits byte-identical token ids. + Treat as experimental until quality gates (e.g. `./coli bench`) pass; do not default `CACHE_ROUTE=1`. diff --git a/docs/ENVIRONMENT.md b/docs/ENVIRONMENT.md index b3d362526..e6bee8245 100644 --- a/docs/ENVIRONMENT.md +++ b/docs/ENVIRONMENT.md @@ -2,7 +2,7 @@ Reference for the environment variables read by the colibrì engine. -**Generated from `dev @ def8419`** by scanning every `getenv()` / `getenv_utf8()` site in `c/*.c`, `c/*.h`, `c/*.cu` and `c/*.mm`. Defaults and behavior are taken from the source; see [MAINTAINING-DOCS.md](MAINTAINING-DOCS.md) to regenerate this after the code changes. +**Baseline generated from `dev @ def8419`** by scanning every `getenv()` / `getenv_utf8()` site in `c/*.c`, `c/*.h`, `c/*.cu` and `c/*.mm`. Individual entries are also maintained with their owning source. Defaults and behavior are taken from the source; see [MAINTAINING-DOCS.md](MAINTAINING-DOCS.md) to regenerate the full inventory after the code changes. ## Which program reads these? @@ -15,8 +15,8 @@ what follows, but the sister engines read their own: | `colibri` | `c/colibri.c` | everything below except the three sections named for another engine | | `kimi_k3` | `c/kimi_k3.c` | the `K3_*` family — see [Kimi K3 engine](#kimi-k3-engine-kimi_k3) | | `inkling` | `c/inkling.c` | `INK_*`, plus `CTX_MAX`, `PIN_N`, `REP_PEN`, `GPU_DEV`, `NOGPU` — see [Inkling engine](#inkling-engine-inkling) | -| `qwen36` | `c/qwen36.c` | `QWEN_*`, `Q36_*`, and its dense/CUDA-tier controls — see [Qwen3.6 engine](#qwen36-engine-qwen36) | -| `qwen38` | `c/qwen38.c` | `Q38_MAXT`, `Q38_EOS`, `Q38_NATIVE_FP8`, `Q38_NATIVE_BF16`, `Q38_PREFILL_BATCH`, `COLI_TIMERS` — see [Qwen3.8 engine](#qwen38-engine-qwen38) | +| `qwen36` | `c/qwen36.c` | `QWEN_*`, `Q36_*`, its dense/CUDA-tier controls, and the `CACHE_ROUTE` family (VRAM tier over RAM cache) — see [Qwen3.6 engine](#qwen36-engine-qwen36) | +| `qwen38` | `c/qwen38.c` | `Q38_MAXT`, `Q38_EOS`, `Q38_NATIVE_FP8`, `Q38_NATIVE_BF16`, `Q38_PREFILL_BATCH`, `Q38_TRUNK_CPU_INT8`, `Q38_FP8_KERNEL`, `COLI_TIMERS` — see [Qwen3.8 engine](#qwen38-engine-qwen38) | | `olmoe` | `c/olmoe.c` | `HOT`, `WIDE`, `SMOOTH`, `CONF_LIMIT`, `MAX_NEW`, `CHAT`, `EXPERT_DROP`, `WARMUP` — see [OLMoE engine](#olmoe-engine-olmoe) | | `deepseek_v4` | `c/deepseek_v4.c` | `CTX`, the `V4_*` / `DSV4_*` families and the two `COLI_CUDA_*_BATCH` gates — see [DeepSeek V4 engine](#deepseek-v4-engine-deepseek_v4); note that the CUDA section below describes `colibri.c` knobs (`COLI_CUDA`, `CUDA_DENSE`, ...) which the V4 engine does not read — its GPU switch is `DSV4_CUDA` | @@ -103,12 +103,12 @@ Format: `VAR` — default — effect. | `COUPLE` | unset | Path to a coupling-score file driving cross-layer expert prefetch (#176). When set, `couple_load` reads it. | | `COUPLE_K` | `8` | Top-K coupled experts per layer when `COUPLE` is set. | | `COUPLE_D` | `1` | Coupling lookahead depth (`1` or `2`) when `COUPLE` is set. | -| `CACHE_ROUTE` | `0` (off) | Opt-in max-rank cache-aware MoE routing (pin∪LRU prefer within top-M). See [CACHE_ROUTE.md](CACHE_ROUTE.md). | +| `CACHE_ROUTE` | `0` (off) | Opt-in max-rank cache-aware MoE routing (pin∪LRU prefer within top-M). Also read by `qwen36`, where the VRAM tier outranks the RAM cache. See [CACHE_ROUTE.md](CACHE_ROUTE.md). | | `ROUTE_J` | `2` | Sacred top ranks always taken when `CACHE_ROUTE=1`. | | `ROUTE_M` | `12` | Max-rank window for resident preference when `CACHE_ROUTE=1`. | | `ROUTE_P` | `0` | Cumulative mass window for CACHE_ROUTE (`0` = fixed M). | | `ROUTE_ALPHA` | `1` | Scale gate mass of substituted experts before renorm (`1` = off). | -| `ROUTE_AGREE` | auto | Overlap% + KL vs true top-K; auto-on when `CACHE_ROUTE=1`. | +| `ROUTE_AGREE` | auto | Overlap% + KL vs true top-K; auto-on when `CACHE_ROUTE=1`. Alone it changes nothing and prints the meters (always 100% / 0). | | `ROUTE_TRACE` | unset | If set to a path, logs every routing decision there (testing/analysis). | | `ABSORB` | `-1` (auto: absorbed for S≤4) | MLA attention absorption mode. | | `IDOT` | `1` | Integer dot-product kernel. `IDOT=0` uses exact f32 kernels (for A/B numerical checks). | @@ -117,7 +117,9 @@ Format: `VAR` — default — effect. | `COLI_NO_FUSED_PAIR` | `0` (off) | `=1` disables the fused-pair matmul kernel. | | `DISK_SPLIT` | `0` (off) | `=1` splits the reported disk-load time across the draft/absorb/forward phases in stats. | | `I4S` | per-ISA (`1` on AVX-512-VNNI / NEON-dotprod, `2` elsewhere) | Engage the int4 `IDOT` kernel for batch `S>=`. `I4S=1` turns IDOT on at decode too: int8-quantized activations on expert matmuls — **not bit-identical** to the f32 decode path (measured 0.39% of scale on the gate output; the same numerics prefill already uses at `S>=2`, and the shipped default on AVX-512-VNNI, measured +5.5% end-to-end there). Attention projections always stay exact regardless. A default flip on AVX-VNNI awaits the quality ablation. | -| `IDOT_GS` | `0` (off) | **Opt-in** grouped planar IDOT for `fmt=4` (gs64/gs128) tensors: int8 activations with the K1 plane layout, one integer dot per scale group. Same numerics family as `I4S=1` — not bit-identical to the f32 grouped kernel, hence off until the ablation. Requires the planar family (AVX2 build, no GPU backend, no `XEXP`). Activation prints `[K1b]` once. | +| `IDOT_GS` | `0` (off) | **Opt-in** grouped planar IDOT for `fmt=4` (gs64/gs128) tensors: int8 activations with the K1 plane layout, one integer dot per scale group. Same numerics family as `I4S=1` — not bit-identical to the f32 grouped kernel, hence off until the ablation. Requires the planar family (AVX2 or AVX-512 build — on AVX-512 only the fmt=4 tensors planarize, fmt=2 keeps the pair layout — no GPU backend, no `XEXP`). Multi-row calls (prefill batch-union, serve-mux decode) take a 1×4 row tile that pays each weight block's unpack once per 4 rows; on AVX-512-VNNI whole 64-element groups go through single `vpdpbusd` zmm ops. All shapes are bit-identical to each other and to the pure-C reference (integer group dots, same per-row fmaf order). Activation prints `[K1b]` once. | +| `AMX` | `1` (on where armable) | `=0` disables the K1c AMX int8 tile kernel inside the `IDOT_GS=1` family (Sapphire Rapids+; Linux arms tile state via `ARCH_REQ_XCOMP_PERM`, Windows 11 via `EnableProcessOptionalXStateFeatures`; other OSes fail closed). With gs a multiple of 64, one `tdpbssd` tile-multiply covers a scale group for 16 output rows × up to 16 activation rows; bit-identical to the vector K1b path. Arming prints `[K1c]` once. | +| `AMX_S_MIN` | `8` | Row threshold for the AMX tile kernel: below it the B-tile unpack does not amortize and the vector 1×4 tile is the better kernel. Measure on your host — the break-even depends on cache level and core count. | | `SPEC_PIN` | `1` (on) | Speculation gate mode. `0` reverts to the legacy S-dependent speculation gates (#163). | | `COLI_RAM_OVERCOMMIT` | off | `=1` overrides the "projected peak > MemAvailable → exit(2)" guard so a run that risks kernel OOM-kill is allowed to proceed. | @@ -275,6 +277,8 @@ These are for testing, benchmarking, or internal use — not part of the everyda | `COLI_CORPUS_MINACC` | `50` | Acceptance floor (percent) for the corpus source. Below it over a 24-proposal window the source pauses for 256 tokens, then re-arms — rejected drafts cost real time. | | `EXPERT_BUDGET` | `0` (off) | Cap experts loaded per layer (MoE-Spec). **Quarantined:** silently forced to `0` unless `EXPERT_BUDGET_EXPERIMENTAL` is set — every tested value is either no faster or incoherent (issue #303). | | `EXPERT_BUDGET_EXPERIMENTAL` | unset | Setting it (any value) allows `EXPERT_BUDGET>0` to actually take effect (expect garbage, #294). | +| `DEGRADE_ZERO` | `0` (off) | **Opt-in approximate mode:** miss slots with per-position gate weight < `DEGRADE_TAU` are zero-filled instead of triggering a blocking disk read. Decode-only (`S≤4`). Changes output — must be set explicitly. Measured on OLMoE-1B-7B: `tau=0.03` → +2.9% ppl, 21.8% slots zeroed; `tau=0.05` → +41% ppl. GLM-5.2 and Kimi K3 router contracts are unmeasured — treat `tau=0.03` as OLMoE-calibrated and tune per-model. `[PROF]` footer reports zeroed slot count and top-3 layers by drop share. See PR #906, issue #865. Calibration assumes a warm expert cache; cold-start transient is not characterized. | +| `DEGRADE_TAU` | `0.03` | Gate weight threshold for `DEGRADE_ZERO` (clamped to `(0, 1]`). Compared per-position, post-`norm_topk`, pre-`routed_scale` — i.e. as a fraction of each position's routed mass. | | `DSA` | on | Dynamic Sparse Attention indexer. `DSA=0` disables. | | `DSA_FORCE` | `0` | Force the DSA path on. | | `DSA_TOPK` | model value | Override the DSA index top-k (testing). | @@ -292,6 +296,8 @@ These are for testing, benchmarking, or internal use — not part of the everyda | `REF` / `REF_FORCE` | `ref_glm.json` | Reference-output comparison mode. | | `REPLAY` | unset | Replay mode. | | `TF` | unset | Teacher-forcing mode. | +| `ORACLE_STRICT` | unset (off) | `colibri` only, env-only. `=1` makes failed teacher-forcing (`TF`) and greedy oracle comparisons exit with status 1. Token-exact by default; only TF can use the mismatch allowance below. Non-finite logits and incomplete generation always fail strict mode; modes that bypass comparison are rejected. Unset or `0` keeps completed comparisons report-only. Invalid reference JSON/arrays fail regardless of this setting. See [CONTRIBUTING.md](../CONTRIBUTING.md) for the strict oracle commands. | +| `ORACLE_TF_MAX_MISMATCHES` | `0` | `colibri` only, env-only. Maximum token mismatches accepted with `ORACLE_STRICT=1` and `TF` set. Must be a nonnegative decimal integer smaller than the number of TF positions. CI uses `2` for the 32-position tiny fixture (30–32 matches); unset or `0` requires exact agreement. Mismatches remain visible in diagnostics. Ignored outside strict TF mode; cannot relax greedy comparison or non-finite-output checks. | | `CHAT_TEMPLATE` | `1` | Apply the GLM chat template (`0` = raw prompt). | | `PPL` | off (`olmoe.c` and `qwen38.c` only) | `PPL=1` enters teacher-forced NLL/perplexity meter mode in the OLMoE and Qwen3.8 sister engines. | | `ABLATE_SCORE` | unset | Causal-ablation sweep over `ABLATE_SCORE=`, with a per-target-position final-logit read-out. Runs before `SCORE` and exits when done. | @@ -387,6 +393,10 @@ and the CPU/GPU execution split. | Variable | Default | Effect | |---|---|---| | `COLI_DENSE_I8` | `1` (on) | Quantize resident dense matrices to per-row int8 at startup. `=0` keeps the f32 reference path for quality A/Bs. | +| `COLI_DENSE_IDOT` | `1` (on) | The dense trunk's GEMVs (DeltaNet projections and out_proj, attention q/k/v/o, shared expert, lm_head) quantize the activation to int8 once per call and run integer dot products (maddubs on AVX2, vpdpbusd on AVX-VNNI / AVX-512 VNNI) instead of converting every int8 weight to f32. Not bit-identical to the f32 path; measured +1.0% perplexity, lm_head 12.6 to 10.2 ms/token. `=0` restores the f32-activation kernel. | +| `QWEN_EXPERT_ACT` | `i8` | The routed experts' activation quantized to int8 once per row (expert_ffn.h mode 1). Measured +0.1% perplexity, expert compute 22.7 to 15.9 ms/token. `=f32` restores f32 activations and the bit-identical contract with the pair kernels. | +| `COLI_DENSE_BITS` | `8` | `=4` stores the dense trunk as int4 in blocks of 64 with one scale per block (the K1b planar layout, half the bytes), served by the grouped integer kernel; implies the integer dot. Opt-in: on the 35B it costs +10% perplexity on the whole trunk, +2.4% on lm_head alone (see `COLI_DENSE_INT4`). | +| `COLI_DENSE_INT4` | all components | With `COLI_DENSE_BITS=4`, a comma list of the components that take int4: `lmhead`, `dnproj`, `dnout`, `attn`, `shexp`, `router`. Measured on the 35B: `lmhead` +2.4% perplexity for 254 MB less per token; `lmhead,dnproj,dnout` +5.6%; everything +10%. | | `QWEN_EXPERT_KERNEL` | `1` (on) | Routed experts run through the shared `expert_ffn.h` kernel: the int4 stays packed in RAM (planar layout, half the expert-cache RSS of the int8 unpack), gate+up are one pass, and a layer is two OpenMP regions over (expert, row-chunk) items instead of 3 x top-k GEMV regions. Takes effect on an int4 gs=64 container whose hidden and expert widths are multiples of 64, and not under the CUDA expert tier. `=0` restores the unpack-to-int8 path; the two produce the same tokens (1024-token decode on the real container byte-identical; pinned on the tiny int4 fixture in CI), only the f32 accumulation order inside a dot differs. Measured at cap 256 on the real container: 12.8 -> 15.7 tok/s, peak RSS 29 -> 17 GB. | | `QWEN_DENSE_BATCH` | `1` (on) | On AVX2/FMA, reuse each dense-int8 weight decode across two prompt rows. `=0` restores one GEMV call per row. Decode `S=1` is unchanged. | | `QWEN_SHARED_BATCH` | bounded by 32 MiB scratch | Batch the CPU shared expert across prompt rows. `=0` restores scalar calls; a positive integer caps rows per chunk. The CUDA-tier overlap path is unchanged. | @@ -404,6 +414,8 @@ checkpoint layout and the text-only capability boundary. | `Q38_NATIVE_FP8` | `1` (on) | Keep routed E4M3 expert bytes and their F32 128×128 block scales native in the LRU. `=0` restores expanded-FP32 slots for A/B validation. | | `Q38_NATIVE_BF16` | `1` (on) | Keep resident and routed BF16 matrices in two-byte storage while retaining FP32 activations/accumulation. `=0` restores the expanded-FP32 reference. | | `Q38_PREFILL_BATCH` | `1` (on) | Route prompt rows in bounded expert-major chunks and batch resident shared-expert/DeltaNet projections. `=0` restores row-at-a-time prompt execution for A/B diagnosis; decode is unchanged. | +| `Q38_TRUNK_CPU_INT8` | `1` (on) | The dense trunk (DeltaNet and attention projections, hyper-connection mixers, shared expert, router, lm_head; every matrix of at least `Q38_TRUNK_MIN_KB`) is kept on the CPU as int8 rows with one scale per row and the BF16 copy is released; `q38_weight_matmul` quantizes the activation to int8 and uses the integer kernels of `idot.h` for decode and prefill. `=0` keeps the BF16 rows and the f32 kernel (the numeric reference). See [qwen38.md](qwen38.md#the-trunk-on-the-cpu-int8-rows). | +| `Q38_FP8_KERNEL` | vector | The routed experts' e4m3 blocks are decoded eight at a time in registers and multiplied with FMA (AVX2 builds); `scalar` restores `quant.h`'s table kernel, which differs only by float summation order inside a block. | | `COLI_TIMERS` | `0` (off) | Set to `1` for the detailed Qwen3.8 phase breakdown on stderr. The shared per-request `PROF` frame is emitted regardless. | ## DeepSeek V4 engine (`deepseek_v4`) @@ -461,6 +473,7 @@ These are read by the Python programs (not the `glm` engine), so they don't appe | `COLI_DEBUG` | `0` (off) | Tee the engine transaction to stderr, by level. **`1`** = decoded model output stream only (byte-by-byte, on both the tool-call and plain paths). **`2`** = both sides — the fully-rendered prompt the engine received *and* the output, bracketed and correlated by request id, so stderr reads as the whole conversation. Invaluable for seeing what the model received vs. emitted during an OpenCode session. | | `COLI_TOOL_SALVAGE` | `0` (off) | Opt-in de-mangler: reconstruct a malformed int4 tool call by mapping its lone payload onto the tool's primary parameter. Never rewrites well-formed output; recommended for int4 deployments. | | `COLI_THINK` | `0` (off) | Make thinking the default when the client sends *neither* `reasoning_effort` nor `enable_thinking`. Any explicit client value still wins. | +| `COLI_CONTINUE_ASSISTANT` | `1` (on) | On the OpenAI- and Anthropic-compatible chat endpoints, continue a trailing `assistant` message — render its turn open and resume from it, dropping the turn terminator and the generation cue — instead of opening a new turn, the same contract as Anthropic's API. On by default: a message list ending in a non-empty `assistant` turn continues. Set `0` to restore the old behavior (append a fresh generation cue). Refused with `tools`/`tool_calls`, and the turn must carry text not ending in whitespace. Every shipped family supports it, Kimi K3 included (its open turn is framed engine-side in `kimi_k3.c`). Unrelated to `COLI_PREFILL_CHUNK`, which is the compute phase. | | `COLI_MODEL` | unset | Default model directory (fallback for `--model`). | | `COLI_MODEL_ID` | `glm-5.2-colibri` | Model id reported by the API. | | `COLI_API_KEY` | unset | Required bearer token for the server. | diff --git a/docs/FORMATS.md b/docs/FORMATS.md index c7724d8aa..374741607 100644 --- a/docs/FORMATS.md +++ b/docs/FORMATS.md @@ -42,9 +42,9 @@ own verification anchor in its sources bullet): the stamp+registry series (#529) stacked directly on the fp8-passthrough series (#528), which is in turn based on dev `292ed4c` (post-#465, post-#457 Metal grouped-GEMV merge, post-#705 Vulkan/Kimi-K3 MXFP4 merge) — no -cross-tree line-number mixing. Every `c/colibri.c`/`c/quant.h` line number -in this document reflects that restack; re-verify them again if this branch -is rebased further. The fmt=6 and fmt=7 rows are upstream's own merged code +cross-tree mixing. Every `c/colibri.c`/`c/quant.h` symbol named in this +document is verified present at this branch's own head. The fmt=6 and +fmt=7 rows are upstream's own merged code (this branch's only fmt=6-adjacent change is the collision handling inside `qt_resolve_fmt`, `c/colibri.c`; it does not touch fmt=7/MXFP4 at all). @@ -94,39 +94,40 @@ fused path) — AND the other fused-bound tensors (`q_a`, `q_b`, `kv_a`, `o`, and on sparse layers `sh_gate`/`sh_up`/`sh_down`) sit on the fmt 1/2/3/4 allowlist; any other format on any of those tensors makes the affected layers' decode take the CPU path instead, announced by a one-line-per-tensor-kind -`[METAL]` stderr notice at load. Single source of truth (all anchors -`c/colibri.c` at branch head `kvb/fmt-gate-notice-r4`): the shared per-layer -predicate `metal_fused_layer_fmt_miss` (:3396, over the `metal_fused_fmt_ok` -allowlist, :3379), consulted by both gate sites — `attention_rows` (:3448) and -`layer_forward_rows` (:5792) — and by the load-time notice -`metal_fmt_gate_notice` (:1866, called from `model_init`). - -Sources for all rows (`c/quant.h`/`c/colibri.c` line numbers at this PR -pair's current restack, base dev `292ed4c`): - -- **fmt=0/1/2/3** — allocation policy: `qt_alloc`, `c/colibri.c:1105` +`[METAL]` stderr notice at load. Single source of truth in `c/colibri.c`: the +shared per-layer predicate `metal_fused_layer_fmt_miss`, over the +`metal_fused_fmt_ok` allowlist, consulted by both gate sites — +`attention_rows` and `layer_forward_rows` — and by the load-time notice +`metal_fmt_gate_notice`, called from `model_init`. + +Sources for all rows (`c/quant.h`/`c/colibri.c` symbols named below, +verified present at this branch's own head -- originally identified +against base dev `292ed4c`, reconfirmed here against the current +restack): + +- **fmt=0/1/2/3** — allocation policy: `qt_alloc`, `c/colibri.c` (`bits>=16→fmt=0`, `bits>=5→fmt=1`, `bits>=4→fmt=2`, else `fmt=3`). - Kernels: `matmul_q` (`quant.h:105`, fmt=1), `matmul_i4` (`quant.h:125`, - fmt=2), `matmul_i2` (`quant.h:251`, fmt=3); pack/quantize helpers - `quantize_rows` (`quant.h:928`, fmt=1) and `pack_int2` (`quant.h:980`, - fmt=3). Byte-count formulas: `qt_bytes`, `c/colibri.c:183`. -- **fmt=4** (`int4-grouped`) — kernel `matmul_i4_grouped`, `quant.h:168`; + Kernels: `matmul_q` (`quant.h`, fmt=1), `matmul_i4` (`quant.h`, + fmt=2), `matmul_i2` (`quant.h`, fmt=3); pack/quantize helpers + `quantize_rows` (`quant.h`, fmt=1) and `pack_int2` (`quant.h`, + fmt=3). Byte-count formulas: `qt_bytes`, `c/colibri.c`. +- **fmt=4** (`int4-grouped`) — kernel `matmul_i4_grouped`, `quant.h`; group size `gs` is per-tensor, not fixed at 64 (contrast fmt=5). Byte-count: - `qt_bytes`'s `fmt==4` branch (inside `c/colibri.c:183`); scale-count split: - `qt_scale_bytes`, `c/colibri.c:263`. + `qt_bytes`'s `fmt==4` branch (inside `c/colibri.c`); scale-count split: + `qt_scale_bytes`, `c/colibri.c`. - **fmt=5** (`int3-g64`) — group size is fixed (`I3_GROUP=64`, - `quant.h:293`; `I3_GBYTES=24`, `quant.h:294`); helpers `i3_groups` - (`quant.h:295`), `i3_rowbytes` (`quant.h:296`); kernel `matmul_i3` - (`quant.h:354`); pack helper `pack_int3_g64` (`quant.h:956`). Allocation: - `qt_alloc`'s `bits==3` branch (inside `c/colibri.c:1105`). + `quant.h`; `I3_GBYTES=24`, `quant.h`); helpers `i3_groups` + (`quant.h`), `i3_rowbytes` (`quant.h`); kernel `matmul_i3` + (`quant.h`); pack helper `pack_int3_g64` (`quant.h`). Allocation: + `qt_alloc`'s `bits==3` branch (inside `c/colibri.c`). - **fmt=6** (`e8-iq3-lattice`) — upstream's merged code: format section header - precedes `quant.h:1008`; constants `E8_QK=256` (`quant.h:1008`), - `E8_SUB=32` (`quant.h:1009`), `E8_BBYTES=98` (`quant.h:1010`); row-byte - helpers `e8_blocks`/`e8_rowbytes` (`quant.h:1011-1012`); rotation contract - documented at `quant.h:1305` ("fmt=6 stores W@Q, so activations must be + precedes `quant.h`; constants `E8_QK=256` (`quant.h`), + `E8_SUB=32` (`quant.h`), `E8_BBYTES=98` (`quant.h`); row-byte + helpers `e8_blocks`/`e8_rowbytes` (`quant.h`); rotation contract + documented at `quant.h` ("fmt=6 stores W@Q, so activations must be transformed before"). Loader discriminator, upstream form (dev, ns==4 tag check at the top of `qt_resolve_fmt`): this branch's SECOND DESIGN - LANDMINE comment (`qt_resolve_fmt`, `c/colibri.c:1356`) hardens that check + LANDMINE comment (`qt_resolve_fmt`, `c/colibri.c`) hardens that check against the degenerate collisions below without changing any genuine-fmt=6 outcome. - **fmt=7** (`mxfp4`, upstream's merged code) — Vulkan-only decode: shader @@ -139,27 +140,27 @@ pair's current restack, base dev `292ed4c`): CPU (`quant.h`) or Metal kernel exists for it, and `qt_resolve_fmt` has no byte-arithmetic branch that returns 7. - **fmt=8** (`fp8-e4m3-b128`, this branch) — decode table `E4M3_LUT` - (`quant.h:446`) / `e4m3_decode` (`quant.h:480`), block size - `FP8_BLOCK=128` (`quant.h:482`), kernel `matmul_fp8` (`quant.h:491`). + (`quant.h`) / `e4m3_decode` (`quant.h`), block size + `FP8_BLOCK=128` (`fp8_format.h`), kernel `matmul_fp8` (`quant.h`). Disambiguation from fmt=1 ("THE DESIGN LANDMINE" — the two formats' weight bytes are byte-identical and can only be told apart by scale-array geometry, which is ambiguous for some small shapes) and the fmt=6 collision ("SECOND DESIGN LANDMINE") both live in `qt_resolve_fmt` - (`c/colibri.c:1356`), which now also consults an optional `stamped_name` + (`c/colibri.c`), which now also consults an optional `stamped_name` parameter (this PR): for the fmt=6 collision, a stamp resolves what an absent stamp still refuses; for the fmt=1-vs-fmt=8 collision, an absent stamp already resolves to `int8-row` since the #528 INVERSION, and a stamp's role there is instead letting a genuinely-stamped `fmt=8` tensor override that default — see "The metadata stamp" below for the exact rule in both cases. FMT_NAMES table (`name string` to `fmt int`): - `c/colibri.c:1316`. -- **no ordinal** (`int4-rans256-g0`, merged tools-only tier — line numbers - at dev `7fb1159`, post-#671 merge `a3a5a75`, not at this PR pair's - restack base) — codec + record reader/writer: `c/rans.h` - (`RANS_NSTREAMS 256`, `c/rans.h:93`; the record layout in the file-header - comment, `c/rans.h:17-34`; that same header names its engine consumer - "a future engine decode stage", `c/rans.h:4` — the format's own statement - that none exists yet). Identity constants: `c/tools/rans_format.py:40-46` + `c/colibri.c`. +- **no ordinal** (`int4-rans256-g0`, merged tools-only tier — symbols + verified at dev `7fb1159`, post-#671 merge `a3a5a75`, not at this PR + pair's restack base) — codec + record reader/writer: `c/rans.h` + (`RANS_NSTREAMS 256`, `c/rans.h`; the record layout in the file-header + comment, `c/rans.h`; that same header names its engine consumer + "a future engine decode stage", `c/rans.h` — the format's own statement + that none exists yet). Identity constants: `c/tools/rans_format.py` (`FORMAT_NAME = "int4-rans256-g0"`; `METADATA_KEY = "colibri.fmt"` — the same key/shape as the fmt=8 stamp convention below, but MANDATORY here rather than a cross-check, because there is no byte arithmetic to fall @@ -169,15 +170,15 @@ pair's current restack, base dev `292ed4c`): named refusal classes in its module docstring). Full specification: `docs/int4-rans256-g0.md`. Engine-interaction status, stated precisely (why the ordinal column is empty, and what still runs): `qt_resolve_fmt` - (`c/colibri.c:1374`) has no branch that returns this format, and + (`c/colibri.c`) has no branch that returns this format, and `c/colibri.c`/`c/quant.h`/`c/st.h` contain no reference to it — no decode path, hence no ordinal. But a repacked shard is not invisible to - the engine: `st_fmt_stamp_ingest` (`c/st.h:317`, called from + the engine: `st_fmt_stamp_ingest` (`c/st.h`, called from `st_init_multi`'s discovery loop) parses its mandatory `colibri.fmt` stamp map at container-discovery time, and the three routed-expert load sites — exactly this format's target population — resolve formats by - byte arithmetic alone with `stamped_name=NULL` (`c/colibri.c:2217`, - `c/colibri.c:2386`, `c/colibri.c:2580`, each marked + byte arithmetic alone with `stamped_name=NULL` (`c/colibri.c`, + `c/colibri.c`, `c/colibri.c`, each marked `/* routed expert: never stamped */`), so the stamp is never consulted where it would matter most. A future consumer must wire stamp-gated dispatch AHEAD of that inference — `docs/int4-rans256-g0.md`'s diff --git a/docs/api.md b/docs/api.md index 6e620dd36..50034b494 100644 --- a/docs/api.md +++ b/docs/api.md @@ -22,11 +22,47 @@ curl http://127.0.0.1:8000/v1/chat/completions \ ``` Implemented endpoints are `GET /v1/models`, `GET /v1/models/{model}`, -`POST /v1/chat/completions`, and legacy `POST /v1/completions`. Chat and +`POST /v1/chat/completions`, legacy `POST /v1/completions`, `POST /v1/brio` +(closed-set scoring, [brio.md](brio.md)) and `POST /v1/systemone`, the +request and reply of TypeSafe's Jev API served by the same channel. Chat and completion requests support JSON responses, SSE streaming, usage counts, `max_tokens`/`max_completion_tokens`, `temperature`, `top_p`, and up to four -custom `stop` sequences. Stop sequences are removed from the response and end -generation early in both JSON and streaming modes. The extension +custom `stop` sequences. `max_tokens` is a ceiling, not a target: when the +prompt leaves less room than the budget asks for, every engine clamps the +budget to what the context holds and the reply ends with `finish_reason: +"length"`; only a prompt that does not fit is refused, with +`context_length_exceeded` (#260, #1641). Stop sequences are removed from the response and end +generation early in both JSON and streaming modes. + +A trailing `assistant` message *continues* that turn instead of starting a new +one: the prompt ends inside it, which is the official template rendered with +`add_generation_prompt=False`. This is on by default — a message list ending in a +non-empty `assistant` turn continues, the same contract as Anthropic's API — and +there is no request field for it, because a trailing assistant turn already says +"continue me" and a body extension would only be reachable by hand-written JSON +rather than from the clients that want it. The server-side switch +`COLI_CONTINUE_ASSISTANT=0` restores the old behaviour (fold the turn into a +completed one and append a fresh cue). Continuation is refused together with +`tools`/`tool_calls`, because the tool-call parsers read an assistant turn from +its start, and the turn must carry text not ending in whitespace: the template +strips trailing whitespace, so the model would resume from different bytes than +the ones sent. A family whose renderer has no open-turn shape yet falls through +to the old behaviour rather than erroring — though every shipped family supports +continuation today, Kimi K3 included (its open turn is framed engine-side, in +`kimi_k3.c`, not derived in the gateway renderer). + +A continuation resumes from the exact bytes you send, which makes the split +point part of the prompt. Splitting mid-word puts the model at a token boundary +it would not have produced itself, and the first generated token is conditioned +on that split: measured on GLM-5.3-Flash, `The capital of France is Par` +completes to `París`, not `Paris` — deterministically, across every effort level +and both endpoints. The continuation is real (the model finished the partial +word rather than restarting, which is the behaviour this feature exists for); +the spelling is an artefact of where the split fell. This is inherent to +resuming from an arbitrary byte offset rather than specific to this engine, and +it is the same hazard as the trailing whitespace above — in the one form that +cannot be refused, because splitting mid-word is sometimes exactly what the +caller wants. Split at a token-ish boundary when the spelling matters. The extension `x_colibri_ignore_leading_stop: true` discards leading stop sequences until the first non-whitespace response content, which is useful for local templates that occasionally emit a role marker before the answer; strict OpenAI stop @@ -43,10 +79,17 @@ The server serves one generation at a time: the model stays in one persistent process, so concurrent HTTP requests queue instead of loading duplicate model copies. Tool calling depends on the active engine; see the support matrix below. Images, log probabilities, and token penalties return an explicit error rather -than being silently ignored. Audio is accepted only by Inkling checkpoints with +than being silently ignored. `seed` is accepted and ignored (see below). +Audio is accepted only by Inkling checkpoints with audio support. The default bind address is localhost; set `COLI_API_KEY` before exposing the server beyond the machine. +### `seed` + +`seed` is accepted (not rejected) for OpenAI-API request-shape compatibility; +the value is not validated. This server sends no per-request seed on the +wire, so the value has no effect at any temperature. + ### Tool-calling support | Engine | OpenAI `tools` | Anthropic `tool_use` | Native format | @@ -89,9 +132,76 @@ admission queue instead of pretending to run unsafe parallel sequences. Configure it with `--max-queue N` (default 8) and `--queue-timeout SECONDS` (default 300), or the `COLI_MAX_QUEUE` / `COLI_QUEUE_TIMEOUT` environment variables. Saturated and timed-out requests receive OpenAI-shaped HTTP 429 -errors before streaming headers are sent. `GET /health` exposes +errors before streaming headers are sent. With `--max-queue 0`, a request +pinned to an occupied KV slot is rejected immediately even if another slot +is free. Queue deadlines are checked before slot assignment: an expired +waiter receives `queue_timeout` even if a slot is now available. `GET /health` exposes active/queued/completed/rejected counters, and successful generation responses include `x-colibri-queue-wait-ms`. +Requests targeting any slot may use a free slot not reserved by earlier waiters. +An earlier pinned request keeps priority for its target; an earlier any-slot +request keeps priority across all slots. A full waiting queue does not reject +a request that can immediately take an unreserved free slot; the queue limit +bounds waiting requests, independently of active capacity. + +## Prometheus metrics + +`GET /metrics` returns Prometheus text exposition (version 0.0.4). When an +API key is configured, supply the same `Authorization: Bearer ...` or +`x-api-key` header used for generation; missing or invalid credentials return +401. Without an API key, the endpoint follows the server's usual unauthenticated +access policy. Metrics contain no prompts, model paths, or request-ID labels. + +All names start with `colibri_scheduler_`: + +| Suffix | Type | Meaning | +|---|---|---| +| `active`, `queued`, `capacity`, `max_queue` | gauge | Admitted requests, waiters, KV slot capacity, and queue limit | +| `admitted_total` | counter | Requests admitted to a KV slot | +| `completed_total` | counter | Admitted requests that returned normally | +| `failed_total` | counter | Admitted requests that raised an error, excluding `ClientCancelled` | +| `rejected_total`, `timed_out_total` | counter | Queue-full refusals and queue timeouts | +| `cancelled_total` | counter | Cancellations detected before admission or during admitted work | +| `queue_wait_seconds` | histogram | Wait until admission, for admitted requests only | +| `slot_duration_seconds` | histogram | Slot occupancy until completion, failure, or cancellation | +| `first_output_seconds` | histogram | Engine-call start to first nonempty text or tool-output callback | +| `engine_call_seconds` | histogram | Duration of each finished engine generation call, including failure/cancellation | + +Histogram buckets are 0.001, 0.01, 0.05, 0.1, 0.5, 1, 5, 10, 30, 60, 300 seconds, +and `+Inf`; each histogram exposes `_bucket`, `_sum`, and `_count`. +Counters reset when the gateway restarts. Collection does not call the engine +or consume a generation slot. `failed` is also included in `/health`'s +authenticated scheduler snapshot; failures no longer increment `completed`. + +Cancellation is checked before acquiring an available KV slot, including when a +waiting request wakes as capacity becomes free. A request cancelled at this +point increments `cancelled_total`, but not `admitted_total`, and contributes +no admission-wait or slot-duration sample. Queue-full and scheduler-closed +checks can still reject a request before the cancellation check is reached. + + +The engine-call histograms exclude admission queue wait and prompt rendering. +First output is observed before the gateway's stop filtering, reasoning split, +or HTTP serialization: it can be reasoning or tool data, not necessarily +user-visible answer text. Empty callbacks, ACCEPT frames, and SSE keepalives do +not count. Calls that finish or fail without output add no first-output sample; +a failure after output retains that sample. Engine-call duration includes callback +processing and response writes during generation. One request can invoke the +engine multiple times (for example Brio scoring), so these histogram counts are +engine calls, not HTTP request counts. + +These are gateway observations, not end-to-end client TTFT, per-token latency, +or GPU kernel measurements. Output callbacks need not correspond one-to-one to +tokens. Slot occupancy includes any response handling while the slot is held. Validation/authentication failures before admission are not counted. +`completed` means the admitted handler returned normally, not that the client +received every response byte. Request exceptions can include client input or +transport errors as well as engine failures. + +Example PromQL for the admitted-request queue-wait p95: + +```promql +histogram_quantile(0.95, sum by (le) (rate(colibri_scheduler_queue_wait_seconds_bucket[5m]))) +``` ## Anthropic-protocol endpoint (`/v1/messages`) @@ -132,11 +242,19 @@ tool declarations and choices explicitly instead of feeding another architecture's markers to an incompatible tokenizer. Not supported, and refused explicitly rather than ignored: `stop_sequences`, -`top_k`, and non-text content blocks (images, documents). Errors use Anthropic's -own `{"type":"error","error":{...}}` envelope on this path. Architecture-local +`top_k`, and non-text content blocks (images, documents). Errors use Anthropic's own +`{"type":"error","error":{...}}` envelope on this path. Architecture-local features that have not been wired to this protocol are likewise rejected with an explicit error. +A trailing `assistant` message continues that turn by default on both the +Anthropic- and OpenAI-compatible endpoints (`COLI_CONTINUE_ASSISTANT=0` restores +the old behavior, where this endpoint appended a fresh cue). Note this changes +what an existing Anthropic client sees on `/v1/messages`: a trailing assistant +turn now continues rather than starting fresh — which is the real Anthropic +contract — and the off-switch is the escape hatch for anyone relying on the old +behavior. + > The prefill warning below applies here too, and applies *hardest* to Claude Code: > its system prompt and tool catalog are large, and on a disk-streaming CPU path > that is a long silent wait before the first token. Read it before you connect. @@ -194,6 +312,46 @@ The `"api_key": "local"` dummy is what satisfies clients that demand a key. `context_window` is only the client's budget display — set it to whatever your KV configuration actually allows. +**pi** — add a custom provider to `~/.pi/agent/models.json` ([pi](https://github.com/earendil-works/pi-coding-agent) loads every OpenAI-compatible server through its `openai-completions` API): + +```json +{ + "providers": { + "colibri": { + "baseUrl": "http://localhost:8000/v1", + "api": "openai-completions", + "apiKey": "local", + "compat": { + "supportsDeveloperRole": false, + "supportsReasoningEffort": false + }, + "models": [ + { + "id": "glm-5.2-colibri", + "name": "GLM-5.2 (Colibri)", + "contextWindow": 131072, + "maxTokens": 1024 + } + ] + } + } +} +``` + +The `apiKey` dummy satisfies pi's auth requirement; colibri only enforces a +key if you set `COLI_API_KEY`. The `compat` flags tell pi to send the system +prompt as a plain `system` message and to omit `reasoning_effort`, which keeps +the request inside what the gateway accepts by default. If you serve GLM-5.2 +and want its reasoning block, set `"reasoning": true` on the model and drop +`supportsReasoningEffort` — the standard `reasoning_effort` field enables +thinking on that engine. Then select the model with `pi --list-models` or the +`/model` picker (`colibri / GLM-5.2 (Colibri)`). + +`contextWindow` is only the client's budget display — set it to whatever your +KV configuration actually allows. Tool calling in pi works on the engines in +the [tool-calling matrix](#tool-calling-support) above; on unsupported engines +pi's tool calls fail the same way any OpenAI `tools` request does. + **Continue, Cline / Roo, `llm`, the OpenAI SDKs, …** — set the provider's base URL to `http://localhost:8000/v1`, the model to `glm-5.2-colibri`, and any dummy key (`OPENAI_API_KEY` / `OPENAI_BASE_URL` for env-based tools). diff --git a/docs/baselines/README.md b/docs/baselines/README.md new file mode 100644 index 000000000..cad336299 --- /dev/null +++ b/docs/baselines/README.md @@ -0,0 +1,130 @@ +# Colibri, SGLang and vLLM serving baseline + +This is a reproducible collection protocol, **not a measured performance ranking**. +`c/tools/benchmark_baseline.py` reuses the existing HTTP harness. It does not start, +stop, reconfigure, or install engines. Run one engine at a time on an explicitly +allocated host; three servers sharing the same GPU would invalidate isolation. +No model or GPU is needed to plan or summarize an experiment. + +## Freeze the experiment + +Copy `three-engine.example.json` and `workload.jsonl` into an experiment directory. +Fill every `REPLACE` value before running; the example intentionally fails +validation. Workload paths are relative to the manifest, not the working directory. +Keep the same manifest for every engine/round. Do not store keys in launch commands: +use environment-variable references. Only `api_key_env` names are read by the +collector; their secret values are never included in results. + +Record: + +- The physical host, CPU/NUMA topology, RAM, GPU count/model/VRAM, storage, + OS and driver/runtime versions. Use the same allocated hardware for every arm. +- The exact source model revision, tokenizer and rendered chat-template SHA-256s. + Hash a sorted inventory of weight files for each `artifact_sha256`; retain the + inventory and conversion commands. Include weight **and KV** precision in + `quantization`. These are operator declarations, not remote attestations. +- Engine commits or image digests and complete launch commands, including TP/PP/EP, + context limit, admission limits, memory budget and CPU thread placement. + Aliases and URLs may differ; they do not identify the checkpoint. +- Reasoning, speculation, prefix-cache and warmup policies. Disable speculation in + the initial baseline, then create a separate experiment to measure it. Warmup is + performed before **each concurrency cell**. Caches persist across cells; the + collector never flushes them. Reset/restart before each round if the declared + policy calls for it. Cold and warm results belong in separate experiments. +- A fixed quality evaluation dataset/revision, metric and acceptance threshold. + Run this independently on all three configurations and retain the results. + +Choose the comparison explicitly: + +| Mode | Enforced identity | Permitted interpretation | +| --- | --- | --- | +| `matched_artifact` | All three weight inventory hashes, formats and quantization descriptions must match | Matched-artifact performance observations, conditional on external quality and configuration checks | +| `deployment` | Shared source model/tokenizer/template and workload; engine artifacts may differ | A comparison of complete deployment configurations, **not an isolated engine speedup** | + +A Colibri converted INT4 artifact and a vLLM NVFP4 checkpoint belong in +`deployment`, even if they originated from the same model. Matching labels or +hashes does not prove equal quality or backend numerical behavior. + +Use separate manifests/directories for `fully_resident` and `offload`. Verify +residency with engine counters and I/O measurements; do not infer it from model +size or successful startup. Collect RAM/VRAM peaks, disk/PCIe traffic and power in +external telemetry, aligned to each report's UTC `started_at`. The manifest's +hardware data is descriptive; the tool does not measure memory or energy. + +## Collect the matrix + +From the repository root: + +```sh +python3 c/tools/benchmark_baseline.py plan --manifest /path/experiment/manifest.json +``` + +The default example describes C1/C4/C8/C16, three independent rounds, and 16 +measured requests per cell. **The two included prompts are smoke inputs only**; +replace them with representative short, long and shared-prefix workloads and +increase the request count before interpreting P95/P99. Request counts must at +least reach the largest concurrency; this does not guarantee sustained occupancy. +`repeats` repeats requests within a cell; `rounds` repeats complete measurements. + +The plan rotates engine order to reduce systematic order bias: + +1. Colibri, SGLang, vLLM. +2. SGLang, vLLM, Colibri. +3. vLLM, Colibri, SGLang. + +After manually starting the selected engine with the recorded configuration: + +```sh +python3 c/tools/benchmark_baseline.py run \ + --manifest /path/experiment/manifest.json \ + --engine colibri --round 1 --results /path/experiment/results +``` + +Repeat for each plan entry, releasing the previous engine's allocated hardware +before starting the next. The collector only contacts the selected endpoint. +Each cell writes `-r-c.json`, with the full manifest, +workload and harness hashes, raw per-request records, warmup and summary. +Existing cells are never overwritten. A failed warmup saves evidence and stops +the round; measured failures are retained and make the command exit nonzero. +For a failed/invalid campaign, keep the evidence and use a new results directory +for a replacement campaign. Do not silently substitute the fastest repeat. + +## Summarize and interpret + +```sh +python3 c/tools/benchmark_baseline.py compare \ + --manifest /path/experiment/manifest.json \ + --results /path/experiment/results > /path/experiment/comparison.json +``` + +The comparison rejects mismatched manifests/workloads, mixed collector/harness +versions, duplicate or unexpected cells, altered summaries and missing requests. +Missing cells, failed requests, missing usage and empty/filtered output are listed +as issues and return a nonzero exit. No failed cell is silently dropped. + +For each engine/concurrency it reports min/median/max **across rounds** for: + +- Aggregate successful completion tokens per full batch wall second. +- Client first-output P50/P95/P99 and completion-duration P95. +- Failure rate and latency-SLO request goodput. +- It also reports actual server-reported output-length distribution across rounds. + +Round percentiles are not pooled percentiles. Throughput is aggregate, never +per-user decode speed, and includes HTTP, queueing and prefill. First output is +the first nonempty content/reasoning/tool delta, not an instrumented first token. +SSE chunk gaps are **not ITL**, and no token-latency/TPOT estimate is manufactured. +Instrument the engines or use their native benchmarks for token timestamps; +report those separately with their definitions. The underlying HTTP timing, +usage and socket-timeout boundaries are documented in [benchmarking.md](../benchmarking.md). + +`protocol_complete` means the matrix completed without those protocol/data gaps. +It does **not** mean quality passed, outputs have equivalent lengths, SLOs were +met, resources were isolated, residency was verified, or memory/power was measured. +`quality: not_assessed` remains explicit. This report deliberately emits neither +winner labels nor speedup ratios. Inspect lengths, reasoning accounting, external +quality and telemetry before publishing any performance conclusion. + +The matrix currently uses closed-loop traffic. Rate-driven interference tests +remain available in `benchmark_http_serving.py`; do not mix them into this matrix. +A useful follow-up workload adds a long prefill while short decodes are active, +with a separate report of first-output and ongoing token-latency interference. diff --git a/docs/baselines/three-engine.example.json b/docs/baselines/three-engine.example.json new file mode 100644 index 000000000..b8e1d074f --- /dev/null +++ b/docs/baselines/three-engine.example.json @@ -0,0 +1,73 @@ +{ + "schema": "colibri.serving-baseline/v1", + "experiment": "REPLACE-experiment-id", + "comparison": "deployment", + "residency": "fully_resident", + "workload": "workload.jsonl", + "hardware": { + "host": "REPLACE-host", + "cpu": "REPLACE-cpu", + "ram": "REPLACE-ram", + "gpu": "REPLACE-gpu", + "storage": "REPLACE-storage", + "os": "REPLACE-os", + "driver": "REPLACE-driver" + }, + "model": { + "source_revision": "REPLACE-source_revision", + "tokenizer_sha256": "REPLACE-tokenizer_sha256", + "chat_template_sha256": "REPLACE-chat_template_sha256" + }, + "cache_policy": "REPLACE-reset-before-round; warm-with-recorded-workload", + "reasoning_policy": "REPLACE-explicit-thinking-setting", + "speculation_policy": "REPLACE-disabled-for-initial-baseline", + "quality_protocol": "REPLACE-quality-dataset-revision-metric-and-acceptance-threshold", + "matrix": { + "concurrency": [ + 1, + 4, + 8, + 16 + ], + "rounds": 3, + "repeats": 8, + "warmup_requests": 2, + "max_tokens": 128, + "temperature": 0, + "timeout": 120, + "slo_first_output": 5, + "slo_duration": 120 + }, + "engines": { + "colibri": { + "base_url": "http://127.0.0.1:8000/v1", + "served_model": "REPLACE-model-alias", + "revision": "REPLACE-commit-or-image-digest", + "launch_command": "REPLACE-exact-command-with-secret-environment-references", + "artifact_sha256": "REPLACE-weight-manifest-sha256", + "weight_format": "REPLACE-format", + "quantization": "REPLACE-weight-and-KV-precision", + "api_key_env": "OPENAI_API_KEY" + }, + "sglang": { + "base_url": "http://127.0.0.1:8000/v1", + "served_model": "REPLACE-model-alias", + "revision": "REPLACE-commit-or-image-digest", + "launch_command": "REPLACE-exact-command-with-secret-environment-references", + "artifact_sha256": "REPLACE-weight-manifest-sha256", + "weight_format": "REPLACE-format", + "quantization": "REPLACE-weight-and-KV-precision", + "api_key_env": "OPENAI_API_KEY" + }, + "vllm": { + "base_url": "http://127.0.0.1:8000/v1", + "served_model": "REPLACE-model-alias", + "revision": "REPLACE-commit-or-image-digest", + "launch_command": "REPLACE-exact-command-with-secret-environment-references", + "artifact_sha256": "REPLACE-weight-manifest-sha256", + "weight_format": "REPLACE-format", + "quantization": "REPLACE-weight-and-KV-precision", + "api_key_env": "OPENAI_API_KEY" + } + } +} diff --git a/docs/baselines/workload.jsonl b/docs/baselines/workload.jsonl new file mode 100644 index 000000000..dd5f416f9 --- /dev/null +++ b/docs/baselines/workload.jsonl @@ -0,0 +1,2 @@ +{"messages": [{"role": "user", "content": "Explain the difference between throughput and latency using a concrete example."}]} +{"messages": [{"role": "user", "content": "Write a short Python function that merges two sorted lists and explain its complexity."}]} diff --git a/docs/benchmarking.md b/docs/benchmarking.md index 9407467c1..3f7a123c5 100644 --- a/docs/benchmarking.md +++ b/docs/benchmarking.md @@ -75,3 +75,184 @@ platform-specific I/O constraints. This protocol originated with the measurements and draft contributed by [@outtodata in #867](https://github.com/JustVugg/colibri/issues/867), including their later correction of the #863 explanation. + +## Compare HTTP serving with a fixed workload + +`c/tools/benchmark_http_serving.py` uses the OpenAI-compatible streaming chat +endpoint and Python's standard library. Run it against Colibri, SGLang, or vLLM +with the same workload and generation settings. This measures the whole HTTP +request path, including server queueing and prefill, unlike a steady-decode-only +engine benchmark. No performance comparison is implied by providing the tool. + +Create a UTF-8 JSONL file with one conversation per line. Rows contain only +`messages`; messages contain a `system`, `user`, or `assistant` role and string +`content`: + +```jsonl +{"messages":[{"role":"user","content":"Explain how a CPU cache works."}]} +{"messages":[{"role":"system","content":"Answer briefly."},{"role":"user","content":"What is a mutex?"}]} +``` + +From the repository root, with the server already running: + +```sh +python3 c/tools/benchmark_http_serving.py \ + --base-url http://127.0.0.1:8000/v1 --model your-served-model \ + --workload prompts.jsonl --concurrency 4 --repeats 3 \ + --max-tokens 128 --temperature 0 --timeout 60 --output colibri-c4.json +``` + +Repeat with the other server's API root, model alias, and a distinct output file. +Authentication uses `OPENAI_API_KEY`, or the environment variable named by +`--api-key-env`; keys and message contents are not copied into reports. The +endpoint and model alias are recorded. URLs with credentials, queries or +fragments are rejected, and redirects are not followed. The requested endpoint +must support `stream_options.include_usage`; errors are recorded rather than +silently changing the workload or retrying. + +The JSON report includes the workload's SHA-256, generation settings, one record +per attempt in input order, HTTP status, failure category, relative start time, +duration, first output time, finish reason, and reported completion tokens. +Failed requests remain in the report, and any failure makes the command exit 1. +Latency summaries use successful requests only, with nearest-rank p50/p95/p99 +and sample counts. Failed requests retain their individual durations and any +observed output/usage. + +Optional `--warmup-requests N` sends N requests before measurement, cycling +through workload rows with the same generation settings and concurrency limit. +Warmup is closed-loop even when the measured phase uses `--request-rate`. +All warmup requests finish before a fresh measurement clock and arrival schedule +start. Their rows and summary are recorded separately under `warmup`; they do +not contribute to measured latency, tokens, throughput, or SLO goodput. The +default is zero (no warmup). If any warmup request fails, the report has +`status: "warmup_failed"`, `summary: null`, and an empty measured `requests` list; +the command exits 1 without starting measurement. Successful warmup does not +prove stable performance. It can populate prefix caches, so record the same +warmup and cache policy when comparing engines or runs. + +Measurement boundaries: + +- Without `--request-rate`, concurrency is closed-loop: at most that many + requests are in flight, and each worker starts its next request after its + previous stream ends. There is no automatic retry or cache flush. Repeats reuse the + conversations in file order; prefix caching and scheduling can affect results. +- Request timing begins inside the worker, before HTTP connection setup, and + excludes waiting for a local worker. Each request uses a new connection. + `--timeout` limits individual socket operations, **not total request time**; + a stream that keeps sending data can last longer. +- `first_output_seconds` measures receipt of the first nonempty content, + reasoning, or tool-function name/arguments delta (`tool_calls` or legacy + `function_call`). Role-only and empty deltas + do not count. This is client-visible first output latency, not necessarily + time to a visible answer or to exactly one token. Empty successful output + has no first-output sample. SSE chunk gaps are not reported as token latency. +- Success requires a supported finish reason (`stop`, `length`, `tool_calls`, + `function_call`, or `content_filter`) and `[DONE]`. Error/unknown finish reasons, + HTTP errors, stream errors, malformed responses, and incomplete streams fail. + With `n=1`, each nonempty choices array must contain exactly one choice with + integer index 0. Output text and tool-function fields must be strings or null. + Once a choice has finished, further choice chunks are rejected; a trailing + usage chunk with empty choices is accepted. + This establishes protocol completion, not output correctness; filtered or + length-limited output can still count as protocol success. +- Token counts come only from `usage.completion_tokens`. Successful completion + token throughput divides those counts by the **entire batch wall time**, + including failed attempts. It is null if any successful request lacks usage + (or no request succeeds). Counts from failed streams are excluded. Inspect + failure rate and usage coverage alongside throughput; backend tokenizers, + reasoning-token accounting, and stopping policies may differ. + +For a meaningful comparison, record the model weights, quantization, tokenizer, +chat template, reasoning mode, server commands/versions, cache state and hardware +beside the report. Keep requested settings equal, check actual output lengths, +and run an independent quality check: this tool deliberately does not save +response text or assess correctness. Retain the workload file with its hash, +interleave server runs, and report repeated-run spread. Start at concurrency 1, +then increase it to expose queueing and prefill interference. + +### Measure throughput within a latency target + +Add `--slo-first-output 1 --slo-duration 15` to require first output within one +second and protocol completion within fifteen seconds. Either flag can be used +alone; values are finite positive seconds and the boundary is inclusive. +`summary.latency_slo` reports the thresholds, timing basis, `requests_met`, the fraction of +**all attempts** meeting them, and `goodput_requests_per_second` (qualifying +successful requests divided by the entire batch wall time). Without thresholds, +this field is null. Failures never qualify, even if they emitted output before +failing. When a first-output target is set, empty-output successes also do not +qualify. Missing token usage does not prevent evaluating these latency targets. In +closed-loop mode the targets start at the HTTP request; scheduled-arrival mode uses +the scheduled arrival time and includes client dispatch delay. + +This follows the latency-constrained goodput approach used by +[vLLM's serving benchmark](https://docs.vllm.ai/en/latest/api/vllm/benchmarks/serve/), +with the client first-output boundary defined above. It does not assert TPOT or +per-token SLOs, output quality, a minimum response length, or an equivalent vLLM +TTFT definition. SLO misses alone do not change the CLI exit status: exit 1 still +means at least one protocol/transport failure. Record output-length and quality +controls alongside goodput so short or empty answers cannot masquerade as an +improvement. + +Keep the load model fixed when comparing reports. +[SGLang's serving benchmark](https://github.com/sgl-project/sglang/blob/main/python/sglang/benchmark/serving.py) +also supports request-rate-driven arrivals and trace timestamps. This tool uses +closed-loop concurrency by default; `--request-rate` selects periodic arrivals +by default, or Poisson arrivals with `--arrival-distribution poisson`, +as described below. None of these modes alone establishes a production SLO guarantee; +warmup/cache policy, workload representativeness and repeated-run controls still +matter. + +### Schedule arrivals independently of response time + +Add `--request-rate 5` to schedule five arrivals per second. Request `i` is due at +`i / rate` seconds from batch start; the first is due immediately. Absolute +monotonic deadlines prevent accumulated timer drift. This is a deterministic +periodic schedule. Add `--arrival-distribution poisson --seed 42` for +independent exponential intervals with mean `1 / rate`. The first request is +still immediate; subsequent deadlines accumulate sampled intervals. This follows +the arrival model supported by +[vLLM](https://docs.vllm.ai/en/latest/cli/bench/serve/#--burstiness) and SGLang; +it does not reproduce their random-number sequences or implement trace replay. +The default seed is 0. Equal seeds, rates and request counts reproduce planned +arrivals independently of warmup and response time, not actual network timing. +Finite Poisson samples need not realize the configured mean rate. The report +records the distribution and seed; retain per-request scheduled times for exact +schedule comparison. Compare identical schedules across servers, then repeat +with multiple seeds to measure sensitivity to arrival patterns. + +Requests are not retried +or dropped, and the finite workload drains before the report is written. + +`--concurrency` still caps simultaneous HTTP requests. Arrivals accumulate in +the client's executor queue when all workers are busy. Therefore the configured +rate is **scheduled arrivals, not guaranteed wire or server arrival rate**. +Delayed producer wakeups also contribute to dispatch delay; inspect client load +before attributing all delay to the server. The queue and retained results can +grow to the workload's total request count, so size workloads accordingly. + +Scheduled-arrival per-request records add: + +- `scheduled_seconds`: planned arrival relative to batch start; +- `dispatch_delay_seconds`: actual worker start minus scheduled arrival; +- `arrival_first_output_seconds`: dispatch delay plus HTTP first-output latency, + or null if no output was observed; +- `arrival_duration_seconds`: dispatch delay plus HTTP request duration. + +`summary.arrival_timing` reports dispatch-delay samples for all attempts and +arrival-based latency samples for successes. Existing HTTP timing fields retain +their original meaning. With a rate set, SLO evaluation uses arrival-based times +and records `timing_basis: scheduled_arrival`; without one it records +`request_start`. This prevents a request waiting two seconds for a worker and +then completing in 100 ms from meeting a one-second total-latency target. +Goodput still divides qualifying completions by the entire batch wall time, +including the drain after the last scheduled arrival. Keep rate, concurrency, +request count, latency targets and timing basis equal across compared reports. + +### Repeated Colibri / SGLang / vLLM baseline + +The [three-engine baseline protocol](baselines/README.md) provides a manifest, +rotating run plan and C1/C4/C8/C16 collector using this HTTP harness. Its comparison +checks workload/configuration identity, keeps failed and missing cells visible, +and reports min/median/max across rounds. It distinguishes matched artifacts from +deployment comparisons with different weight formats. Quality, token-level latency +and hardware telemetry require separate evidence; no performance ranking is bundled. diff --git a/docs/brio.md b/docs/brio.md index 0cc134fc2..89a0ae739 100644 --- a/docs/brio.md +++ b/docs/brio.md @@ -181,6 +181,74 @@ const { json, fields } = await r.json(); // json.queue, json.urgent are guaranteed to be values from your lists ``` +### For code written against Jev: `POST /v1/systemone` + +If your application already speaks TypeSafe's Jev API, point it at colibri +and change the base URL; the request and the reply are the same shape. The +route accepts any `model` name (a Jev client sends `jev-latest`) and answers +with the model the server actually runs. + +```bash +curl -s http://127.0.0.1:8000/v1/systemone \ + -H "Authorization: Bearer $COLI_API_KEY" -H "Content-Type: application/json" \ + -d '{ + "state": "Hi, I have been trying to connect my Stripe account for 3 days and the integration keeps failing. I am losing sales. Please help ASAP.", + "model": "jev-latest", + "questions": { + "urgency": {"type": "noul", "instructions": "Does this message express urgency?"}, + "department": {"type": "choice", "instructions": "Which team should handle this?", + "criteria": {"billing": "payments, invoices, Stripe payouts", + "technical": "bugs, outages, integration errors", + "sales": "pricing, plans, upgrades"}}, + "severity": {"type": "score", "instructions": "How severe is the customer impact?", + "criteria": ["no impact", "minor inconvenience", "blocked on one task", + "losing money", "business down"]} + }}' +``` + +The reply, from Qwen3.6-35B-A3B on a CPU box (1m46 with the experts streamed +from disk, the state read once for the three questions): + +```json +{"model": "qwen36", + "answers": { + "urgency": {"type": "noul", "noul": 0.847606}, + "department": {"type": "choice", "choice": "technical", + "probabilities": {"billing": 0.059806, "technical": 0.939983, "sales": 0.000211}, + "confidence": 0.909974}, + "severity": {"type": "score", "score": 3.831772, + "legend": {"1": "no impact", "2": "minor inconvenience", "3": "blocked on one task", + "4": "losing money", "5": "business down"}, + "probabilities": {"1": 0.044403, "2": 0.02624, "3": 0.034899, "4": 0.842097, "5": 0.052361}, + "confidence": 0.802621}}, + "usage": {"input_tokens": 86, "output_tokens": 15}} +``` + +How the three primitives map onto the `questions` form above, so you know +what the model is actually asked: + +| Jev primitive | what is scored | what goes into the question text | reply | +|---|---|---|---| +| `noul` | `yes` / `no` | `instructions`, then `yes: ` and `no: ` when given, then "Answer yes or no." | `noul` = probability of yes | +| `choice` | the labels of `criteria` (up to 255) | `instructions`, then one line per label with its description | `choice`, `probabilities` by label, `confidence` | +| `score` | the level numbers `"1".."n"` (2 to 10 levels) | `instructions`, then `k: ` per level, then "Answer with the number." | `score` = expected value under the distribution, `legend` by number, `probabilities`, `confidence` | + +`state` and `instructions` may be a string, an object or an array; JSON is +serialized as text. `confidence` is the formula their documentation gives, +`(n * peak - 1) / (n - 1)`: 1 when all the mass sits on one label, 0 when +flat; `noul` carries none, as in theirs. + +What differs, stated rather than hidden: + +- `model` in the reply is the served model, not a Jev version. +- `usage.output_tokens` counts the option tokens the engine **read**; nothing + is generated. `input_tokens` is the longest prompt of the request. +- Validation errors are `422`, as theirs, with this server's error envelope + (`error.message`, `error.param`). +- The probabilities are this model's, normalised over the options with the + `mean` rule above. The numbers will not match Jev's on the same input; the + contract is the same, the model is yours. + ### Do not put the options in the prompt Write the state and the question; leave the option list to the `options` field. On a diff --git a/docs/deepseek-v41.md b/docs/deepseek-v41.md index a4cee5b30..b81d8d94d 100644 --- a/docs/deepseek-v41.md +++ b/docs/deepseek-v41.md @@ -41,6 +41,18 @@ decimal **strings**: they run past a double's 53-bit mantissa, and a rounded multiplier would silently hash every n-gram into a different -- and perfectly valid-looking -- row. +When `--cap` is omitted, `coli chat`, `coli serve`, and `coli web` size the expert +cache from the resource plan. `--ram 120` (or `RAM_GB=120`) supplies its RAM budget; +without either, the planner uses available memory. Dense weights, context state, +and runtime reserves are subtracted before choosing slots per layer. `--cap N` +still selects an explicit slot count, and an existing `--auto-tier` plan or measured +profile retains precedence over this fallback. The startup line reports the chosen +cap. The budget is a ceiling, not a promise to fill RSS: cache slots fill on demand +and engram tables continue streaming from disk. + +The standalone `deepseek_v41` binary still takes its cap as its first argument; +`RAM_GB` planning happens in the Python gateway. + ## What the engine implements Each mechanism mirrors a named piece of the vendor's `inference/model.py`: diff --git a/docs/experiments/cnre-offline-simulator.md b/docs/experiments/cnre-offline-simulator.md index 2b3573785..490c28678 100644 --- a/docs/experiments/cnre-offline-simulator.md +++ b/docs/experiments/cnre-offline-simulator.md @@ -31,9 +31,20 @@ model this existing overlap. Long-prefill felt-wait estimates are therefore conservative and must be checked with lower `felt_fraction` sensitivity. Kimi K3 traces are not currently safe input because that engine emits routes -without advancing the trace call ID. Inkling and OLMoE initialize routing -telemetry but do not currently emit route records. Their cache traversal also -differs from GLM, so each needs an explicit engine profile before inclusion. +without advancing the trace call ID. Inkling initializes routing telemetry but +does not currently emit route records. Their cache traversal also differs from +GLM, so each needs an explicit engine profile before inclusion. + +OLMoE **does** emit route records now. It announced `ROUTE_TRACE` at startup +and wrote a zero-byte file for as long as the stream existed, because +`rt_init()` opens the file but nothing in `moe()` ever called `rt_trace()` — a +consumer saw "no data" rather than an error. `moe()` calls `rt_route()` per +position and `rt_trace_end()` once per invocation, so the stream has one line +per `(moe call, position, layer)` with `call` advancing once per layer per +forward (16 calls per forward on OLMoE, rows `0..S-1` within each). It still +needs an engine profile before this simulator accepts it, because its cache +traversal is per-layer LRU with a `--cap` slot budget and its gates are raw +softmax weights rather than a renormalised top-k (`norm_topk_prob=false`). ## Trace contract @@ -240,7 +251,9 @@ first useful campaign needs: 3. Exact commit, model/container identity, DRAFT/sampling settings, cache state, prefill chunk, and frozen `.coli_usage` snapshot. 4. Separate instrumentation fixes and engine profiles before Kimi K3, Inkling, - or OLMoE data is interpreted by this simulator. + or OLMoE data is interpreted by this simulator. OLMoE's instrumentation fix + (it emitted no route records) has landed, so only its engine profile and a + trace campaign remain. No server is required to run the simulator. A model-capable machine is needed only to collect missing traces and later validate a policy end to end. diff --git a/docs/experiments/qwen36-resident-batch-2026-09-22.jsonl b/docs/experiments/qwen36-resident-batch-2026-09-22.jsonl new file mode 100644 index 000000000..1e93a53f2 --- /dev/null +++ b/docs/experiments/qwen36-resident-batch-2026-09-22.jsonl @@ -0,0 +1,3 @@ +{"input":512,"output":1024,"rows":32,"pairs":9,"serial_median_ms":2.974592,"batch_median_ms":0.293992,"paired_speedup_median":10.018000,"cpu_sample_max_abs_error":0,"serial_ms":[2.879376,3.728909,2.975827,2.811392,2.950584,2.995621,2.974592,2.966765,3.317602],"batch_ms":[0.291110,0.293992,0.297048,0.313669,0.287397,0.288322,0.299809,0.318531,0.286065]} +{"input":2048,"output":2048,"rows":32,"pairs":9,"serial_median_ms":3.588027,"batch_median_ms":0.812655,"paired_speedup_median":4.314296,"cpu_sample_max_abs_error":0,"serial_ms":[2.770069,4.095639,2.697552,3.588027,3.637298,4.933534,4.105374,2.849211,2.814900],"batch_ms":[0.793348,0.812655,0.815726,0.831660,0.812434,0.848347,0.814705,0.811320,0.785599]} +{"input":2048,"output":2048,"rows":128,"pairs":9,"serial_median_ms":12.715961,"batch_median_ms":2.912792,"paired_speedup_median":4.365558,"cpu_sample_max_abs_error":0,"serial_ms":[12.715961,13.312398,13.366993,13.219180,15.129256,11.415760,11.160518,10.967418,11.883601],"batch_ms":[2.912792,2.887985,2.912167,2.882372,2.944258,2.923530,2.899905,3.928513,2.944615]} diff --git a/docs/glm53-flash.md b/docs/glm53-flash.md index 322f3bf43..0f4b155ba 100644 --- a/docs/glm53-flash.md +++ b/docs/glm53-flash.md @@ -143,7 +143,7 @@ Tool calling is complete. GLM-5.3 declares tools differently from GLM-5.2 (its own preamble, its own JSON serialisation, its own spacing inside ``) but emits calls identically, so the existing parser handles them unchanged. The whole rendering is pinned byte for byte against `chat_template.jinja` -(`tests/test_glm53_chat_template.py`). +(`tests/glm53_chat_template_harness.py`). ## Environment @@ -157,18 +157,26 @@ measured from available memory when unset), `GLM53_MAX_IMAGE_TOKENS`, ``` python3 tools/make_glm53_multimodal_tiny.py --output ~/glm53_mm_tiny python3 tools/make_glm53_streaming_pair.py --fixture ~/glm53_mm_tiny --output ~/glm53_stream -python3 tests/test_glm53_multimodal_tiny.py --binary ./glm53 --fixture ~/glm53_mm_tiny -python3 tests/test_glm53_streaming.py --binary ./glm53 \ +python3 tests/glm53_multimodal_tiny_harness.py --binary ./glm53 --fixture ~/glm53_mm_tiny +python3 tests/glm53_streaming_harness.py --binary ./glm53 \ --quantized ~/glm53_stream-i4 --dequantized ~/glm53_stream-deq -python3 tests/test_glm53_serve.py --binary ./glm53 --fixture ~/glm53_mm_tiny -python3 tests/test_glm53_vision_serve.py --binary ./glm53 --fixture ~/glm53_mm_tiny -python3 tests/test_glm53_chat_template.py --template /chat_template.jinja -make VK=1 glm53 && python3 tests/test_glm53_vulkan.py --binary ./glm53 --fixture ~/glm53_mm_tiny +python3 tests/glm53_serve_harness.py --binary ./glm53 --fixture ~/glm53_mm_tiny +python3 tests/glm53_vision_serve_harness.py --binary ./glm53 --fixture ~/glm53_mm_tiny +python3 tests/glm53_chat_template_harness.py --template /chat_template.jinja +make VK=1 glm53 && python3 tests/glm53_vulkan_harness.py --binary ./glm53 --fixture ~/glm53_mm_tiny ``` The generators want transformers 5.16.1, pinned because an oracle written by a -different version is a different oracle. Each test skips with the command that -builds what it is missing rather than throwing. +different version is a different oracle. Each harness skips with the command that +builds what it is missing rather than throwing, and exits 2 when it does: a skip +verified nothing and must not read as a pass. + +The harnesses are named `glm53_*_harness.py` so `make test-python` does not +collect them as empty unittest modules. `tests/test_glm53_oracles.py` wraps the +two stdlib-only oracles for unittest; it runs them when `GLM53_TINY` +(`tools/make_glm53_tiny.py`) and `GLM53_MM_TINY` (the multimodal fixture above) +point at their fixtures, as the GLM-5.3 CI job does, and skips with that reason +otherwise. Two of the generators refuse to write a fixture that cannot fail: one rejects a degenerate model that answers the same token everywhere, the other a fixture diff --git a/docs/int4-rans256-g0.md b/docs/int4-rans256-g0.md index b61dcb7d0..38526890d 100644 --- a/docs/int4-rans256-g0.md +++ b/docs/int4-rans256-g0.md @@ -77,7 +77,9 @@ Validity rules every conformant record satisfies (and validators enforce): least 4 bytes** — even a stream that encodes zero symbols carries its 4 flushed state bytes; - the record's total length equals the derived framing exactly (no trailing - bytes), and every derived padding byte is zero; + bytes), and every derived padding byte is zero. That length is the + **physical extent**, known from stored framing before decode — not the + identity inference the stamp section forbids (`expected_bytes(O, I)`); - **amplification bound**: no admissible table can encode more than `payload_len * 8 * M_max` symbols into a payload (`M_max = 2^15`, the format's largest table size — a symbol's cost is bounded below by @@ -165,18 +167,13 @@ For the existing formats the engine can infer identity from byte arithmetic **That inference is structurally impossible here**: entropy-coded size is data-dependent — there is no `expected_bytes(O, I)` to compare against. The stamp is therefore the **only** signal that a `U8` tensor is entropy-coded -at all. This is a statement about *identity*, not about extent: the -physical length of every record is still fully determined by its stored -framing (header, stream offsets, payload, the two `round16()` pads; see the +at all. Data-dependence applies to **identity inference**, not to **extent**: +the record's physical length is fully determined by the stored framing before +decode (header, stream offsets, payload, the two `round16()` pads; see the validity rules above, "the record's total length equals the derived framing -exactly"), and is known before decode. Only the payload's content length is -data-dependent, so page-aligned, fixed-size I/O planning over records remains -valid (#1273). Data dependence prevents **identity inference**, not extent planning: -the stored framing determines the record's complete physical length before -decode, so page-aligned reads (including 4 KiB-padded `O_DIRECT` extents) -remain possible. The stamp is still mandatory because the format identity, -not the physical extent, is what cannot be inferred. Consequences any consumer -must respect: +exactly"), so page-aligned I/O (4 KiB-padded extents, `O_DIRECT`) remains +available. The stamp is still mandatory because identity, not size, is what +cannot be inferred. Consequences any consumer must respect: - an *unstamped* `U8` tensor must never be presumed entropy-coded by any size heuristic; diff --git a/docs/multidisk.md b/docs/multidisk.md new file mode 100644 index 000000000..e3bff2c00 --- /dev/null +++ b/docs/multidisk.md @@ -0,0 +1,106 @@ +# Multi-disk streaming + +Use extra drives when model reads limit inference. `COLI_MODEL_MIRROR` points +to read-only copies of the primary model's shards; no RAID setup is required. +These examples target the GLM-5.2 engine. Other engines have their own I/O +paths, so check their model guides before applying the same settings. + +## Run with two or more copies + +Run from `c/` in a source checkout, or the directory containing `coli` in an +unpacked release. Replace the paths with your existing model copies. + +Bash, with one primary and two mirrors: + +```bash +COLI_MODEL_MIRROR='/mnt/nvme2/glm52_i4;/mnt/nvme3/glm52_i4' \ + python3 ./coli chat --model /mnt/nvme1/glm52_i4 +``` + +PowerShell, with the same three-drive layout: + +```powershell +$env:COLI_MODEL_MIRROR = 'E:\models\glm52_i4;F:\models\glm52_i4' +python ./coli chat --model 'D:\models\glm52_i4' +``` + +For two drives, supply just one mirror directory. Quote the list: an unquoted +semicolon separates shell commands. The primary is supplied by `--model` +(or `COLI_MODEL`); do not repeat it in the mirror list. + +With `COLI_DISK_WEIGHTS` unset, the engine probes bandwidth at startup and prints +the measured split. An optional override is a comma-separated list of positive +relative weights, primary first, followed by each usable mirror in list order. +For example, `9,3,3` requests shares of 60%, 20%, and 20% across three drives; +it is not a measured speedup. Check the startup log for skipped mirrors and +probe failures before trusting the split. See the +[environment reference](ENVIRONMENT.md#dual-ssd-streaming) for the variables. + +Missing or divergent mirror shards fall back to the primary. For a smaller +drive, use the [partial-mirror planner](../README.md#multiple-ssds-stream-model-copies-from-more-than-one-drive) +to stage a usage-ranked subset. `COLI_MODEL_DIRS` is a different layout: it +spreads distinct shards across drives to combine capacity, rather than adding +replicas of the same shards. + +## What has been measured + +More storage bandwidth does not translate directly into more tokens per second. +Compute, cache residency, and shared PCIe/controller bandwidth can dominate. +Compare the complete inference workload, not just the drive probe. + +- **Two independent NVMe drives helped.** The Threadripper PRO 7965WX report in + [#1249](https://github.com/JustVugg/colibri/issues/1249), also recorded in + [the benchmark table](benchmarks.md), measured 0.80 to 1.10 tok/s (+37.5%) + with `DIRECT=1`; the buffered comparison gained about 16%. +- **Mixed-speed drives exposed a bug that has since been fixed.** + [#1270](https://github.com/JustVugg/colibri/pull/1270) replaced equal-sized + stripes with bandwidth-weighted chunks. Before that fix, adding a SATA drive + to two NVMe drives reduced GLM-5.2 decode from 0.900 to 0.666 tok/s in the + reported experiment. This is historical evidence, not a current slowdown + attributed to every SATA mirror. +- **The post-fix third drive was neutral on that host.** The contributor's + [follow-up on `dev`](https://github.com/JustVugg/colibri/pull/1270#issuecomment-5466324415) + reported 1.030 tok/s with two NVMe drives and 1.006 with the added SATA drive + (five interleaved runs per arm, reported standard deviations 0.041 and 0.034). + The author judged the difference within run-to-run variation. That Windows + 11 / i9-14900K / RTX 3090 result does not promise a gain from a slower disk + on another machine. + +These are community measurements on specific configurations, not new results +from this guide. [#1137](https://github.com/JustVugg/colibri/issues/1137) contains +the multi-disk discussion and an earlier mixed-speed report. + +## Compare against one drive + +1. Record the commit, model/container, hardware and drive/controller layout. + Keep the prompt, seed, output-token limit, RAM/VRAM budgets, and I/O mode + fixed. Keep background disk work idle. Use the same learned usage history + and cache preparation for each arm, and report cold and warm runs separately. +2. Start a fresh process for each arm. For the primary-only baseline, clear + `COLI_MODEL_MIRROR`, its legacy alias `SNAP_MIRROR`, and any + `COLI_DISK_WEIGHTS` override. For the mirror arm, set only + `COLI_MODEL_MIRROR` to the copies being tested and let the probe choose the + weights. Keep any `COLI_MODEL_DIRS` layout fixed; if it spans drives, label + this a split-layout baseline rather than a single-drive baseline. +3. Set `PROF=1` in both arms. Save startup `[MIRROR]` messages, throughput, + TTFT, expert hit rate, and per-drive `MIRROR:` byte/read counters. The GLM + serve loop emits cumulative profile counters on clean shutdown; do not + assume they appear after every chat response. +4. Interleave at least three repetitions per arm. Report the median and + spread, output/correctness checks, and raw logs, including excluded runs. + If testing `DIRECT=1`, run it as a separate comparison and verify that the + platform and filesystem actually use direct I/O. See the + [benchmarking protocol](benchmarking.md) for cache-state and reporting rules. + +To return to the primary-only configuration in your shell: + +```bash +unset COLI_MODEL_MIRROR SNAP_MIRROR COLI_DISK_WEIGHTS +``` + +```powershell +Remove-Item Env:COLI_MODEL_MIRROR, Env:SNAP_MIRROR, Env:COLI_DISK_WEIGHTS -ErrorAction SilentlyContinue +``` + +PowerShell's `$env:` assignments persist in that session; remove `PROF` too if +you enabled it only for this comparison. diff --git a/docs/qwen36-cuda-tier.md b/docs/qwen36-cuda-tier.md index 57e6a195c..d96bd5aea 100644 --- a/docs/qwen36-cuda-tier.md +++ b/docs/qwen36-cuda-tier.md @@ -46,6 +46,28 @@ OMP_NUM_THREADS= OMP_WAIT_POLICY=ACTIVE OMP_PROC_BIND=close \ SNAP= N_NEW=200 ./c/qwen36 256 4 prompt.txt ``` +### Windows (CUDA_DLL=1) + +MinGW cannot link CUDA directly, so the backend is built into `coli_cuda.dll` +with nvcc + MSVC and `qwen36.exe` reaches it through `backend_loader.c`. +`CUDA=1` is rejected on Windows by design. From an *x64 Native Tools* prompt +with MSYS2's `mingw64\bin` and `usr\bin` on `PATH`: + +```cmd +cd c +make cuda-dll CUDA_ARCH=sm_89 +make qwen36.exe CUDA_DLL=1 ARCH=native +set COLI_CUDA=1 +set COLI_GPUS=0 +set CUDA_EXPERT_GB=auto +qwen36.exe +``` + +Keep `coli_cuda.dll` next to `qwen36.exe`, built from the same checkout, and +the CUDA toolkit's `bin` directory on `PATH` for `cudart`. A startup line +`[gpu] MoE experts -> CUDA VRAM tier` confirms the tier is active; without it +the run is CPU-only. + `cap` (argv[1]) must equal `n_experts` (full RAM residency). int4 containers only (the int8 container keeps the CPU path). `COLI_TIMERS=1` prints per-phase timings and tier telemetry. @@ -60,10 +82,49 @@ the trunk on the CPU is why a 6 GB card sees the hit rate stop mattering (#1040): the GPU is doing the cheap job. By default the engine now places the trunk itself. Before the tier decides its -budget, the engine offers each trunk component with its size (lm_head once, the -fused DeltaNet projection of every DeltaNet layer), and the tier prices them -against the experts they would displace, in **bytes saved on the memory bus per -token, per byte of VRAM**: +budget, the engine offers each trunk component with its size, and the tier +prices them against the experts they would displace, in **bytes saved on the +memory bus per token, per byte of VRAM**. The components, with their int8 size +on Qwen3.6-35B-A3B (hidden 2048, 30 DeltaNet and 10 attention layers): + +| component | what | per layer | total | +|---|---|---|---| +| `lmhead` | the output head, once | | 508 MB | +| `dnproj` | DeltaNet in_proj qkv ++ z, fused | 25.2 MB | 755 MB | +| `dnout` | DeltaNet out_proj | 8.4 MB | 252 MB | +| `attnproj` | attention q, k, v, o (one item, four matrices) | 27.3 MB | 273 MB | +| `shexp` | the shared expert's gate, up, down | 3.1 MB | 126 MB | + +Offer order is the placement priority once the budget runs short: `lmhead`, +then `dnproj`, `dnout`, `attnproj`, `shexp`, each in layer order, so a partial +placement is whole layers. `dnout`, `attnproj` and `shexp` were added after +the M10 datapoint in #1652 showed the un-offloaded dense path as the ceiling +on a CPU without AVX2; measured on a 16-core CPU, of the 37.5 ms a decoded +token spends in the DeltaNet stack 23.4 are the input projections, 8.3 the +out_proj and norm, 3.3 the convolution and 2.4 the recurrence -- the matmuls, +not the recurrence, are what the trunk costs. They are served from VRAM on +decode (one GEMV each); a prompt batch keeps the batched CPU matmul. + +Placed DeltaNet input projections (`dnproj`, qkv ++ z) also run as CUDA +batches during prefill: at most 256 rows per call, with input/output staging +bounded to 32 MiB (or one row if that alone is larger). The convolution and +recurrent state still advance one token at a time. An unavailable or failed +projection uses the existing per-token CPU path. This changes dispatch count, +not the recurrent update order; it is not a full GPU DeltaNet implementation. + +**Measured, not assumed.** The pricing rule below presumes the GPU answers a +GEMV faster than the CPU does. Four Tesla M10 (sm_50, no tensor cores, four +GPUs on one PCIe board) said otherwise in #1652: with every expert +VRAM-resident, placing the trunk made every component slower (lm_head 68.8 ms +against 41.7 on the CPU, the 30 DeltaNet projections 106 against 66) and +decode fell from 3.56 to 2.68 tok/s. So the engine measures before it uploads +a byte of trunk: one DeltaNet input projection is timed both ways on the +device that would host it (ten GEMVs, best of three rounds, after a warm-up), +one `[place] probe:` line reports both times, and if the GPU loses the whole +automatic placement is withdrawn (`[place] trunk stays on the CPU`) and its +bytes go back to the expert budget before the warmstart. Only the automatic +placement is questioned: a hand-written `COLI_PLACE` stands, and +`COLI_TRUNK_PROBE=0` skips the probe. The pricing rule: - a dense component is read every token: 1.0 per byte; - a routed expert is read with the probability a token routes to it -- its heat @@ -82,6 +143,7 @@ of that device's expert budget, and each decision prints as a `[place]` line. | unset or `auto` | automatic, as above | | `off` | nothing placed: experts only (the behaviour before this) | | `lmhead=0,dnproj=0:20+1:20,experts=0` | hand-written list (the measurement tool); obeyed as written, trunk bytes still charged to the budget | +| `dnout=0,attnproj=0,shexp=0` | the same list form for the newer components; any of the five names, per layer or split with `+` | First calibration, one Quadro RTX 4000 (8 GB), per-row int4 container, 200-token decode, same prompt, output bit-identical in all four runs: @@ -157,3 +219,46 @@ particular has no int8 analogue (see **Memory**). CPU-only baseline of this engine before the tier: 0.35 tok/s. Numerics: logits cosine vs the f32 CPU reference 0.9992 (dense int8 on), bit-identical GPU-vs-CPU on the same container (cosine 1.0000001). + +## Resident projection batching microbenchmark + +Build and run a bounded, model-free comparison of `S` resident-int8 one-row GPU +calls with one `S`-row GPU call: + +```sh +make -C c tests/bench_cuda_resident_batch CUDA=1 CUDA_ARCH=sm_89 +c/tests/bench_cuda_resident_batch > resident-batch.jsonl +``` + +Set `CUDA_HOME` if the toolkit is not on PATH and select the architecture for +your device. The harness uses CUDA device 0, uploads each matrix once, warms +both paths twice, then alternates their order over nine paired measurements. +The host wall time includes synchronous input/output copies and compute, but +excludes upload and validation. All output elements are checked between arms +after every pair; sampled elements also have an independent CPU reference. +JSONL retains all nine timings per arm and the median of the paired ratios. +An execution/validation failure exits nonzero; absent CUDA initialization exits +77. The target is explicitly run and is not part of ordinary CI. + +A single run on 2026-09-22 (RTX 4070 12 GB, driver 591.86, CUDA 12.9, `sm_89`, +Core Ultra 9 285K, WSL2 Linux 6.6.114.1, `-O3 -ftz=false`) produced: + +| Input × output | Rows | Serial median ms | Batch median ms | Median paired ratio | +|---|---:|---:|---:|---:| +| 512 × 1024 | 32 | 2.975 | 0.294 | 10.02 | +| 2048 × 2048 | 32 | 3.588 | 0.813 | 4.31 | +| 2048 × 2048 | 128 | 12.716 | 2.913 | 4.37 | + +[Raw paired timings](experiments/qwen36-resident-batch-2026-09-22.jsonl) +include zero sampled CPU-reference error for these deterministic synthetic +inputs. The device was not isolated: telemetry before/after showed 4% GPU +utilization, approximately 3 GB occupied VRAM and 2550 MHz SM clock. Timings show +visible variation; one nine-pair run does not establish reproducibility across +sessions, devices or real activation distributions. + +The serial GPU baseline reflects the projection-call pattern replaced by the +DeltaNet input batching change. These are synthetic shapes, not complete +DeltaNet layers: convolution, recurrent updates, normalization, other projections, +model loading and HTTP queueing are excluded. This is **not** an end-to-end model +speedup, nor an attention-prefill speedup: the old attention prefill path used +CPU batch matmul, which is not an arm in this benchmark. diff --git a/docs/qwen36.md b/docs/qwen36.md index 32cfcdd4f..98dd906e9 100644 --- a/docs/qwen36.md +++ b/docs/qwen36.md @@ -104,6 +104,71 @@ overlap, not this kernel. `tests/test_expert_ffn` holds the numerics. keeps its own path: it uploads the pair-layout int4 and computes misses from the int8 copy. +## The dense trunk: integer dot products + +Every dense GEMV of a token, the DeltaNet in and out projections, the +attention q/k/v/o, the shared expert and lm_head, used to multiply int8 +weights by f32 activations: each weight byte converted to f32 and fed to an +FMA, eight weights per instruction. On a 16-core AVX-512 host lm_head +(248320 x 2048 int8, 508 MB) ran at 29 GB/s on a memory bus that does 80: +the kernel was the limit, and the dense part of the token is 1.9 GB of int8 +on the 35B, three times what the routed experts read. + +Since 1.12.1 the activation is quantized to int8 once per call (one scale, +amax/127, the same contract the expert integer kernels use) and the products +are integer: 32 weights per instruction on AVX2, 64 on AVX-512 VNNI, exact +int32 sums scaled once per output. The routed experts take the same path for +their activations. Measured on Qwen3.6-35B-A3B, 8 threads, 300 decoded +tokens, every expert resident, perplexity on 4 x 512 tokens of English text: + +| | tok/s | ms/token: DeltaNet proj / out, attention, lm_head, expert compute | perplexity | +|---|---|---|---| +| f32 activations (1.12.0) | 6.71 | 21.9 / 8.6, 9.5, 12.6, 22.7 | 13.79 | +| int8 activations, dense trunk (`COLI_DENSE_IDOT=1`) | 7.35 | 17.4 / 6.3, 7.7, 10.2, 22.7 | 13.92 (+1.0%) | +| int8 activations, routed experts (`QWEN_EXPERT_ACT=i8`) | 7.20 | 21.9 / 8.6, 9.5, 12.6, 15.9 | 13.80 (+0.1%) | +| both (the 1.12.1 default) | **8.23** (+22.6%) | 17.2 / 6.3, 7.5, 10.1, 15.6 | 13.97 (+1.3%) | + +Both are the default; `COLI_DENSE_IDOT=0` and `QWEN_EXPERT_ACT=f32` +restore the f32 kernels, which stay bit-identical to their references. + +**int4 for the trunk is opt-in, per component.** `COLI_DENSE_BITS=4` stores +the dense matrices as int4 in blocks of 64 with one scale per block (the +planar layout of the grouped expert kernel) and halves the bytes the token +reads, but the parts of the trunk pay 4 bits very differently, so +`COLI_DENSE_INT4` picks which ones take it: + +| int4 on | tok/s | perplexity | +|---|---|---| +| nothing (int8) | 7.35 | 13.92 | +| `lmhead` | 8.15 (lm_head 10.1 to 8.4 ms; the total is within noise of the cache warming) | 14.13 (+2.4% vs f32) | +| `lmhead,dnproj,dnout` | | 14.56 (+5.6%) | +| `lmhead,dnproj,dnout,shexp` | | 14.94 (+8.3%) | +| everything (`attn` included) | 8.09 | 15.17 (+10%) | + +A least-squares refinement of the block scale was tried and changes nothing +(15.16 against 15.17): the loss is the matrices' sensitivity, not the +quantizer. If you take one, take `lmhead`: 254 MB less per token for the +smallest cost. + +## Cache-aware routing (`CACHE_ROUTE`, off by default) + +The residual misses above are the lever's target. `CACHE_ROUTE=1` ports the +GLM engine's max-rank re-routing ([CACHE_ROUTE.md](CACHE_ROUTE.md), +arXiv:2412.00099) to this engine with two residency levels: inside the top-`M` +window, a slot past the sacred top-`J` prefers an expert already in the VRAM +tier, then one in the RAM cache, then the plain ranking. It is **lossy**: it +changes which experts run, so the semantic contract is off while it is set and +the footer prints what it cost, `route_agree` (overlap with the true top-K) +and `route_kl` (mass KL), next to the swap and hit rates. Unset, the router +is the original loop and the token ids are byte-identical; `ROUTE_AGREE=1` +alone prints the meters at 100 % / 0 without touching routing. + +Qwen3.6 routes top-8 (plus the shared expert), so the default `ROUTE_J=2` +leaves six substitutable slots per token; the tiny fixture routes top-2 and +needs `ROUTE_J<2` to show any swap at all. A/B it the way the GLM doc does: +same prompt and seed, tok/s and hit rate against agreement and KL, and treat +`PPL=1` on a teacher-forced reference as the quality bar. + ## Which container? The gs64 container carries one scale per 64-weight group instead of one per @@ -114,6 +179,19 @@ per-row quantization error concentrates the same way. The gs64 container costs ~1.7 GB more on disk and a few percent on cold-start; warm decode speed is the same or slightly better. +**Mixed: int8 `down`, int4 gate/up.** `convert_qwen36.py --ebits 4 --gs 64 +--down-bits 8` (`--down-gs` for grouped down scales, 0 = per row) writes one +slab per expert with `down_proj` in int8 and gate/up as above -- 5.7 bits per +weight against gs64's 4.5. It is the knob that produced the #1370 numbers on +wikitext-2 (16 x 512 tokens): gs64 7.325, mixed 7.281, all experts int8 7.153, +Ollama's Q4_K_M 7.147 -- `down` alone recovers a quarter of the gap to int8, +the rest sits in gate/up, and at equal bits Q4_K_M's asymmetric quantizer is +ahead. Keep it as a measurement tool and a middle step for boxes with RAM to +spare; it is not the answer to the gap. The engine tells the layout apart by +size and reads each matrix in its own format on the CPU path; the CUDA VRAM +tier takes one format per expert and refuses a mixed container with a line +(`COLI_CUDA=1 ignored`), so such a container runs CPU-only for now. + ## Which checkpoints, and what the banner calls them Two Qwen checkpoints declare `model_type: qwen3_5_moe_text` and resolve to diff --git a/docs/qwen38.md b/docs/qwen38.md index 7d355347b..d67c8ccbd 100644 --- a/docs/qwen38.md +++ b/docs/qwen38.md @@ -3,9 +3,9 @@ `c/qwen38.c` runs the language model in [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) directly from the official safetensors shards. No conversion or second copy of -the weights is required. This contribution is deliberately **text-only**: -Colibri does not load or advertise the checkpoint's vision encoder, and it does -not use the optional MTP layer. +the weights is required. The engine supports text and images through the +checkpoint's vision encoder; see **Vision** below. It does not use the optional +MTP layer. The upstream language model has 125B ordinary parameters with 6B activated, plus a 51B hashed n-gram embedding. It has 48 layers arranged as 12 repetitions @@ -27,7 +27,7 @@ make -C c qwen38 COLI_MODEL=~/Models/Qwen3.8-Flash-Next-FP8 ./c/coli chat ``` -`coli serve` and `coli web` use the same text-only gateway path. Qwen3.8 thinks +`coli serve` and `coli web` use the same gateway path. Qwen3.8 thinks by default; `reasoning_effort` accepts `low`, `medium`, `high`, and `xhigh`, and `enable_thinking: false` emits the model's official empty thinking prefix. Audio and grammar constraints are rejected explicitly. Images are supported; @@ -66,7 +66,6 @@ are FP32; BF16 matrices use FP32 accumulation, while native FP8 uses FP32 dot products within each 128-column block and FP64 accumulation across the scaled blocks. The bounded per-layer cache therefore spends about one quarter of the previous memory per FP8 expert. Context state grows by about 54 KiB per token. -There is currently no Qwen3.8 GPU backend. ## What stays on disk @@ -101,7 +100,12 @@ blocks are pooled and scored, the best 512 blocks are retained, and a causal tail of up to three tokens is appended. The main 24-head attention then operates only on those original tokens. The native model limit is 262,144 tokens; `Q38_MAXT` defaults to 8,192 and may raise the server limit up to that native -ceiling when the required RAM is available. +ceiling when the required RAM is available. `coli serve --ctx N` (and `coli +chat --ctx N`) reach the engine as `Q38_MAXT=N`, not as `CTX`: that is the +variable to look for in the engine process's environment. `max_tokens` is a +ceiling: a budget the prompt leaves no room for is clamped to the room left +(one `[serve] max_tokens ... clamped` line on stderr), and only a prompt that +does not fit is refused. ## Memory and speed @@ -244,8 +248,8 @@ whatever the placer accepts is quantized to **int8 per row** (scale = max|w| / 127, the qwen36 dnproj/lmhead format) when the tier starts -- about 2 s for the 553 matrices, 3.96 GiB on one card -- and answers decode GEMVs from there, one round trip per matmul (`x` up, `y` down; activations stay on the -CPU). Prefill rows (S > 1) and any backend failure take the BF16 CPU path, -which stays in RAM. The placer takes the trunk before the experts (it is +CPU). Prefill rows (S > 1) and any backend failure take the CPU path, which +holds the same int8 rows (see [the trunk on the CPU](#the-trunk-on-the-cpu-int8-rows)). The placer takes the trunk before the experts (it is read on every token), so on an 8 GB card about 2 GB remain for hot experts; `coli plan` prices it the same way (`4.0 GB int8 trunk + ... hot tier`). @@ -255,7 +259,7 @@ read on every token), so on an 8 GB card about 2 GB remain for hot experts; | `Q38_TRUNK_MIN_KB=` | offer matrices of at least n KiB (default 1024; a round trip costs more than a tiny GEMV saves) | | `Q38_TRUNK_SKIP=name,name` | leave the named components on the CPU (bisecting, or a component that does not pay) | | `QT_UPLOAD_SYNC=1` | the tier's `qt_issue` waits for every in-flight upload first (tests and diagnostics: deterministic residency, no upload/compute overlap) | -| `Q38_TRUNK_CPU_INT8=1` | the same int8 rows on the CPU instead -- what the quantization alone does to the output, no GPU needed (perplexity, token parity); a matrix the GPU holds is still answered from VRAM | +| `Q38_TRUNK_CPU_INT8=0` | keep the trunk BF16 on the CPU with the f32 kernel (the numeric reference); the default is int8 rows with integer dot products, GPU or not | | `Q38_TRUNK_SELFTEST=1` | at start, every placed matrix is checked once: GPU GEMV against the same int8 rows on the CPU (relative error printed per matrix) | **Numerics.** GPU int8 against CPU int8 on the same rows: relative error @@ -307,6 +311,75 @@ error, only wrong tokens, worse the more experts were resident. The dense path has its own buffers now ([qwen36-cuda-tier.md](qwen36-cuda-tier.md)); qwen36 never called the dense path inside that window. +### The trunk on the CPU: int8 rows + +Without a GPU the same trunk is the decode's floor: 8 GiB of BF16 read on +every token, multiplied by a scalar loop. Since 1.12.1 the engine keeps the +trunk on the CPU as **int8 rows with one scale per row** (the format the +tier uploads, so GPU or not the rows are the same bytes), releases the BF16 +copy once the rows exist, and multiplies with the integer kernels of +`idot.h`: the activation is quantized to int8 once per row and meets the +weights with maddubs / vpdpbusd (AVX2 / AVX-512 VNNI) or NEON dot products. +Decode and prefill take this path. Every matrix of at least +`Q38_TRUNK_MIN_KB` (default 1 MiB) is in; the small ones stay BF16 because +they cost nothing either way. `Q38_TRUNK_CPU_INT8=0` keeps the BF16 rows and +the f32 kernel, the numeric reference. + +The routed experts stay e4m3 with their 128 x 128 block scales, but the +kernel decodes eight bytes at a time in registers and multiplies with FMA +(`q38_matmul_fp8_vec`); the block scale still applies once per block and the +blocks still add in double, so it differs from the table kernel only by the +float summation order inside a block. For a batch of prefill rows the block +is decoded once and every row runs through it. `Q38_FP8_KERNEL=scalar` +restores the table kernel. + +**Measured** on the released Qwen3.8-Flash-Next-FP8, a 16-core server shared +with a training job, `OMP_NUM_THREADS=8`, RAM LRU cap 96 per layer, 100 +generated tokens, `COLI_TIMERS=1` decode bank: + +| | BF16 trunk, table FP8 (1.12.0) | BF16 trunk, vector FP8 | int8 trunk, vector FP8 (default) | +|---|---:|---:|---:| +| decode, tok/s | 0.61 | 0.77 | **1.42** | +| resident-mm (dense trunk), ms/token | 434 | 437 | 85 | +| lm-head, ms/token | 76 | 79 | 13 | +| routed-expert GEMV, ms/token | 388 | 154 | 140 | +| shared-expert, ms/token | 43 | 44 | 15 | +| expert-read (page cache), ms/token | 156 | 155 | 140 | +| peak RSS, GB | 32.2 | 32.2 | 28.5 | + +The middle column reproduces the first's text byte for byte over the 100 +tokens: the vector kernel is a faster way to compute the same thing. The +int8 trunk is a quantization, and its cost is measured as perplexity, not +assumed. Teacher-forced NLL on four 512-token chunks of the repository +docs (8 prompt tokens, 504 scored), same machine: + +| chunk | BF16 trunk | int8 trunk | BF16 trunk, vector FP8 | +|---|---:|---:|---:| +| 0 | 1.4527 | 1.4416 | 1.4527 | +| 1 | 2.4594 | 2.4544 | | +| 2 | 2.3845 | 2.3929 | | +| 3 | 2.7463 | 2.7753 | | +| mean nats/token | 2.2607 | 2.2661 | | + ++0.24% nats, about +0.5% perplexity, two chunks lower and two higher: the +per-row int8 error is noise at this size. The vector FP8 kernel alone +reproduces the BF16 run to four decimals. + +Prefill takes the same paths, and it is where the scalar BF16 loop hurt +most. The 512-token chunk 0 as a prompt, one generated token: + +| | BF16 trunk, table FP8 | int8 trunk, vector FP8 | +|---|---:|---:| +| wall, 512 prompt tokens | 495 s | 149 s | +| resident-mm | 205 s | 12 s | +| routed-expert GEMV | 172 s | 34 s | +| shared-expert | 15.5 s | 1.8 s | +| expert-read | 28 s | 26 s | + +The trunk quantization takes 3.3 s at load for the 553 matrices and the +RSS after load grows by 0.6 GB while the int8 rows and the BF16 copy +coexist; once the BF16 is released the peak RSS of a run is 3.7 GB lower. + ## Performance telemetry Every served request reports its own routed-expert cache hit rate; persistent @@ -363,12 +436,9 @@ checkpoint. They are not redistributed by Colibri. The released checkpoint is multimodal and the engine already reads its `model.language_model` prefix, so the vision tensors are present and reachable. -The tower itself is not implemented yet; images are still refused rather than -silently dropped. - -What exists today is the half that decides whether vision is *correct* rather -than nearly correct: `tools/qwen38_image.py`, pinned against the official -`Qwen2VLImageProcessor` in `tests/test_qwen38_image.py`. +The tower and gateway image path are implemented; their checks are described +below. Image preprocessing is checked with `tools/qwen38_image.py`, pinned +against the official `Qwen2VLImageProcessor` in `tests/test_qwen38_image.py`. ``` python3 tests/test_qwen38_image.py --config /preprocessor_config.json diff --git a/docs/tuning.md b/docs/tuning.md index 4d221ced7..f9ed37dbc 100644 --- a/docs/tuning.md +++ b/docs/tuning.md @@ -203,6 +203,50 @@ COLI_CUDA=0` if you also want kernel-family/GPU independence. Acceptance percentages are not comparable across engine versions under `--topp` ([#163](https://github.com/JustVugg/colibri/issues/163) has the full story). +## Approximate mode: `DEGRADE_ZERO` (opt-in, OLMoE-calibrated) + +`DEGRADE_ZERO=1` enables an opt-in degraded inference policy: when a prefetch +deadline is missed, experts whose per-position gate weight falls below +`DEGRADE_TAU` (default 0.03) are **zero-filled instead of loaded from disk**. +The slot contributes nothing to the layer output; the approximation is the +dropped mass, not a rescaled version of it (renorm is catastrophically worse — +see issue #865 for the measured A/B). + +This reduces blocking disk reads on NVMe-bound workloads at the cost of a small +quality hit. Measured on OLMoE-1B-7B: + +| `DEGRADE_TAU` | slots zeroed | ppl delta | +|---|---|---| +| 0.03 | ~22% | +2.9% | +| 0.05 | ~60% | +41% | + +**These numbers are OLMoE-specific.** GLM-5.2 (`norm_topk=1`) and Kimi K3 have +different router contracts and expert counts — their operating points have not +been measured. Until they are, treat `tau=0.03` as a starting point and verify +quality on your model before relying on it. + +**These numbers assume a warm expert cache.** Cold-start sessions — where the cache +begins empty and all experts miss initially — will see higher drop rates until the LRU +fills. The steady-state perplexity delta above is what was measured; cold-start transient +behavior has not been separately characterized. + +The feature is decode-only (`S≤4` guard, same as `EXPERT_BUDGET`): dropping +experts during prefill corrupts the KV cache. A rescue rule ensures no token +position is left with zero routed experts. Resident (pinned or LRU-cached) +experts are never dropped regardless of weight. + +The `[PROF]` footer reports the total zeroed slot count and the top-3 layers by +drop share when the flag is active, so a miscalibrated tau is visible rather +than silent. + +```bash +DEGRADE_ZERO=1 DEGRADE_TAU=0.03 COLI_MODEL=/nvme/glm52_i4 ./coli chat +``` + +See [ENVIRONMENT.md](ENVIRONMENT.md) for the full variable reference and +[issue #865](https://github.com/JustVugg/colibri/issues/865) for the +measurement methodology. + ## Conversations reopen warm `coli chat` persists the compressed MLA KV-cache to disk after every turn diff --git a/flake.lock b/flake.lock index a89029e8b..355b0fccf 100644 --- a/flake.lock +++ b/flake.lock @@ -20,11 +20,11 @@ }, "nixpkgs": { "locked": { - "lastModified": 1784280462, - "narHash": "sha256-DtoqIqM7VkR6NxAkcLpMwmi02USwWb3JdmNGLyhthc0=", - "owner": "NixOS", + "lastModified": 1789149629, + "narHash": "sha256-H6GwaZzZf+4npqv0tph94w9tZddSjFjmQrVsW0z78uk=", + "owner": "nixos", "repo": "nixpkgs", - "rev": "293d6abedf0478e681a4dfcfcb35b30fc796a32f", + "rev": "eaad089433ca2bb662274377d33df3d0e51ef28b", "type": "github" }, "original": { diff --git a/flake.nix b/flake.nix index d1a3a8482..d82817abc 100644 --- a/flake.nix +++ b/flake.nix @@ -105,6 +105,8 @@ # defers to a future `make install` that stages them itself. [ -e "$out/lib/colibri/v4_dsml.py" ] || \ install -m 644 c/v4_dsml.py "$out/lib/colibri/" + [ -e "$out/lib/colibri/v41_dsml.py" ] || \ + install -m 644 c/v41_dsml.py "$out/lib/colibri/" [ -e "$out/lib/colibri/tools/iq3xxs_grid.json" ] || \ install -m 644 c/tools/iq3xxs_grid.json "$out/lib/colibri/tools/" diff --git a/site/index.html b/site/index.html index 292ba59c7..ed5bc7422 100644 --- a/site/index.html +++ b/site/index.html @@ -488,7 +488,7 @@

Get up and running on hardware you already own