diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml
index e3daad8e8..c25acd068 100644
--- a/.github/workflows/check.yml
+++ b/.github/workflows/check.yml
@@ -23,7 +23,9 @@ jobs:
linux:
name: Linux
runs-on: ubuntu-latest
- timeout-minutes: 15
+ # make check takes about 14 minutes here now (14.9 at worst this week);
+ # at 15 the job was cut off with nothing failed.
+ timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- name: make check
@@ -128,13 +130,21 @@ jobs:
python c/tools/make_edge_tiny_tokenizer.py --vocab-size 281 /tmp/glm53_stream-i4
cd c && COLI_GLM53_FIXTURE=/tmp/glm53_stream-i4 python -m unittest -v tests.test_glm53_dashboard
COLI_GLM53_FIXTURE=/tmp/glm53_stream-i4 python -m unittest -v tests.test_glm53_context_exceeded
+ # The two stdlib-only oracles used to be argparse scripts that `make
+ # test-python` collected as zero tests (#1700). They run here, on the
+ # fixtures generated above, token-exact against transformers in f32.
+ - name: Run the GLM-5.3 tiny oracles
+ run: |
+ cd c && GLM53_TINY=/tmp/glm53_tiny GLM53_MM_TINY=/tmp/glm53_mm python -m unittest -v tests.test_glm53_oracles
macos:
# clang; libomp for the threaded path (Makefile falls back to
# single-threaded automatically if it's ever missing).
name: macOS (colibri + V4 platform gate)
runs-on: macos-latest
- timeout-minutes: 15
+ # `make check` here now takes about 14.5 minutes; at 15 the job was cut
+ # off inside the Python suite with no test failed, and showed as red.
+ timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- name: install libomp
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index c9af31a35..1471b14da 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -147,7 +147,7 @@ jobs:
musl:
name: musl libc (Alpine)
runs-on: ubuntu-latest
- container: alpine:3.21
+ container: alpine:3.24
steps:
# #1430: colibri did not compile against musl, because malloc_trim is a
# glibc extension guarded on __linux__ rather than on __GLIBC__. Nothing in
@@ -547,6 +547,31 @@ jobs:
echo "cap=$cap: identical ($(cat ids_${cap}_1.txt))"
done
make tests/test_expert_ffn && ./tests/test_expert_ffn
+ - name: Mixed expert layout (int4 gs64 gate/up, int8 down) loads and is cap-independent
+ run: |
+ cd c
+ # The #1370 experiment knob: one slab per expert with int4 gate/up and int8
+ # down (2*inter*hidden bytes). The engine must tell it apart from int4 and
+ # int8 by size, read every matrix in its own format, and give the same ids
+ # at every cache capacity (cap=1 recycles the single slot after each expert,
+ # which is where a wrong slab offset would show). The ids are not compared to
+ # the torch oracle: int4 gate/up drift from f32 by design, as in the A/B above.
+ python3 tools/convert_qwen36.py --model qwen36_tiny64 --out qwen36_tiny64_d8 --ebits 4 --gs 64 --down-bits 8
+ python3 - <<'PY'
+ import json; m = json.load(open("qwen36_tiny64_d8/qwen36_meta.json"))
+ assert m["expert_down_bits"] == 8 and m["expert_down_gs"] == 0 and m["expert_gs"] == 64, m
+ PY
+ for cap in 1 2 8; do
+ COLI_DENSE_I8=0 SNAP=qwen36_tiny64_d8 ./qwen36 "$cap" 4 qwen36_tiny64/ref_full.json > mixed_$cap.log 2>&1 || true
+ grep -q "expert format on disk: int4 gate/up + int8 down" mixed_$cap.log || { echo "FAIL: mixed layout not detected at cap=$cap"; cat mixed_$cap.log; exit 1; }
+ grep -E "^C engine" mixed_$cap.log > mixed_ids_$cap.txt
+ test -s mixed_ids_$cap.txt || { echo "FAIL: no ids at cap=$cap"; cat mixed_$cap.log; exit 1; }
+ done
+ cmp mixed_ids_1.txt mixed_ids_8.txt && cmp mixed_ids_2.txt mixed_ids_8.txt || { echo "FAIL: ids differ across caps"; cat mixed_ids_*.txt; exit 1; }
+ echo "mixed layout: identical ids at cap 1/2/8 ($(cat mixed_ids_8.txt))"
+ # COLI_CUDA=1 on a mixed container is refused with a line, the CPU path stands
+ COLI_CUDA=1 COLI_DENSE_I8=0 SNAP=qwen36_tiny64_d8 ./qwen36 8 4 qwen36_tiny64/ref_full.json > mixed_cuda.log 2>&1 || true
+ grep -q "COLI_CUDA=1 ignored: the VRAM expert tier does not take the mixed layout" mixed_cuda.log || { echo "FAIL: tier refusal line missing"; cat mixed_cuda.log; exit 1; }
- name: A malformed container is refused, not read
run: |
cd c
@@ -843,7 +868,9 @@ jobs:
# the oracle fixture has no tokenizer; serve mode needs one
python3 tools/make_edge_tiny_tokenizer.py --vocab-size "$(python3 -c 'import json;print(json.load(open("qwen38_tiny/config.json"))["vocab_size"])')" qwen38_tiny
make qwen38 >/dev/null
- QWEN38_TINY=qwen38_tiny python3 -m unittest -v tests.test_qwen38_dashboard
+ QWEN38_TINY=qwen38_tiny python3 -m unittest -v tests.test_qwen38_dashboard tests.test_qwen38_brio
+ python3 tools/make_edge_tiny_tokenizer.py --vocab-size 64 qwen38_tiny_fp8
+ QWEN38_TINY=qwen38_tiny_fp8 python3 -m unittest -v tests.test_qwen38_brio
inkling-oracle:
name: Inkling oracle (token-exact vs transformers)
@@ -923,10 +950,15 @@ jobs:
/tmp/tike
- name: Generate the glm_tiny fixture
run: cd c && python3 tools/make_glm_oracle.py
- - name: Token-exact oracle (teacher forcing)
- # The fixture is only trustworthy if the engine reproduces it, so assert
- # that before reading anything else out of a run.
- run: cd c && SNAP=./glm_tiny TF=1 COLI_TEMP=0 ./colibri 64 16 16
+ - name: Oracle (30–32/32 teacher forcing + exact greedy)
+ # Preserve CONTRIBUTING.md's two TF near-tie allowance. Greedy remains
+ # exact, and invalid/non-finite results cannot consume the allowance.
+ run: |
+ cd c
+ SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16
+ SNAP=./glm_tiny COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16
+ - name: Oracle allowance and failure regressions
+ run: cd c && python3 tests/test_glm_oracle.py
- name: Structural efficiency tests
# test_cpu_vs_cpu_tok_s_stability is NOT in this list: it is a tok/s
# bound (two runs within 25%) over a ~15 ms tiny replay -- it measured
@@ -1204,6 +1236,30 @@ jobs:
- name: Python test suite
run: cd c && python3 -m unittest discover -s tests -p 'test_*.py'
+ gguf:
+ name: GGUF reader + converter tests
+ runs-on: ubuntu-latest
+ steps:
+ - uses: actions/checkout@v4
+ - uses: actions/setup-python@v5
+ with:
+ python-version: '3.12'
+ cache: pip
+ cache-dependency-path: c/tools/requirements-gguf.txt
+ # This job is what makes the numerical GGUF evidence real: without these
+ # deps those tests skip at import, so the generic `python` job alone only
+ # exercises test_gguf_reader.py (pure stdlib).
+ - name: Install GGUF test dependencies
+ run: pip install -r c/tools/requirements-gguf.txt
+ - name: GGUF reader, dequant, profile and converter tests
+ run: |
+ cd c
+ python3 -m unittest -v \
+ tests.test_gguf_reader \
+ tests.test_gguf_dequant \
+ tests.test_gguf_olmoe_profile \
+ tests.test_convert_gguf_to_olmoe
+
windows-python-focused:
name: Python tests (Windows focused)
runs-on: windows-latest
@@ -1229,7 +1285,7 @@ jobs:
# This job builds every engine on arm64 (NEON compile coverage), runs the
# integer-kernel bit-exactness gate with the NEON branches live, and replays
# the glm_tiny teacher-forcing oracle against a fixture generated on THIS
- # runner (same-machine torch reference, so no cross-ISA float excuses).
+ # runner. As on x86, allow two TF near ties; greedy remains exact.
oracle-arm:
name: ARM (engines + NEON kernel exactness + tiny oracle)
runs-on: ubuntu-24.04-arm
@@ -1245,6 +1301,12 @@ jobs:
run: pip install -r c/tools/oracle-requirements.txt
- name: Build every engine (NEON branches must compile)
run: make -C c colibri inkling kimi_k3 olmoe
+ - name: "DeepSeek V4 FP4 expert kernels: NEON arm bit-exact against the scalar arm (#1696)"
+ run: |
+ cd c
+ # batch (prefill) and S=1 (the matvec the decode uses)
+ bash tools/bench_fp4_matmul.sh 32 256 128 1
+ bash tools/bench_fp4_matmul.sh 1 256 128 1
- name: Integer-kernel exactness, NEON branches live (#1081)
run: |
cd c
@@ -1252,5 +1314,10 @@ jobs:
/tmp/tike
- name: Generate the glm_tiny fixture on this runner
run: cd c && python3 tools/make_glm_oracle.py
- - name: Token-exact oracle on ARM (teacher forcing)
- run: cd c && SNAP=./glm_tiny TF=1 COLI_TEMP=0 ./colibri 64 16 16
+ - name: Oracle on ARM (30–32/32 teacher forcing + exact greedy)
+ run: |
+ cd c
+ SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16
+ SNAP=./glm_tiny COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16
+ - name: Oracle allowance and failure regressions on ARM
+ run: cd c && python3 tests/test_glm_oracle.py
diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml
index 1ef3d3af7..4afa56adc 100644
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -84,7 +84,7 @@ jobs:
- uses: actions/setup-node@v4
with:
- node-version: '20'
+ node-version: '22'
cache: npm
cache-dependency-path: web/package-lock.json
diff --git a/.gitignore b/.gitignore
index d1564447e..eeba7b078 100644
--- a/.gitignore
+++ b/.gitignore
@@ -30,6 +30,8 @@ c/deepseek_v4.exe
c/deepseek_v41
c/deepseek_v41.exe
c/COLI_V4_UNIT_*.o
+c/deepseek_v4.cflags
+c/deepseek_v4.cudaflags
# ...and the ownership-test objects, which the same Makefile puts in a build/
# subdirectory (V4_OWN_DIR) rather than next to the sources. #868 caught the
# twelve in c/, not the four in here, so `make check` still left `?? c/build/`.
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 39d9bb452..e6a5f546d 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -3,6 +3,395 @@
All notable changes to colibrì are documented here.
Format follows [Keep a Changelog](https://keepachangelog.com/).
+## [1.12.1] — 2026-09-24
+
+95 pull requests since v1.12.0, 82 of them from contributors. Two tokenizers
+brought back to the reference, brio on the ninth engine, `coli chat` working
+again at the default context on two families, and a placement decision that
+is now measured on the card in front of it instead of predicted.
+
+### Tokenizers, measured against the reference
+
+- **#1654**: qwen36 tokenized differently from HF `tokenizers` in two ways.
+ An added token right after punctuation was encoded as text (`X.<|im_end|>`
+ was 7 tokens instead of 3, every chat turn ending in punctuation paid +4,
+ #1653), and a whitespace run followed by a non-space was one piece where
+ the regex's `\s+(?!\S)` leaves the last char to the next one, so every
+ indented line of code tokenized differently. Measured on the real
+ vocabulary: 2,803 lines and blocks of code, Markdown, Chinese and
+ Japanese went from 757 identical to 2,803, with 5.4% fewer tokens.
+- **#1656**: OLMoE's `tokenizer.json` has no Split, a bare ByteLevel with
+ `use_regex`, for which HF runs the original GPT-2 pattern; `tok.h` applied
+ cl100k. A GPT-2 family in `tok.h`: 1,560/1,708 identical before, 1,708/1,708
+ after. The same measurement on GLM-5.2/5.3/5.3-Flash, DeepSeek V4 and V4.1,
+ Inkling and Qwen3.8 came back identical on every case.
+
+### Brio and the serve contract
+
+- **#1662**: `POST /v1/systemone`, the request and the reply of TypeSafe's
+ Jev API, served by the brio channel: a client written for it points at
+ colibri and changes the base URL. `noul` is a yes/no question, `choice`
+ scores the labels with their descriptions in the text, `score` the level
+ numbers with the expected value and the legend; `confidence` by their
+ documented formula. Any `model` name is accepted on that route. Measured on
+ the real Qwen3.6: the three-question example of the docs in 1m46 with the
+ state read once.
+- **#1655**: the DeepSeek V4 engine speaks the numeric channel (`logprobs=k`,
+ `pin=1`, `max_tokens=0`), so `/v1/brio` works on the ninth engine instead
+ of answering 500 (#1648). The head that used to keep only its argmax now
+ returns the whole row; `ECHO` per prompt position during prefill, the
+ prompt-end scores kept with a state snapshot on `pin`, and a logprob tail
+ on every `DATA` frame during generation. The tiny fixture pins that the
+ best `ECHO` token equals the greedy token from the same prefix, and that
+ the pinned predictor equals a cold prefill's.
+- **#1659**: qwen36 and qwen38 refused a request when `prompt + max_tokens`
+ exceeded the context, and the gateway's default budget for these two
+ families is 8192, the whole default context: every request without
+ `max_tokens` and every `coli chat` message answered 400 on a two-token
+ prompt (#1641). `max_tokens` is now a ceiling, clamped to the room the
+ prompt leaves, as GLM and DeepSeek V4 already did; only a prompt that does
+ not fit is refused. `docs/api.md` states the rule, `docs/qwen38.md` names
+ `Q38_MAXT` as the variable `--ctx` becomes.
+
+### The dense trunk in VRAM, measured before it is placed
+
+- **#1657**: qwen36 offers the rest of its dense trunk to the VRAM placer:
+ the DeltaNet out_proj (`dnout`), the attention q/k/v/o (`attnproj`) and
+ the shared expert (`shexp`), about 650 MB more of int8 on the 35B beside
+ `lmhead` and `dnproj`. Measured on four Tesla M10 by the reporter of
+ #1652, every placed component ran slower than the CPU (lm_head 68.8 ms
+ against 41.7), so the engine now times one GEMV both ways at startup and
+ withdraws the whole automatic placement when the GPU loses, giving the
+ VRAM back to the experts: `auto` equals `off` on that box, byte-identical
+ output. A hand-written `COLI_PLACE` stands; `COLI_TRUNK_PROBE=0` trusts
+ the placer.
+
+### Performance
+
+- **#1664**: qwen36's dense trunk and routed experts multiply with integer
+ dot products. The activation is quantized to int8 once per call and the
+ weights, int8 rows or int4 planar blocks, meet it with maddubs / vpdpbusd
+ instead of a float conversion per weight; the integer kernels move from
+ `quant.h` into `idot.h`, shared by every engine. Measured on the 35B, 8
+ threads, every expert resident: decode 6.71 to 8.23 tok/s (+22.6%), lm_head
+ 12.6 to 10.1 ms/token, the expert compute 22.7 to 15.6, for +1.3%
+ perplexity on 4 x 512 tokens. Both are the default (`COLI_DENSE_IDOT=0`,
+ `QWEN_EXPERT_ACT=f32` restore the f32 kernels). `COLI_DENSE_BITS=4` with
+ `COLI_DENSE_INT4=` stores part of the trunk as int4 in blocks
+ of 64: opt-in, with the perplexity it costs per component in the docs
+ (lm_head alone +2.4%, everything +10%).
+- **#1668**: qwen38's dense trunk (553 matrices, 3.6 G weights, 8 GiB of
+ BF16 read on every token, more than the ten routed experts) is kept on the
+ CPU as int8 rows with the BF16 copy released, and multiplied with the same
+ integer kernels; the routed experts' e4m3 blocks are decoded eight at a
+ time in registers and multiplied with FMA instead of one table lookup per
+ weight. Measured on the released Qwen3.8-Flash-Next-FP8, 8 threads, RAM
+ LRU 96 per layer: decode 0.61 to 1.42 tok/s, the trunk 434 to 85 ms/token,
+ lm_head 76 to 13, the expert GEMVs 388 to 140, peak RSS 32.2 to 28.5 GB;
+ prefill of 512 tokens 495 to 149 s. Perplexity on 4 x 512 tokens +0.5%
+ (two chunks lower, two higher); the vector FP8 kernel alone reproduces the
+ BF16 run to four decimals. Both are the default (`Q38_TRUNK_CPU_INT8=0`
+ keeps the BF16 trunk, `Q38_FP8_KERNEL=scalar` the table kernel).
+
+### Performance, from contributors
+
+- **#1606**: the K1b grouped int4 family gets a multi-row tile and AVX-512
+ and AMX arms, and is no longer switched off on AVX-512 builds; exact on
+ all eight engines on a 16-core AVX-512 host.
+- **#1239**: an SSE4.1 tier for the olmoe and qwen36 int8 GEMV, for hosts
+ with SSE4.1 but no AVX2; on the Sandy Bridge of #1652 decode went from
+ 2.45 to 3.54 tok/s.
+- **#1313**: `matmul_fp8` computes four output rows per pass under clang,
+ where the contraction makes it bit-exact; GCC keeps the one-row kernel.
+- **#1612**: qwen36 gains the GLM engine's `CACHE_ROUTE` lever, with the
+ VRAM tier as the first residency level, opt-in.
+- **#906**: `DEGRADE_ZERO`, an opt-in policy that zero-fills a missed
+ expert slot below a gate-weight threshold instead of blocking on the
+ load (#865).
+- **#1677**: qwen36 projects a prompt's DeltaNet inputs (qkv and z) on the
+ card in blocks of up to 256 rows instead of one row at a time; a 259-row
+ prefill makes 2 projection calls instead of 259, with the convolution
+ history and the recurrent state checked against the CPU run. The paired
+ microbenchmark on an RTX 4070 read 4 to 10x per projection.
+- **#1674**: qwen36's attention projections the tier placed in VRAM answer
+ a whole prompt batch with one call per matrix; a failed call turns only
+ that handle off and the prompt continues on the CPU.
+- **#1676**: Kimi K3's streaming CUDA expert keeps the gate, up and SiTU
+ intermediates on the device and applies down there (one fused entry
+ point, optional in the DLL: an older backend keeps the three-call path).
+- **#1673**: the streaming MXFP4 matmul reuses one grow-only device scratch
+ per card instead of allocating and freeing weights and scales on every
+ call.
+- **#1559** (kreuzzelg): `convert_qwen36.py --down-bits 8` writes the mixed
+ expert layout, int4 gs64 gate/up and int8 down in one slab (5.7 bits per
+ weight against gs64's 4.5); the engine tells it apart by size and reads
+ each matrix in its own format on the CPU path, and refuses the VRAM tier
+ with a line. It is the knob behind the #1370 numbers: on wikitext-2 the
+ int8 down alone recovers a quarter of the gap between gs64 and all-int8,
+ the rest sits in gate/up. A measurement tool and a middle step, not the
+ answer to the gap.
+- **#1730** (mfethe1): DeepSeek V4's FP4 expert kernels, the prefill batch
+ and the decode matvec, get a NEON arm; arm64 used to take the scalar
+ arm, which is why Apple Silicon prefilled at decode speed (#1696). Bit
+ identical to the scalar arm, and the ARM CI job now checks that on every
+ change; 17 to 22x on the kernel at the V4 expert shapes on an M-series
+ Mac, as measured by the author.
+- **#1716** (jtinbergen): qwen36 quantizes its dense weights to int8 while
+ loading instead of keeping an f32 copy first, and converts f16/bf16 with
+ SIMD. On the 35B the resident set after load goes from 9.2 to 4.8 GB;
+ the generated text and the perplexity are identical to before.
+- **#1286** (cameron): the grouped int4 GEMV and the fused gate/up GEMV get
+ an SSE4.1 arm for CPUs without AVX2 (Ivy Bridge and older). It vectorizes
+ across output rows, so each lane runs the scalar row's exact sequence and
+ the result is bit-identical; forced-SSE4.1 tests at -O1, -O3 and without
+ FP contraction pin it. 2.2 to 2.7x on the isolated kernel on a dual
+ E5-2680 v2.
+- **#1686** (DebugSultan): qwen38's prefill chunk (`Q38_PREFILL_BATCH_ROWS`)
+ and workspace (`Q38_PREFILL_WORKSPACE_MIB`) are runtime knobs, the expert
+ load batch is no longer capped at top-k, and the QSA ranking and
+ attention run per position in parallel at prefill. Measured on the
+ released checkpoint on top of the int8 trunk: output byte-identical at
+ every chunk width, no speed change on our 16-core server; the knobs are
+ there for hardware where the chunk binds.
+
+### Fixed
+
+- **#1650** (bokiko): a Qwen3.8 pin snapshots the recurrent and PLE state
+ but reuses the live attention and indexer rows; after an unrelated prompt
+ overwrote those rows, returning to the pin could change brio logprobs
+ without a warning. The engine now records the token identity of the live
+ rows (`kv_prefix.h`) and refuses a stale pin or prefix restore; image rows
+ are tainted. Wire regressions run on the BF16 and FP8 fixtures.
+- **#1626**: `SNAP` is the model directory for every non-GLM engine, so
+ `coli run` stops handing them a leftover environment (#1600).
+- **#1604**: glm53 honours `Mat.resident` in the Vulkan gate and frees the
+ Vulkan handle in `mat_release`.
+- **#1321**: glm53 sizes its expert cache around the model rather than
+ around `MemAvailable`, which the page cache had been inflating.
+- **#1588**: qwen36 refuses loudly on a failed encode-buffer realloc instead
+ of writing through NULL.
+- **#1546**: `coli convert` routes OLMoE to `convert_olmoe_merged.py`.
+- **#1630**: olmoe emits the `ROUTE_TRACE` records it announced; the stream
+ used to be a zero-byte file.
+- **#1610**: `v41_dsml.py` is staged during installation (and the nix flake
+ bumped).
+- **#1511**: the GPU test suite builds under HIP on gfx1151.
+- **#1658**: `test_mem_available` compared two reads of available memory
+ with `==` and failed on a busy Windows runner; a quarter of a GB of
+ tolerance.
+- **#1670**: DeepSeek V4.1 read only its argv cache cap, so `RAM_GB=120`
+ on a 128 GB box left the engine at eight expert slots per layer and
+ 23.8 GB of RSS (#1666). With `--cap` omitted, `coli chat`, `coli serve`
+ and `coli web` now size the cache from the resource plan, with `RAM_GB`
+ or `--ram` as the budget; an explicit `--cap`, a measured profile and an
+ auto-tier plan keep precedence.
+- **#1671**: DeepSeek V4.1 treats `max_tokens` as a ceiling like the other
+ engines (#1641): a fitting prompt with a large request generates what
+ the context leaves, a score-only prompt may fill the context, and a
+ prompt one token over it is refused instead of silently truncated.
+- **#1675**: resident MXFP4 tensors on CUDA carried O float scales where
+ the kernel reads O x ceil(I/32) exponent bytes: short buffers were
+ over-read and long ones truncated. One format-aware size for upload,
+ refresh, accounting and release.
+- **#1678**, **#1679**, **#1680**, **#1682**, **#1683**, **#1684**: the
+ Qwen CUDA tier's lifecycle, end to end. Shutdown releases every resident
+ expert, projection handle and host table after parked callers resume;
+ a failed gate, up or down upload frees what it had already allocated;
+ the expert budget charges the three scale buffers at their own sizes
+ (two experts used to be admitted where one fit); a failed result
+ collection stops inference instead of publishing a partial MoE sum; a
+ failed or explicitly disabled tier start unwinds its storage and
+ synchronization objects, and a second init cannot overwrite a running
+ tier; the CUDA backend validates the whole device list before touching
+ state and keeps live contexts on a repeated init. Fault-injected on the
+ fake backend, then run together on an RTX 4070 under compute-sanitizer
+ with zero errors and zero bytes leaked.
+- **#1669**: `test_systemone_api` imports its scoring engine relative to
+ its package, so an installed `tests` package no longer breaks discovery.
+- **#1697** (kevin9327): the dashboard redesign had dropped the reasoning
+ stream: thinking tokens arrived on `delta.reasoning_content` and vanished,
+ the bubble stayed empty until the answer and a stop during thinking lost
+ the turn. The stream is read again, rendered as its own folding block,
+ counted in the rate and the time to first token, and a unit test pins the
+ split.
+- **#1693** (namespaceMarcello): with `PILOT` on, OLMoE could read the same
+ expert twice, once from the prefetcher and once from the forward pass,
+ into two slots; a slot being read now keeps a reservation in the index
+ (the `colibri.c` pattern) and the second caller waits for the first read
+ to publish. Three model-free scenarios pin it.
+- **#1695** (namespaceMarcello): the prefill echo state and `serve_echo` sit
+ under the same `QWEN36_NO_MAIN` guard, so the segment build no longer
+ warns about a function it never gets; the full build is byte-identical.
+- **#1724** (GenericRikka): Qwen3.6 decoded ``, `` and the
+ tool tags to nothing, because they live only in the tokenizer's
+ `added_tokens`; with thinking on, the closing tag never reached the
+ gateway and the whole answer came back as `reasoning_content`. The
+ non-special added tokens are decoded now; special ones such as
+ `<|im_start|>` still decode to nothing.
+- **#1734** (tarazum): stopping `coli serve` closes the engine's stdin and
+ waits for it to exit on its own before the hard-stop ladder, so the
+ engine's teardown runs; qwen36 never saved its `HEAT_FILE` under `coli
+ serve` (#1733). On Windows the gateway handles SIGBREAK and the engine
+ runs in its own process group.
+- **#1726** (kevin9327): Inkling measured no RAM on Windows and sized its
+ expert cache to 16 per layer; it uses the shared probe now, which on
+ Linux and macOS reads the same numbers as before.
+- **#1735**: on GNU Make 3.81, the system make on macOS, `.build-config`
+ was never written and every build relinked (#1732).
+- **#1731** (bokiko): the DeepSeek V4 CUDA object rebuilds when the nvcc
+ command changes, so a new `CUDA_ARCH` no longer links the old object.
+- **#1728** (crichalchemist): `make test-c VK=1` built 43 test binaries
+ without the Vulkan object; they link it now.
+- **#1712** (kevin9327): Kimi K3, Inkling and OLMoE now treat `max_tokens`
+ as a ceiling like the other engines; `coli chat`'s default of 16384
+ answered 400 on every Kimi and Inkling message against their 8192-token
+ window.
+- **#1713** (kevin9327): `coli plan`, `doctor` and `--auto-tier` export the
+ variable that actually sizes the expert cache on Kimi K3
+ (`K3_EXPERT_GB`) and GLM-5.3 (`GLM53_EXPERT_GB`); only `RAM_GB` was
+ exported, which neither engine reads as the cache size.
+- **#1711** (kevin9327): on Windows, a Kimi K3 `CUDA_DLL` build and a HIP
+ host were refused by `--gpu` as CPU-only; the probe reads the backend DLL
+ name the host was built with, and the launcher maps `--gpu` onto Kimi's
+ `K3_CUDA`.
+- **#1710** (kevin9327): a `tools[]` entry whose `function` is not an object
+ answered HTTP 500 from the GLM and DeepSeek renderers; it is the 400 that
+ `generation_options` already had.
+- **#1721** (monotophic): every frame the gateway writes to the engine is
+ checked, short writes are completed, and a failed `CANCEL` or `STOP`
+ drops the request's pending entry and answers a named 500 instead of a
+ silent close.
+- **#1714**, **#1719** (benmaster82): the brio options form pins the shared
+ state, so per-question requests on one document read it once; the web
+ page can stop a scoring run, and duplicate options are removed on both
+ clients.
+- **#1709** (kevin9327): regenerating a turn with pictures sends them again
+ and leaves the composer alone.
+- **#1646** (Stamina9): qwen38 says once, on stderr, why the parallel expert
+ read path is not taken (disabled, cache smaller than the route, no FP8
+ scale bank, repeated expert, converted layout).
+- **#1708** (wittchen): every `VK=1` build of glm53 failed to compile on a
+ misplaced parenthesis.
+- **#1707** (namespaceMarcello): the DeepSeek V4 unit objects rebuild when
+ the build flags change, so a CUDA engine build followed by `make test-c`
+ no longer links the wrong objects (#1702).
+- **#1715** (namespaceMarcello): seven GLM-5.3 harnesses matched the
+ unittest glob and counted as zero tests; they are renamed, a skip exits 2,
+ the two tiny oracles run in CI, and a discovery test catches the next
+ empty module (#1700).
+- **#1622**: DeepSeek V4's REAP checkpoints store each expert as six
+ per-matrix records; the engine read them through buffered pread and
+ counted every one as a direct-I/O fallback (36% of expert reads on the
+ 150B, #1615). Each segment now goes through the aligned direct window,
+ with a regression on a generated per-matrix fixture.
+- **#1597**: a replayed tool call whose `arguments` parsed as JSON but was
+ not an object (`"[1, 2]"`, `"5"`) answered HTTP 500 from the GLM renderers
+ before the engine was asked anything; it renders the call without
+ arguments, as every other renderer already did.
+- **#1624**: the five gcc 13 warnings left in `make check` are gone, and
+ `st_index_load` refuses an index path that would not fit its buffer
+ instead of opening a truncated one, with a long-path case in the tests.
+- **#1651**: a `pyflakes` pass over the launcher, autotune, the family
+ registry and the qwen36 converter: a `measure()` defined twice, a
+ `readline` import without a fallback, a stray f-string, dead variables.
+- **#1580**: `make qwen36 CUDA_DLL=1` on Windows reached GNU make's implicit
+ rule and built a CPU-only binary; a bare `qwen36` alias, a `.build-config`
+ prerequisite so a CUDA_DLL change rebuilds, a loader-against-header parity
+ test, and the Windows CUDA tier documented.
+- **#1556**: `coli plan` on macOS said "no supported GPU detected" on every
+ Mac; it now lists the Metal device by name, as identity only, without
+ pretending unified memory is a VRAM budget.
+- **#1691**: the installed launcher invoked as `/bin/coli` or `/sbin/coli`
+ on a merged-/usr system derived `/libexec/colibri` instead of
+ `/usr/libexec/colibri`, because `abspath` kept the alias (#1689,
+ florin65's patch): `realpath` first. A test runs the launcher through
+ such an alias, and another checks that every root module the launcher
+ reaches is in the `make install` list, the gap #1610 closed by hand.
+
+### Tools and the gateway
+
+- **#1425**: a general GGUF reader, pure stdlib, and a converter from GGUF
+ OLMoE checkpoints to a colibri container, with the numerical evidence in
+ its own CI job.
+- **#1497**: opt-in prompt-injected tool calling for the families without
+ native tool tokens (OLMoE, Qwen3.6), behind `COLI_TOOL_FALLBACK=1`, with a
+ two-turn end-to-end test.
+- **#1355**, **#1357**: durable per-request results and strict stdout
+ classification in the eval harness, and the logprob-gap check gated on
+ the engine preamble.
+- **#1687**: `GET /metrics` in Prometheus text format, behind the API key:
+ four gauges, six outcome counters and four histograms (queue wait, slot
+ occupancy, first output, engine call), no request labels, no new
+ dependency. The admission scheduler distinguishes completion, failure
+ and cancellation, lets a request use a free slot that no earlier waiter
+ reserved, and joins the keepalive pump before the slot is released.
+- **#1717** (enitimeago): the web chat offers Continue on the last
+ assistant message when it stopped at the token limit, by hand or on an
+ error, and only when `/health` says the server continues assistant
+ turns (#1699).
+- **#1402** (enitimeago): a request whose last message is a non-empty
+ `assistant` turn continues that turn instead of answering in a new one, on
+ `/v1/chat/completions` and `/v1/messages`, for all nine families (Kimi K3
+ frames the open turn engine-side); the prompt ends inside the turn as the
+ official template renders it without a generation cue. On by default,
+ `COLI_CONTINUE_ASSISTANT=0` restores the old behaviour; refused together
+ with tools or a turn ending in whitespace, with a 400 that says why. Each
+ renderer is pinned against the vendored template (#1401).
+- **#1102** (monotophic): checkpoint-faithful FP8 containers that store
+ `kv_b_proj` as fmt=8 could load but not decode attention; the absorb path
+ now decodes fmt=8 on CPU (bit-exact against the reference) and CUDA
+ (within the documented tolerance), and the kv_b sharding refuses by name
+ the formats it cannot serve, which also closes two silent misreads of
+ fmt=5 and fmt=6.
+- **#1395** (rybruscoe): `COLI_EXACT_VERIFY=1` makes the speculative verify
+ batch token-exact against sequential decode, at a measured cost on the
+ dot itself; off by default, the default path is unchanged.
+- **#1720** (monotophic): a request carrying `seed` is accepted and the seed
+ ignored, as `docs/api.md` now says, instead of a 400; no determinism is
+ implied.
+- **#1605**: `ORACLE_STRICT=1` makes a GLM oracle comparison exit non-zero
+ when it fails, token-exact by default with `ORACLE_TF_MAX_MISMATCHES` for
+ the documented teacher-forcing allowance; references are validated before
+ the comparison and non-finite logits cannot pass. Both oracle CI jobs run
+ real-process regressions against it.
+- **#1705**: `tools/benchmark_baseline.py`, a collection protocol on top of
+ the HTTP harness for a repeated three-engine serving baseline: one frozen
+ manifest (hardware, model and template identity, per-engine launch
+ settings, cache and speculation policy), a rotating plan over a
+ concurrency matrix, one collector per engine and round that manages no
+ server, and a comparison that keeps failed and missing cells visible and
+ distinguishes matched artifacts from deployment comparisons. No results
+ are bundled and no ranking is emitted.
+- **#1688**: `tools/benchmark_http_serving.py`, a stdlib HTTP streaming
+ benchmark over fixed JSONL conversations: closed-loop or paced arrivals
+ (periodic or Poisson, seeded), warmup separated from measurement,
+ first-output and duration SLOs, latency percentiles and usage-based
+ token throughput; a truncated or malformed stream is a failure, not a
+ sample.
+
+### Docs
+
+- **#1639**, **#1644**: the README shows brio mode and the dashboard as it
+ is: the workspace, the Brain page (the measured expert atlas as a cortex,
+ and a region inside it) and the Profiling page, in four languages;
+ `docs/api.md` describes the four pages instead of the old console.
+- **#1492**, **#1643**: a Japanese README, and its banner at the shipping
+ version, which the banner test now checks in every language.
+- **#1634**: the multi-disk guide states measured gains and limits instead
+ of "twice the bandwidth", with Bash and PowerShell examples.
+- **#1617**: connecting the pi coding agent to `coli serve`.
+- **#1619**: `expected_bytes` identity versus physical extent for
+ int4-rans256-g0 (#1273).
+- **#1618** (bherald): `docs/qwen38.md` no longer calls the engine text-only;
+ the vision tower and the gateway image path shipped in 1.12.0.
+- **#1649**, **#1647** (Suraj2105-1): the musl CI job runs on Alpine 3.24
+ and the release pipeline on Node.js 22, ahead of the 3.21 and Node 20
+ end of life.
+- **#1568** (Yoruxyv): an Indonesian translation of the dashboard.
+- **#1681** (XBold): the README and `docs/qwen38.md` no longer say Qwen3.8
+ has no GPU backend; the CUDA VRAM expert tier and the int8 trunk in VRAM
+ shipped in 1.12.0.
+
## [1.12.0] — 2026-09-20
81 pull requests since v1.11.0. A new way to ask a model a closed question,
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index 28cebe5da..687351c7a 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -5,7 +5,7 @@ Keep changes focused and preserve Colibri's dependency-free default CPU path.
## Branches
- **`main`** is the stable branch. It's what users clone, and it stays known-good
- (engine always passes the token-exact oracle: `SNAP=./glm_tiny TF=1 ./colibri 64 16 16`,
+ (engine always passes the oracle: `SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16`,
run from `c/`). The `glm_tiny` fixture is generated, not committed --
`python3 c/tools/make_glm_oracle.py` builds it (needs torch).
- **`dev`** is the integration branch. **Open your PR against `dev`.** Reviewed PRs
@@ -13,9 +13,37 @@ Keep changes focused and preserve Colibri's dependency-free default CPU path.
it into `main`. This keeps `main` clean instead of taking every PR one at a time.
Every PR — on either branch — is reviewed for a clean build (0 warnings), the oracle
-(~30-32/32 TF depending on floating-point near-ties + 20/20 greedy), and its own
+(30–32/32 TF depending on floating-point near ties + 20/20 greedy), and its own
targeted validation before merge.
+After generating the fixture, run both enforced comparisons from `c/`:
+
+```sh
+SNAP=./glm_tiny TF=1 COLI_TEMP=0 ORACLE_STRICT=1 ORACLE_TF_MAX_MISMATCHES=2 ./colibri 64 16 16
+SNAP=./glm_tiny COLI_TEMP=0 ORACLE_STRICT=1 ./colibri 64 16 16
+python3 tests/test_glm_oracle.py
+```
+
+`ORACLE_STRICT=1` makes a failed comparison exit with status 1. Strict mode is
+token-exact by default. The teacher-forcing command above explicitly allows at
+most two mismatches, preserving the 30–32/32 acceptance range for this fixture;
+greedy comparison always requires every continuation token to match. Non-finite
+output and incomplete generation fail regardless of the mismatch allowance.
+Invalid reference arrays/JSON fail in either mode.
+Without strict mode (or with `ORACLE_STRICT=0`), a completed comparison remains
+report-only for diagnostic/benchmark callers; its exit status is not a correctness
+gate. Strict mode rejects `REPLAY`, `CONSIST`, serving, text generation, and other execution
+modes that would bypass the comparison.
+
+`ORACLE_TF_MAX_MISMATCHES` is a nonnegative integer smaller than the number of
+TF positions; it is used only for strict teacher-forcing comparisons. Unset or
+`0` requires exact TF agreement. Generate the reference on the test host and
+inspect near ties with `TF=1 DEBUG_LOGITS=1`; an allowed mismatch is still printed
+and does not establish token-exact agreement. The regression script checks the
+two-mismatch boundary, optional exact mode, and rejection of a single greedy
+mismatch. It validates the original TF fixture within the same allowance. Routine
+`make check` covers reference validation without needing torch or a generated model.
+
## Local checks
Run the lightweight checks locally:
diff --git a/README.it.md b/README.it.md
index a7de181fe..86dbed613 100644
--- a/README.it.md
+++ b/README.it.md
@@ -4,7 +4,7 @@
Discord ·
- English · 简体中文 · 繁體中文 · Italiano
+ English · 简体中文 · 繁體中文 · Italiano · 日本語
**Motore piccolo, modello immenso.** Esegui **modelli MoE di frontiera — da 744
@@ -34,7 +34,7 @@ ma non ridefinire il modello di nascosto.
```
$ ./coli chat
- 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
+ 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU
✓ ready in 32s · resident 9.9 GB
› ciao!
◆ Ciao! 😊 Come posso aiutarti oggi?
diff --git a/README.ja.md b/README.ja.md
new file mode 100644
index 000000000..9f4bbdca9
--- /dev/null
+++ b/README.ja.md
@@ -0,0 +1,707 @@
+
+
+
+
+
+
+
+
+
+
+ Website ·
+ Discord ·
+ English · 简体中文 · 繁體中文 · Italiano · 日本語
+
+
+**小さなエンジン、巨大なモデル。** ストレージ・RAM・VRAM を単一の推論階層として扱う
+(AI メモリのマルチティア化)ことで、**744B から 2.8T パラメータのフロンティア MoE モデル**を、
+コンシューマー向けや異種混在のハードウェア上で、エンジン依存ゼロの純粋な C で実行します。
+
+現在動作するのは 9 つのファミリーです: **GLM-5.2/5.3**(744B)、**GLM-5.3-Flash**(321B、
+ビジョン対応)、**Inkling**(975B)、**Kimi K3**(2.8T)、**DeepSeek V4 Flash**(284B)、**DeepSeek V4.1 Flash**(552B、ビジョン対応)、
+**Qwen3.8-Flash-Next**(125B + 51B n-gram)、**Qwen3.6**(35B-A3B)、そして
+**OLMoE**(7B)——
+それぞれが C ファイル 1 つで、同じ `coli chat` / `coli serve` / `coli web` フロントエンドを共有します。
+[全モデル一覧 ↓](#other-supported-models)
+
+> **Colibrì は今日すぐに動かせる推論エンジンであり、同時にオープンな研究
+> プラットフォームでもあります。** 主な目標は、ソフトウェアとハードウェアの境界全体——
+> モデルフォーマット、メモリ階層、ストレージ I/O、配置、スケジューリング、カーネル、
+> 投機的デコード、CPU/GPU のオーバーラップ——にわたって推論側の性能を追求し、
+> 大規模モデルが希少なハードウェアに依存せず、より低コストで動くようにすることです。
+
+Colibrì は VRAM・RAM・ストレージを単一のマルチティア階層として扱い、意図的に
+攻めたシステム上のアイデアを試す場となっています。そのため **速度に SLA はありませんが、
+セマンティクスは厳格に保証します**。実験は再現可能なエンドツーエンドの計測によって
+採用に値することを示さなければならず、デフォルトのポリシーは **モデルの精度やルーターの
+セマンティクスを黙って変更することは決してありません**。高速メモリが不足すると速度は
+落ちるかもしれませんが、それによってモデルが密かに別物になることはあってはなりません。
+
+```
+$ ./coli chat
+ 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU
+ ✓ ready in 32s · resident 9.9 GB
+ › ciao!
+ ◆ Ciao! 😊 Come posso aiutarti oggi?
+```
+
+## 動作の様子
+
+
+
+
+Web ダッシュボード(./coli web): 744B モデルが 4 tok/s、TTFT 1.6 秒、ディスク 0 で動作 —
+6× RTX 5090 上でエキスパートを完全常駐させ、ライブのトークンメトリクス、ターンごとの時間内訳、
+VRAM/RAM/ディスクのティアバー、隅にはライブのミニ脳を表示しています。
+
+
+
+
+Brain ページの Explore 表示: GLM-5.2 の計測されたエキスパートアトラスを皮質として描画します。
+特性が明らかになった 13,260 個のエキスパートが 10 の領域(Python、SQL、数学、詩、法律、中国語…)に分かれ、位置は学習された埋め込みではなく
+計測されたルーティング親和性です。領域を選ぶとその中に入れます。Live routing 表示は実際に動いているモデルに切り替わり、
+エキスパートごとに 1 セル、色はストレージのティア、1 ターンでルーティングされたエキスパートは白く光ります。
+
+
+
+
+Python 領域の内部: 1,142 個のエキスパートが星座として並び、それぞれにレイヤーと番号のラベルが付きます。パネルはその 1 つ、
+レイヤー 17 のエキスパート 178 を表示しています。エントロピー 3.13 のジェネラリストで、計測された親和性は Python 20.2%、JSON 14.6%、
+会話 14.2%、SQL 13.3% です。
+
+
+
+
+Profiling ページ: エンジンが各ターンで時間を使う場所をフェーズごとに示し、直近 30 ターンを推移として表示します。
+ここでは CPU マシン上の Qwen3.6: プロンプト 36 トークンと生成 55 トークンで壁時計時間 19.0 秒、2.9 tok/s、
+ディスクサービス 11.4 秒は計算と重なっています。
+
+## 研究のミッション
+
+Colibrì があれば、プライベートなフロンティアモデルへのアクセスが、ハイパースケーラー級ハードウェアの入手可能性に制限されることはありません。
+
+マルチティア機能によって、Colibrì は **推論エンジンのパイプラインを積極的に最適化し、
+プロプライエタリなハードウェアへの依存を取り除きます**。
+
+私たちの実務上のミッションには、重みの表現方法と移動方法を変えること、何を VRAM・RAM・
+ストレージに置くかを決めること、異種計算資源をオーバーラップさせること、起動と同期の
+オーバーヘッドを減らすこと、スパース性と再利用を活用すること、そして新しいデコード
+アルゴリズムを試すことが含まれます。慣習的だからという理由だけで守られるものはなく、
+マイクロベンチマークで速く見えるという理由だけで採用されるものもありません。決め手となるのは
+実機でのエンドツーエンドの推論結果であり、スループット、レイテンシ、メモリ、コストと
+並んで、正しさと品質も計測されます。
+
+その実際的な帰結が **アクセシビリティ** です。すでに持っているハードウェアで
+744B パラメータのモデルを動かし、すべてのエキスパートが発火する様子をリアルタイムで眺め、
+それを実現しているコードを変更できます。API の向こうにある知能を借りるのではなく、
+それを *手にする* ——調べ、計測し、改善する——のです。エンジンは意図的に小さく保たれており、
+次の有用な最適化は、それを計測しようとする誰からでも生まれ得ます。
+
+## コア技術と計測結果
+
+- **ティア容量に縛られない単一の階層。** VRAM・RAM・NVMe は同じ重みを置く配置ティアです。
+ 高速メモリの制約は速度を変えるだけで、モデルのセマンティクスは変えません。
+- **重みのための JIT。** すべてのエキスパートをロードする代わりに、計測されたルーティングの熱量が
+ レイヤーごとの LRU、学習されたピン留めホットストア、1 レイヤー先のプリフェッチを駆動します。
+ 繰り返しのあるワークロードでは効果がありますが、履歴は過学習し得るし、先読みは一部の
+ ホストでは逆効果になり得るため、どちらも約束ではなく計測可能なポリシーとして扱われます。
+- **I/O はエンジンの一部。** バッチ化されたエキスパートの和集合、読み込みと計算のオーバーラップ、
+ `O_DIRECT`、重み付きデュアル SSD ストライピングによって、ストレージのレイテンシが
+ タダであるかのように装うのではなく、ストリーミング経路そのものに取り組みます。`O_DIRECT` は
+ ドライブ依存であり、デュアル SSD はまだより幅広いコミュニティによるエンドツーエンドの A/B を必要としています。
+- **異種混在実行。** CPU、CUDA、Metal、NUMA メモリ、部分的または完全なエキスパート常駐は
+ 1 つのランタイムを共有し、マシンに応じて組み合わせられます。どの組み合わせが有利かは、
+ 計算能力、帯域幅、常駐状況、ワークロードによって決まります。
+- **別のモデルにすることなく状態を圧縮。** トークン単位で完全一致するフォワード検証、
+ 57 分の 1 に縮小された MLA KV 状態、永続化されたウォームな会話、忠実な DSA により、
+ 最適化を正しさに結び付けています。これらはメモリ・レイテンシ・正しさの特性であり、
+ 一律のスループット向上を主張するものではありません。
+- **元が取れる場合にだけ使う投機的デコード。** ネイティブ MTP と文法強制ドラフトは
+ エンドツーエンドで計測され、受理率が検証コストに見合わない場合は無効化できます。
+
+## 未検証の仮説、実験、そして協力の方法
+
+Colibrì は、制御されたエンドツーエンドの A/B が示すまで、最適化を仮説として扱います。
+現在の主な問いは次のとおりです:
+
+| 仮説 | これまでのエビデンス | まだ必要な実験 |
+|---|---|---|
+| ルーティング履歴は単純な LRU よりもうまくエキスパートを配置できる | 学習されたピンは繰り返しのワークロードを改善するが、プロンプトに過学習し得る | コーディング、チャット、多言語、長コンテキストのワークロードにわたる、ホールドアウトかつセッション横断の A/B |
+| 複数の SSD は独立した帯域幅をデコード速度に変えられる | 重み付きミラー/分割ルーティングは実装・検証済みで、帯域幅モデルは妥当 | 実際に独立したコントローラ上での、コールドキャッシュ状態の 1 ドライブ対 2 ドライブによる GLM-5.2 実行 |
+| ハードウェアを考慮したプランナーは、各マシンの最適構成に自動で近づける | RAM/VRAM の予算といくつかのバックエンドは現在すでに検出される | 生成されたプランを、ラップトップ、ワークステーション、NUMA ホスト、マルチ GPU システムにわたる制御されたパラメータスイープと比較する |
+| ロスレスまたは品質上限付きの表現で、重みの移動を意味のあるほど減らせる | 正しさ/品質ゲート付きのフォーマットと量子化のアブレーションが存在する | 圧縮率だけでなく、品質、移動バイト数、レイテンシ、有用トークンあたりのコストを同時に再現する |
+| ルーティングを考慮した投機的デコードは、ほぼ完全常駐に達する前でも元が取れる | MTP と文法ドラフトは動作するが、MTP はエキスパートヒット率約 85% 付近で 32% の損失も計測されている | 受理率、エキスパートヒット率、バッチの和集合、ドラフト深さにわたる損益分岐面をマッピングする |
+| CPU/GPU のオーバーラップは、ボトルネックを移すだけでなく転送と同期を隠蔽できる | CUDA と Metal での改善はあるが、高速な CPU や低い常駐率ではそれが打ち消され得る | PCIe、ユニファイドメモリ、完全常駐マシンにわたる、ステージごとのプロファイルと 1 変数ずつの A/B |
+
+協力したいですか? いずれかの行を選び、ネガティブな結果も公開してください。ハードウェア、
+コミット、モデル/コンテナ、正確なコマンド、プロンプト、キャッシュ状態、スループット、
+TTFT、エキスパートヒット率、読み込みバイト数、品質チェックを記録し、1 つの変数だけを変えて
+再実行し、生のログを添付してください。まずは
+[CONTRIBUTING.md](CONTRIBUTING.md) から始め、
+[ベンチマークプロトコル](docs/benchmarking.md) と比較したうえで、
+[実験 issue を作成](https://github.com/JustVugg/colibri/issues/new) してください。
+ここでは、説明のつかない速い数値よりも、よく制御された失敗のほうが価値があります。
+
+## アイデア
+
+744B の Mixture-of-Experts モデルは、1 トークンあたり約 40B のパラメータしか活性化せず、
+そのうちトークンごとに入れ替わるのは約 11 GB(ルーティングされるエキスパート)だけです:
+
+
+
+
+
+つまり、モデルは高速メモリに *収まる* 必要はなく、**配置** されればよいのです:
+
+- **密な部分**(アテンション、共有エキスパート、埋め込み — 約 17B パラメータ)は
+ **int4 で RAM に常駐** します(約 9.9 GB)。
+- **19,456 個のルーティングエキスパート**(75 MoE レイヤー × 256 + MTP ヘッド、int4 で各約 19 MB)は
+ **ディスク上** に置かれ(約 370 GB)、レイヤーごとの LRU キャッシュ、学習されたピン留めホットストア、
+ オプションの VRAM ティアとともに **オンデマンドでストリーミング** されます。
+
+コアアルゴリズムは **重みのための JIT** だと考えてください。コンパイラの JIT は
+プログラム全体をコンパイルすることはなく、実際に実行される部分を観察して、ホットパスを
+ジャストインタイムでコンパイルします。colibrì は 744B のパラメータ空間に対して同じ賭けをします。
+パラメータは保持すべき常駐状態ではなく、ルーターが必要だと証明したまさにそのときに、
+異種混在のストレージ階層(VRAM / RAM / NVMe)にわたって **ステージングされるデータ** なのです。
+計測されたルーティングの熱量がどのエキスパートをどのティアに置くかを決め、ルーターは
+1 レイヤー先を走ってプリフェッチがステージングのレイテンシを隠し、そして JIT と同様に、
+エンジンはあなたのワークロードを学習します。使えば使うほど、適切なエキスパートがホットになっていきます。
+これが機能するのは、ルーティングに計測可能な構造があるからです(
+[エキスパートアトラス](https://github.com/JustVugg/colibri/issues/175) を参照)。
+そして構造はキャッシュ可能です。
+
+エンジンは単一の C ファイル(`c/colibri.c`)と小さなヘッダ群だけで構成されています。BLAS も、
+実行時の Python も、GPU も必要ありません。
+
+### ローカルクラスタモード
+
+コーディネーターはトークン生成、ルーティング、KV 状態をローカルに保持し、ディスクを
+バックエンドとするエキスパートワーカーが、ルーティングされた FFN を他の Mac 上で実行します。
+あるレイヤーでルーティングされたバッチの和集合は 1 つの永続 TCP リクエストとして送られるため、
+1 トークンでエキスパートごとにラウンドトリップが発生することはありません。
+
+オプションの登録サービスを起動します:
+
+```bash
+./coli cluster coordinator --host 0.0.0.0 --port 8765
+```
+
+各ワーカーで、同じ変換済みモデルをローカルに用意したうえで:
+
+```bash
+./coli cluster worker --model /nvme/glm52_i4 --port 9100 \
+ --coordinator http://COORDINATOR:8765 --advertise-host WORKER_IP
+```
+
+ディスカバリーを使ってコーディネーターを実行するか、静的な構成では `--cluster-workers
+HOST:PORT,...` を指定します:
+
+```bash
+./coli serve --model /nvme/glm52_i4 \
+ --cluster-coordinator http://127.0.0.1:8765
+```
+
+ワーカーが設定されていない限りトランスポートは無効なので、既存の単一マシンでの経路は
+変わりません。密レイヤーのシャーディングやブラウザ/WebGPU ワーカーは、別の今後の拡張ポイントです。
+
+## 仕組み
+
+### トークンごとの経路
+
+
+
+
+
+すべてのトークンのすべてのレイヤーが、同じ 5 つのステップをたどります。設計上の目標は、
+**配置が決めるのは常に速度だけ** ということです。エキスパートが VRAM から応答しようと
+ディスクから応答しようと、ルーターの判断も重みの精度も同じです。
+
+### 1 つのメモリ要件ではなく、1 つのメモリ階層
+
+
+
+
+
+### デュアル SSD: モデルのコピー 2 つで、読み込み帯域幅を 2 倍に
+
+ほとんどのマシンでデコードはディスク律速であり、エキスパートの読み込みは読み取り専用です。そこで **2 台目の SSD** があるなら、そこにモデルの完全なコピーを置き、エンジンに両方のドライブから同時にストリーミングさせましょう:
+
+```bash
+COLI_MODEL=/fast/glm52_i4 COLI_MODEL_MIRROR=/second/glm52_i4 ./coli chat
+COLI_DISK_WEIGHTS=9,3 ... # オプション: プライマリ,ミラーの帯域幅比(未指定なら起動時に計測)
+```
+
+各エキスパートは、2 台のドライブの計測された(または宣言された)帯域幅で重み付けされた決定論的ハッシュによって一方のドライブに割り当てられます。そのため readahead/PILOT のプリフェッチと要求時の読み込みは常に同じドライブに当たり、二重にキャッシュされることはありません。合計帯域幅は両ドライブの和になります — 9 GB/s + 3 GB/s の組み合わせでは、高速なドライブ単体よりもエキスパートの読み込みが約 33% 速くなり、OMP 並列のピン留め/ウォームアップのロードも両方からストリーミングされます。知っておくべき詳細:
+
+- ミラーは **起動時に検証** されます(ファイルごとのサイズと safetensors ヘッダがプライマリとバイト単位で一致する必要があります)。一致しない、または欠けているファイルは黙ってプライマリのままになるので、**部分的なミラーでも問題ありません** — 一部のシャードしか置けない小さめの 2 台目 SSD でも効果があります。
+- ミラーには **一切書き込まれません**: `.coli_usage`、`.coli_kv` およびすべてのサイドカーファイルはプライマリに残ります。
+- ミラーでの読み込みエラーはプライマリにフォールバックします(警告 1 回、クラッシュなし)。そのため実行中に 2 台目のドライブを抜いても、サーバーが落ちるのではなく性能が低下するだけです。
+- ルーティングがトークンを変えることはありません — 両コピーはバイト単位で同一であり、実行ごとの `MIRROR:` 統計行にはドライブごとに提供した GB 数が表示されます。
+
+同じエンジンがあらゆる規模をカバーします。25 GB のラップトップではすべてがディスクから
+ストリーミングされ(遅いが正しい)、大きなホストではエキスパート全体が常駐し
+(`CUDA_EXPERT_GB=auto PIN_GB=all`)、ディスクはデコード経路から完全に外れます。
+ティアの間には **学習キャッシュ** があります。エンジンは *あなたの* ワークロードがどの
+エキスパートにルーティングされるかを記録し(`.coli_usage`、ターンごとに更新)、最もホットな
+ものを自動でピン留めします — colibrì は文字どおり、使えば使うほど速くなります。マルチソケットの
+ホストでは、`COLI_NUMA=1` によって常駐する重みをメモリコントローラ間でインターリーブします
+([#82](https://github.com/JustVugg/colibri/issues/82))。
+
+モデル全体を置けない 2 台目のドライブ向けに、Colibri はすでに学習しているエキスパート履歴から
+部分ミラーの優先順位を付けられます。まずいくつかの代表的なプロンプトを実行して `.coli_usage` に
+ワークロードを反映させてから、ミラーを計画・ステージング・検証します:
+
+```bash
+./c/coli mirror plan --model /fast/glm52_i4 --mirror /second/glm52_i4 \
+ --budget-gib 200 --reserve-gib 20
+./c/coli mirror stage --model /fast/glm52_i4 --mirror /second/glm52_i4 \
+ --budget-gib 200 --reserve-gib 20
+./c/coli mirror verify --model /fast/glm52_i4 --mirror /second/glm52_i4
+```
+
+プランナーは safetensors ヘッダを直接読み、`COLI_MODEL_DIRS` から分割モデルのディレクトリをたどり、
+最もホットなルーティングエキスパートを提供できるシャードを優先します。ステージングがプライマリの
+モデルを変更することはありません。一時ファイル経由でコピーし、指定された空き容量の予備を確保し、
+すべてのシャードを SHA-256 で検証し、既存のミラーシャードを削除することはなく、選択された
+ミラーの準備が整って初めてレシートをアトミックに公開します。
+
+### ディスクを二度待たない
+
+ミスはコストが高いため、エンジンは工夫の大部分をミスの回避とオーバーラップに費やします。
+各エキスパートの 3 つの行列は隣接して格納され、1 回の `pread` で読み込まれます。上限付きの
+非同期 I/O プール(`PIPE=1`、デフォルト)は、常駐しているエキスパートが計算している間に
+欠けているエキスパートをロードします。バッチ化された位置では一意なエキスパートを 1 回だけ読み込み
+(**batch-union**)、ルーター先読みスレッド(`PILOT=1`)が次のレイヤーのエキスパートを
+プリフェッチします — ルーティングは **1 レイヤー先を 71.6% の精度で予測可能** であることが計測されています。
+GPU では、常駐パイプライン(`COLI_CUDA_PIPE=2`)が残差ストリームをレイヤーをまたいでデバイス上に
+保持するため、CPU のエキスパートループは中断されずに実行されます。Apple Silicon では実験的な
+[Metal バックエンド](docs/metal.md) がユニファイドメモリ GPU 上でバッチ化されたエキスパート演算を行い、
+[Vulkan バックエンド](docs/vulkan.md) はエキスパートティア、密な射影、MLA アテンションのコアを、
+Vulkan 1.2 ドライバを持つあらゆる GPU にもたらします — Mesa/RADV 経由の AMD カードも含まれます
+(RX 580 のようにベンダーのスタックがサポートを終了したカードでは唯一のバックエンドであり、
+RDNA4 では ROCm と互角です — [ベンチマークに関するメモ](docs/vulkan.md) を参照)。
+
+> **実際の NVMe では `DIRECT=1` を計測してください。** O_DIRECT はページキャッシュをバイパスし、
+> DRAM キャッシュと帯域幅に余裕のあるドライブでは大きな改善になることが多いです(Blackwell/Windows
+> マシンで `PIPE=1` と併用してデコード +34% を計測、GB10 の iobench で 4.25→9.69 GB/s)。
+> ただしドライブ依存であり、QLC/DRAM レスや仮想化されたディスクでは効果なし〜逆効果になり得ます。
+> まず試して、ハードウェアが報いてくれる設定を残してください。
+
+### 忠実なモデル、圧縮された状態
+
+フォワードパスは `transformers` のオラクルに対して検証されています(teacher-forcing で
+通常 30〜32/32。小さなオラクルの 2 つの位置は浮動小数点上のほぼ同値で、ツールチェーンに依存します)。
+MLA アテンションは圧縮された KV 状態を保存します — 1 トークンあたり 32,768 個ではなく 576 個の
+浮動小数点数(**57 分の 1**)— そしてそれを再起動をまたいで永続化します(`.coli_kv`)。
+会話は再プリフィルなしでウォームな状態から再開され、中断されなかったセッションとバイト単位で同一です。
+DSA スパースアテンション(GLM-5.2 の lightning indexer)は忠実に実装されており、全キーを強制的に
+選択させたときに密なアテンションを正確に再現することで検証されています。
+
+### 投機的デコードを、誠実に
+
+GLM-5.2 のネイティブ MTP ヘッドがトークンをドラフトし、メインモデルが 1 回のバッチ化された
+フォワードでそれを検証します — 効果がある場合はフォワードあたり 2.2〜2.8 トークン。苦労して
+得た 2 つのルールがデフォルトとして組み込まれています。MTP ヘッドは **int8** でなければならないこと
+(int4 のヘッドは受理率が 0〜4% に崩壊します、[#8](https://github.com/JustVugg/colibri/issues/8))、
+そしてドラフトと検証は **同じ関数** を計算しなければならないこと — `SPEC_PIN=1` は両者を
+1 つのカーネルファミリーに固定します(詳しい調査の経緯は [#163](https://github.com/JustVugg/colibri/issues/163) にあります)。
+文法強制ドラフト([`GRAMMAR=file.gbnf`](docs/grammar-draft.md))は、制約付き JSON 出力で
+ほぼタダで受理率を上げます。投機的デコードが正味でプラスになるかはキャッシュの温まり具合に
+依存します — 計測し、効果がなければ `DRAFT=0` を使ってください。
+
+## 何を実現しているか
+
+
+
+
+
+同じエンジン、同じ int4 コンテナ — ハードウェアが変えるのはエキスパートの置き場所だけです。
+[完全なベンチマーク表](docs/benchmarks.md) からのハイライト:
+
+- **6× RTX 5090、完全常駐:** デコード 5.8〜6.8 tok/s、TTFT 約 13 秒
+ ([実験ログ](docs/experiments/glm52-6x5090-2026-07-12.md))
+- **128 GB の CPU のみのデスクトップ:** ウォーム時 約 1.8 tok/s([#200](https://github.com/JustVugg/colibri/issues/200))
+- **RTX 5070 Ti 1 枚のラップトップ級マシン:** GPU 常駐パイプラインで 1.07 tok/s
+ ([#273](https://github.com/JustVugg/colibri/issues/273))
+- **25 GB の開発マシン:** コールド時 0.05〜0.1 tok/s — このプロジェクトが始まった実証済みの下限であり、
+ 今も誠実なベースラインです。
+
+品質は仮定ではなく計測されています。int4 コンテナの量子化コストと、スケール粒度/回転の
+アブレーションは [docs/benchmarks.md](docs/benchmarks.md#quality-benchmark) と
+[#108](https://github.com/JustVugg/colibri/issues/108)/[#81](https://github.com/JustVugg/colibri/issues/81) にあります。
+
+## はじめに
+
+必要なものは 2 つです: **プログラム**(数百 KB)と **モデル**(372 GB)。
+全プラットフォーム向けの手順は [クイックスタートガイド](docs/quickstart.md) にあります。
+
+### 1. colibri を入手する
+
+**ビルド済みリリースをダウンロード** — Linux、macOS、Windows に対応し、コンパイラは不要です。
+[Releases](https://github.com/JustVugg/colibri/releases) から自分のプラットフォーム用の
+アーカイブを取得して展開します:
+
+```bash
+mkdir colibri && tar xzf colibri-v1.8.0-linux-x86_64.tar.gz -C colibri && cd colibri
+python3 coli info # engine ready ✓
+```
+
+中にはエンジン(`colibri`、Windows では `colibri.exe`)、`coli` ランチャー、その Python
+ヘルパーが入っています。名前の変更や設定は不要です — `coli` は自分の隣にあるエンジンを見つけます。
+必要なのは [Python 3](https://www.python.org/downloads/) のインストールだけです。ランチャーと
+API ゲートウェイは Python スクリプトですが、エンジン自体は依存ゼロの純粋な C です。
+
+**またはソースからビルド** — OpenMP 対応の `gcc`(または clang)が必要です:
+
+```bash
+git clone https://github.com/JustVugg/colibri && cd colibri/c
+./setup.sh # gcc/OpenMP を確認し、ビルドとセルフテストを実行
+```
+
+`coli` を PATH に置きたい場合は、チェックアウトから `pip install -e .` を実行すると登録されます
+(エンジンは引き続き `c/` にあります — wheel ではなく、クローンからの editable インストールです)。
+
+### 2. モデルを入手する
+
+変換済みの **GLM-5.2 int4** コンテナが Hugging Face にあります — **int8 MTP ヘッド** 付きの
+**グループスケール(gs64)** ビルドを使ってください。サイズは約 **372 GB** なので、
+十分な容量のある、できれば高速なディスクに置いてください:
+
+**https://huggingface.co/mastouri/GLM-5.2-colibri-int4-g64-with-int8-mtp**
+
+> ⚠️ 古い行単位 int4 のミラー(`mateogrgic/…`、`jlnsrk/…`)ではなく、上記の **gs64** コンテナを
+> 使ってください。それらは品質が約 9 ポイント劣ることが計測されており、
+> [#455](https://github.com/JustVugg/colibri/issues/455) で報告された当初の思考モードのループや
+> 終わらない生成の根本原因でした。gs64 コンテナは制御された行単位の A/B で見られたそれらの問題を
+> 解消しましたが、繰り返しや EOS 欠乏に対する汎用的なガードではありません。MTP ヘッドも
+> **int4 ではなく int8** である必要があります
+> (int4 → ドラフト受理率 0%、[#8](https://github.com/JustVugg/colibri/issues/8)):
+> `ls -l /out-mtp-*` — int8(正しい)なら 3 ファイルで `3527131672 / 5366238584 / 1065950496`、
+> または単一の `out-mtp-00000.safetensors` で `9959321520` バイトです
+> (推奨コンテナの現在のアップロードは 1 ファイルで配布されています: 中身は同じ
+> int8 テンソルで、1 要素 1 バイトのものが 777 個です)。
+
+あるいは FP8 のソースから自分で変換することもできます — 756 GB 全体を一度にディスクに置く必要のない、
+再開可能なコマンド 1 つで行えます:
+
+```bash
+./coli convert --model /nvme/glm52_i4 # シャードごとにダウンロード+変換(python、初回のみ)
+```
+
+
+#### その他の対応モデル
+
+GLM-5.2 がリファレンスモデルですが、同じストリーミング手法でさらに 6 つのファミリーが動作します。
+それぞれが **兄弟エンジン** です — C ファイル 1 つで独自のアーキテクチャを持ち、同じ
+`coli chat` / `coli serve` / `coli web` フロントエンドを使います(ランチャーはモデルの
+`config.json` からバイナリを選びます):
+
+> **それぞれに必要なもの。** これらは大きく異なり、2 つを並べて読んだ人が要件が矛盾していると
+> 誤解したこともあります([#191](https://github.com/JustVugg/colibri/issues/191))。矛盾してはいません —
+> 別々のモデルなのです。**どれも GPU は必要ありません。**
+>
+> | モデル | 重み用のディスク | RAM | GPU |
+> |---|---|---|---|
+> | **OLMoE** | 約 7 GB(int8 コンテナ) | 8 GB | 不要 |
+> | **GLM-5.2/5.3** | 約 372 GB | 最低 16 GB、快適には 24 GB | 不要 |
+> | **GLM-5.3-Flash** | 変換後 約 195 GB | 25 GB(int4 の重み 12 GB + エキスパートキャッシュ) | 不要 |
+> | **Inkling** | 約 469 GB | int4 密コンテナ使用時 25 GB、未使用時 約 120 GB | 不要 |
+> | **Kimi K3** | 約 1.6 TB | 32 GB 以上 | 不要 |
+> | **DeepSeek V4 Flash** | 約 167 GB(REAP 150B: 約 85 GB) | 最低 16 GB、快適には 32 GB | オプション。GTX 10 シリーズ以降の任意の NVIDIA カード(Pascal/Turing は `CUDA_ARCH=portable-pre-ampere NO_TC=1` で、RTX 50 で最良)により、プリフィルが 5〜10 倍、デコードが約 2.5 倍高速化 |
+> | **Qwen3.8-Flash-Next** | 約 185.5 GB(公式 FP8 チェックポイント) | 最低 16 GB、デフォルトのコンテキストで快適には 24 GB | 非対応。CPU のみ |
+> | **Qwen3.6-35B-A3B** | 約 20 GB(int4-gs64 コンテナ) | 24 GB(RAM への完全常駐が必要) | オプション。CUDA VRAM エキスパートティアは 8 GB カード 2 枚で **1.44 -> 10.05 tok/s(7.0 倍)** を計測、出力は CPU とビット単位で同一 |
+>
+> GPU はあくまで速くするだけです。エキスパートはディスクからストリーミングされるため、速度は
+> ディスクで決まります — 遅いドライブでは 1 秒あたり 1 トークン未満、高速なドライブでキャッシュが
+> 温まっていれば 1 秒あたり数トークンを想定してください。
+
+| ファミリー | 総数 / アクティブ | 重み | ビルド | ドキュメント |
+|---|---|---|---|---|
+| **GLM-5.2/5.3** | 744B / 40B | [`mastouri/…-int4-g64-with-int8-mtp`](https://huggingface.co/mastouri/GLM-5.2-colibri-int4-g64-with-int8-mtp)(372 GB) | `make -C c glm` | このページ |
+| **Inkling**(Thinking Machines) | 975B / 41B | [`nbeerbower/Inkling-colibri-int4`](https://huggingface.co/nbeerbower/Inkling-colibri-int4)(469 GB) | `make -C c inkling` | [inkling.md](docs/inkling.md) |
+| **GLM-5.3-Flash**(Z.ai) | 321B / 40B | [`zai-org/GLM-5.3-Flash`](https://huggingface.co/zai-org/GLM-5.3-Flash) — ルーティングエキスパートを **int4-gs64** に変換、密部分は BF16 のままで精度はロード時に選択。ビジョン対応 | `make -C c glm53` | [glm53-flash.md](docs/glm53-flash.md) |
+| **Kimi K3**(Moonshot) | 2.8T / 104B | [`moonshotai/Kimi-K3`](https://huggingface.co/moonshotai/Kimi-K3) — オリジナルのチェックポイント、ルーティングエキスパートは **ネイティブ MXFP4** のまま | `make -C c kimi_k3` | [kimi_k3.md](docs/kimi_k3.md) |
+| **DeepSeek V4 Flash** | 284B / 13B | 公式のシャード化チェックポイント — ルーティングエキスパートは **ネイティブ fp4**、密部分は fp8-e4m3 のまま。**REAP で枝刈りした 150B**([`puwaer/DeepSeek-V4-Flash-0731-reap-150b`](https://huggingface.co/puwaer/DeepSeek-V4-Flash-0731-reap-150b)、85 GB、256 個中 132 個のエキスパート)も同じエンジンで変換なしにロード可能 | `make -C c deepseek-v4` | [deepseek-v4.md](docs/deepseek-v4.md) |
+| **DeepSeek V4.1 Flash** | 552B / 16B | 公式チェックポイント、**変換不要**: エキスパートはすでに fp4、密部分は fp8-e4m3。そのうち 203 GB は一度に数百バイトずつディスクから読まれる n-gram メモリで、ルーティングエキスパートのコストは GLM-5.2 の 12.7 GB に対して **1 トークンあたり 4.5 GB**。ビジョン、ツール呼び出し、DSpark ドラフトヘッドはすべて有効 | `make -C c deepseek_v41` | [deepseek-v41.md](docs/deepseek-v41.md) |
+| **Qwen3.8-Flash-Next**(Alibaba) | 125B + 51B n-gram / 6B | [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) — オリジナルのチェックポイント。PLE はページング可能なまま、エキスパートは **ネイティブのブロック FP8** のまま | `make -C c qwen38`(CPU のみ) | [qwen38.md](docs/qwen38.md) |
+| **Qwen3.6**(Alibaba) | 35B / 3B | [`Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64`](https://huggingface.co/Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64)(約 20 GB、**推奨**)— Gated Attention + Gated DeltaNet のハイブリッド | `make -C c qwen36`(VRAM エキスパートティアには `CUDA=1`) | [qwen36.md](docs/qwen36.md) |
+| **OLMoE**(AI2) | 7B / 1B | `c/tools/convert_olmoe_merged.py` で変換 — **int8** コンテナ、約 7 GB | `make -C c olmoe` | — |
+
+Qwen3.6 には変換済みコンテナが 3 つあります: **int4-gs64**(推奨 — int8 のアンカーに対するコサイン類似度は
+行単位と比べて 0.98777 → 0.99313、KL は 0.109 → 0.080 と計測されており、量子化誤差が約 44% 少ない)、
+A/B のベースラインとしての [int4 行単位](https://huggingface.co/Kreuzzelg/qwen36-35b-a3b-colibri-i4)、
+そして [KAT-Coder v2.5](https://huggingface.co/Kreuzzelg/kat-coder-v2.5-dev-colibri-i4-gs64) です。
+KAT-Coder は同じエンジンでそのまま動きます — アーキテクチャが同一のチェックポイントであれば、
+専用のコードパスなしで動作します。`CUDA=1` を使うと、VRAM エキスパートティアは
+**8 GB カード 2 枚で 1.44 → 10.05 tok/s(7.0 倍)** を計測し、出力は CPU の経路とビット単位で同一でした。
+
+Kimi K3 は変換不要です。QAT で学習された MXFP4 エキスパートはオリジナルの Hugging Face シャードから
+直接ストリーミングされ、bf16 の密な重みセットはロード時に量子化されます。長いエージェントセッションでは、
+リカレント状態のチェックポイントをオプトインできます(RAM 上に `COLI_K3_CKPT=N` スロット、または
+`COLI_K3_CKPT_DIR` でディスクに退避)。編集されたプロンプトやフォローアッププロンプトは、残っている
+最も深いチェックポイントを復元して末尾だけを再プリフィルするため、会話全体を SSM レイヤーで
+再生し直す必要がありません。Vulkan ホストでは `K3_VK_UP=auto` が、計測された帯域幅から
+エキスパートティアのアップロード量を決めます。エンジンの KDA と MLA の経路は、CI でベンダー実装に
+対してトークン単位で完全一致することが検証されています。
+
+Inkling は int4 のエキスパートと **bf16 の密な重み**(常駐 49.4 GB)で提供されています。それを
+保持できないホスト向けに、[inkling.md](docs/inkling.md) には密な重みセットを 15.3 GB にする
+ワンパスのツールがあり、975B を 25 GB のマシンで動かせます — トレードオフも誠実に書かれています。
+
+### 3. 実行する
+
+```bash
+COLI_MODEL=/nvme/glm52_i4 ./coli chat # RAM 予算、キャッシュ、MTP を自動検出
+COLI_MODEL=/nvme/glm52_i4 ./coli plan # 計画された VRAM/RAM/ディスクの配置を確認
+COLI_MODEL=/nvme/glm52_i4 ./coli doctor # 読み取り専用の準備状況チェック
+COLI_MODEL=/nvme/glm52_i4 ./coli doctor --deep # テンソル/シャード/インデックス/ミラーの厳密な事前チェック
+COLI_MODEL=/nvme/glm52_i4 ./coli tune # このマシンで最速かつ安全な実行プロファイルを計測して保存
+./coli web --model /nvme/glm52_i4 # API + ダッシュボード、ブラウザを開く
+./coli serve --model /nvme/glm52_i4 # API + ダッシュボード、ブラウザなし(ヘッドレス)
+```
+
+Windows ではリリースアーカイブに `coli.cmd` が同梱されています。ダブルクリックでクイックスタート、
+または cmd や PowerShell から `coli.cmd chat --model D:\glm52_i4` を実行してください。
+ソースのチェックアウトからは、同じコマンドを `python coli chat --model
+D:\glm52_i4` として実行できます。`.exe` ファイルはエンジンでありランチャーではありません。
+単体で起動するとロードするモデルがないため、すぐに終了します。
+実行時のエンジンは純粋な C です — Python は初回のみの変換ツールとオプションの
+API ゲートウェイでしか使われません。
+
+#### 同じコマンドでどのモデルも動く
+
+`coli` はモデルの `config.json` を読み、対応するエンジンバイナリを選び、そのファミリーの
+チャットテンプレートを適用します — そのため **モデルが変わってもコマンドラインは何も変わりません**。
+使いたいエンジンを一度ビルドしたら、あとは `COLI_MODEL` を適切なディレクトリに向けるだけです:
+
+```bash
+make -C c glm # GLM-5.2
+make -C c inkling # Inkling
+make -C c kimi_k3 # Kimi K3
+
+COLI_MODEL=/nvme/glm52_i4 ./coli chat # TUI
+COLI_MODEL=/nvme/inkling_i4 ./coli chat
+COLI_MODEL=/nvme/kimi_k3 ./coli chat
+
+./coli web --model /nvme/inkling_i4 # API + ダッシュボード、ブラウザを開く
+./coli web --model /nvme/kimi_k3
+./coli serve --model /nvme/inkling_i4 # API + ダッシュボード、ブラウザなし
+```
+
+GLM 以外のエンジンでは、`coli chat` がローカルでゲートウェイを起動して TUI をそれに接続します。
+そのため TUI、API、ダッシュボードはすべて同じアーキテクチャ対応のチャットテンプレートを通り、
+テンプレートを自分で指定する必要はありません。
+
+モデルごとに異なる点が 2 つあり、どちらも各モデルのページに記載されています:
+
+- **RAM に余裕のないホストでの Inkling** には、int4 の密コンテナと小さなエキスパートキャッシュが
+ 必要です: `./coli chat --model /nvme/inkling_i4 --cap 2`
+ ([inkling.md](docs/inkling.md) を参照 — デフォルトの `--cap 8` では、常駐セットに加えて
+ 約 14 GB のキャッシュが必要です)。
+- **Kimi K3** は MXFP4 エキスパートをオリジナルのチェックポイントからストリーミングするため、
+ 変換するものはありません — ただしスナップショットは約 1.6 TB です
+ ([kimi_k3.md](docs/kimi_k3.md) を参照)。
+
+### 4. さらに詳しく
+
+| トピック | ドキュメント |
+|---|---|
+| ベンチマーク、コミュニティのデータポイント、品質計測 | [docs/benchmarks.md](docs/benchmarks.md) |
+| 再現可能なベンチマークプロトコルと最低限のレポート | [docs/benchmarking.md](docs/benchmarking.md) |
+| チューニング項目、ポリシー、学習キャッシュ、プリフェッチ | [docs/tuning.md](docs/tuning.md) |
+| Windows 11 ネイティブビルド(+ CUDA DLL) | [docs/windows.md](docs/windows.md) |
+| CUDA バックエンド、VRAM エキスパートティア、完全常駐 | [docs/cuda.md](docs/cuda.md) |
+| Vulkan バックエンド(任意の GPU: RADV 経由の AMD、ROCm がサポートを終了したカードを含む) | [docs/vulkan.md](docs/vulkan.md) |
+| Apple Silicon Metal バックエンド | [docs/metal.md](docs/metal.md) |
+| OpenAI 互換 API、KV スロット、Web ダッシュボード | [docs/api.md](docs/api.md) |
+| 実験的なレイヤーセグメント埋め込み ABI | [docs/segment-runtime.md](docs/segment-runtime.md) |
+| 実験的なトークナイザ/埋め込み/ヘッドの Edge ABI | [docs/edge-runtime.md](docs/edge-runtime.md) |
+| 文法強制ドラフト(構造化出力) | [docs/grammar-draft.md](docs/grammar-draft.md) |
+| 環境変数一覧 | [docs/ENVIRONMENT.md](docs/ENVIRONMENT.md) |
+
+## DeepSeek V4
+
+**DeepSeek V4 Flash** は公式チェックポイントを変換なしでストリーミングします。ルーティング
+エキスパートは **ネイティブ fp4** のまま、密な重みセットは UE8M0 ブロックスケール付きの
+**fp8-e4m3** のままです。MLA + DSA スパースアテンション、43 レイヤー、256 個のルーティング
+エキスパートと 1 個の共有エキスパート、top-6。x86-64/aarch64 の Linux と Windows/MSYS2(CPU)に
+対応し、オプションの CUDA ティア(Windows ではランタイム DLL、Linux では `CUDA=1` による直接リンク、
+WSL2 で検証済み)は、すべてのステージを CPU 基準に保ちつつステージごとにフォールバックします。
+
+```bash
+cd c
+make deepseek-v4
+python ./coli chat --model /path/to/DeepSeek-V4-Flash --ram 32
+# coli run / coli serve / coli web も可
+# Windows CUDA ティア: make cuda-dsv4-dll CUDA_ARCH=portable (RTX 50 では + make cuda-dsv4-dg-dll)
+```
+
+新しく追加された 2 つのオプトイン GPU レバーがあり、コミュニティによる計測値を求めています。
+どちらもデフォルトでオフで、未設定時はバイト単位で同一です。`DSV4_HYBRID=1` は、実行時に計測した
+帯域幅を使って、VRAM ティアのミスを GPU のフィル分岐と CPU の分岐に振り分けます。
+`COLI_CUDA_MOE_DOUBLE=1`(`COLI_CUDA_MOE_BATCH=1` と併用)は、現在のレイヤーを計算している間に
+次のレイヤーのエキスパートセット全体を 2 つ目の VRAM バンクにプリフェッチし、VRAM が足りなければ
+単一バンクにフォールバックします。
+CUDA ティアは Pascal と Turing のカード(GTX 10 / RTX 20 シリーズ)でも動作するようになりました:
+`CUDA_ARCH=portable-pre-ampere NO_TC=1` でビルドしてください。
+
+グリーディデコードで KV スロットは 1 つです。ツール呼び出しは、V4 ネイティブのプロンプトと DSML の
+呼び出しブロックで HTTP ゲートウェイを通じて接続されています。文法はサポートされていません。
+[エンジンごとの API マトリクス](docs/api.md#tool-calling-support) を参照してください。プレフィックス
+チェックポイント(メモリ上 + ディスク上)により、システムプロンプトの初回プリフィル後は、
+エージェントセッションやフォローアップのターンが数秒で始まります。RTX 5080 + NVMe 2 台での計測:
+3324 トークンのプリフィルが 90 秒、8.3k トークンの初回ターンが初回のみ約 4 分、以降の
+セッション/ターンは 6〜9 秒、3k コンテキストでデコード約 1.6 tok/s — 詳細は
+[docs/deepseek-v4.md](docs/deepseek-v4.md) を参照してください。
+
+**RAM を与えてください。** 43 × 256 個のルーティングエキスパートはディスク上で約 137 GiB あり、
+1 トークンがそのうち 301 個に触れるため、エキスパートキャッシュのヒット率が tok/s を決めます —
+`--ram` は最も価値のある単一の設定項目であり、変わるのは速度だけで、出力は決して変わりません。
+
+**投機的ドラフトは存在しますが、オフです。** DSpark のマルコフドラフターと完全な MTP はどちらも
+実装・検証済みです。ドラフトはフォワードパスを節約できますが、トークンを変えることは決してありません。
+受理されたトークンはすべてターゲット自身の argmax だからです。実際のマルチターンチャットで計測したところ、
+受理率は 15 個中 1 個と 24 個中 10 個で、このエンジンのリカレントなアテンション状態について
+棄却されたサフィックスを再生するコストがドラフトによる節約を上回りました — 14 トークンの回答 1 つに
+495 秒かかりました。そのため `V4_DRAFT` と `V4_MTP` はデフォルトで `0` とし、より高速なストレージで
+再挑戦する人のために、コードは計測値とともに残してあります。
+
+CUDA ティア(ビルド、DLL の選択、GPU の対応範囲)、環境変数リファレンス、性能値、
+チェックポイントの検証、生成された小さな独立オラクルについては
+[docs/deepseek-v4.md](docs/deepseek-v4.md) を参照してください。
+
+## 今後の展望
+
+- **推論システムの研究こそがプロダクトです。** 現在の階層は LRU + 学習されたピンセットです。
+ 進行中の作業は、モデルフォーマット、圧縮、配置、スケジューリング、I/O、CPU/GPU カーネル、
+ 異種混在のオーバーラップ、KV 状態、ルーティングを考慮した投機的デコードにわたります。
+ 目的はハードウェア要件と有用トークンあたりのコストを下げることです。すべてはこのプロジェクトの
+ やり方で取り込まれます: エンドツーエンドで計測され、レビューされ、オープンに開発されます。
+- **より多くのオープンモデル。** ティアリングアルゴリズムはモデルに依存しません。ルーティング
+ エキスパートを持つ MoE であれば、どれも同じ方法でステージングできます。現在 9 つのファミリーが
+ 動作しています(GLM-5.2、GLM-5.3-Flash、Inkling、Kimi K3、DeepSeek V4 Flash、DeepSeek V4.1 Flash、
+ Qwen3.8-Flash-Next、Qwen3.6、OLMoE)。さらなるオープンウェイトのファミリー — 候補には
+ **MiniMax** も含まれます — は、最初の 8 つと同じ方法でエンジンを獲得します:
+ 誰かがエンドツーエンドで計測したときにです。
+
+## プロジェクトへの支援
+
+colibrì は、RAM 25 GB の 12 コアのラップトップ上で、1 人のプロジェクトとして始まりました。
+今ではその数値は、実機を持つコミュニティから集まっています。役に立ったと感じたら:
+
+- ⭐ リポジトリにスターを付けて共有してください。
+- 🐛 あなたのハードウェアでのベンチマーク数値を添えて issue を作成してください — データポイントは
+ 何よりもこのプロジェクトを前進させます。
+- 💬 [Discord コミュニティ](https://discord.gg/RXV83nSZdk) に参加して、実験、ハードウェアでの結果、
+ 研究の方向性について議論してください。
+- 💬 開発のスポンサーやハードウェアの寄贈については、GitHub の issue からご連絡ください。
+
+## リポジトリ構成
+
+```
+Makefile ルートのビルド/チェックのエントリポイント
+c/
+├── colibri.c GLM-5.2 エンジン (make glm)
+├── inkling.c Inkling エンジン (make inkling)
+├── kimi_k3.c Kimi K3 エンジン (make kimi_k3)
+├── deepseek_v4.c DeepSeek V4 Flash エンジン (make deepseek-v4)
+├── qwen38.c Qwen3.8-Flash-Next テキストエンジン (make qwen38)
+├── qwen36.c Qwen3.6 エンジン (make qwen36)
+├── olmoe.c OLMoE エンジン (make olmoe)
+│
+├── st.h safetensors のインデックスと範囲読み込み
+├── quant.h 正規のコンテナデコーダ
+├── expert_ffn.h MoE エンジン共通のルーテッドエキスパート FFN カーネル(planar int4、レイヤーランナー)
+├── tok.h, json.h トークナイザと JSON パーサ
+├── compat.h Windows/macOS 用シム(POSIX 名を一か所に)
+├── expert_store.h ストリーミングエキスパートキャッシュ
+├── route_trace.h ルーティングのテレメトリと .coli_usage(エンジン非依存)
+├── kv_prefix.h ターンをまたいだ KV プレフィックスの再利用
+│
+├── backend_cuda.* オプションの CUDA ティア (CUDA=1)
+├── backend_metal.* オプションの Metal ティア (METAL=1)
+├── backend_vulkan.* オプションの Vulkan ティア (VULKAN=1)
+│
+├── Makefile ビルドとローカルチェック
+├── coli ユーザー向け CLI
+├── openai_server.py OpenAI 互換 HTTP ゲートウェイ
+├── resource_plan.py `coli plan` と `coli doctor` の背後にある RAM/VRAM プランナー
+├── tools/ オフライン変換、フィクスチャ、ベンチマーク
+├── scripts/ 長時間実行の変換ヘルパー
+└── tests/ 依存関係のない C と Python のテスト
+web/ ブラウザ UI(純粋な OpenAI API クライアント)
+desktop/ Web UI をラップする Tauri v2 デスクトップシェル
+docker/ コンテナイメージ
+docs/ リファレンスドキュメント、実験、メディア
+```
+
+**モデルファミリーごとに `.c` を 1 つ、共有の単一ヘッダの上に。** エンジンは自身のアーキテクチャだけを
+持ち、それ以外は持ちません。2 つのエンジンが共に必要とするもの — safetensors リーダー、コンテナ
+デコーダ、トークナイザ、エキスパートキャッシュ — は両者がインクルードするヘッダに置かれるため、
+修正は一度にすべてのエンジンに届きます。このルールは飾りではありません。ここで繰り返し発生する
+不具合は、ある仕組みが 1 つのエンジンにだけ入り、兄弟エンジンに届かなかったケースなのです。
+
+リポジトリのルートからは、`make`、`make check`、`make clean` がエンジンの Makefile に委譲されます。
+
+## なぜ「colibrì」なのか
+
+ハチドリ(イタリア語で colibrì)は体重わずか数グラムで、空中にとどまり、1 日に千もの花を訪れます。
+このエンジンは、7,440 億パラメータの巨人をハチドリの食事量で生かし続けます: RAM 25 GB、
+CPU 12 コア、そしてディスクへのたっぷりの忍耐です。
+
+## 謝辞
+
+colibrì はエンジンにすぎず、それが動かす知性は贈り物です。フロンティア級の重みをオープンに
+公開しているチーム — **Z.ai**(GLM)、**Moonshot AI**(Kimi)、**Alibaba Qwen**、**MiniMax**、
+**Allen AI**(OLMoE)— そして、ベンチマークを取り、バイセクトし、アトラスの実行を再現し、
+パッチを送ってくれたすべてのコントリビューターに感謝します。
+このプロジェクトは、オープンウェイトが何を可能にするかの証明です。
+
+このプロジェクトのエキスパート配置、圧縮、ルーティングの実験は、以下のオープンな研究と
+システムのアイデアやエビデンスにも基づいています:
+
+- 出力を考慮した、ドメイン固有のエキスパート重要度については
+ [REAP](https://github.com/CerebrasResearch/reap) と
+ [EASY-EP](https://github.com/RUCAIBox/EASYEP)。
+- 類似度に基づくエキスパートの再ルーティングについては [SERE](https://github.com/JL-Cheng/SERE)、
+ キャッシュ局所性を考慮したルーターのファインチューニングについては
+ [ReMoE](https://github.com/BUAA-OSCAR/ReMoE)。
+- ルーティングに導かれたエキスパートのマージと圧縮については
+ [MC-SMoE](https://github.com/UNITES-Lab/MC-SMoE)。
+- 共有エキスパート基底と低ランクのエキスパート差分については
+ [MoBE](https://github.com/inclusionAI/MoBE) と
+ [D²-MoE](https://github.com/lliai/D2MoE)。
+- CPU/GPU ハイブリッドのエキスパートスケジューリングについては
+ [HybriMoE](https://github.com/PKU-SEC-Lab/HybriMoE)、エキスパート通信と計算のオーバーラップについては
+ [ScMoE](https://arxiv.org/abs/2404.05019)、分散オンデマンドのエキスパートロードについては
+ [OD-MoE](https://arxiv.org/abs/2512.03927)。
+- 比較を再現可能にしているオープンな推論システムとエキスパートオフロードの取り組みについては
+ [vLLM](https://github.com/vllm-project/vllm)、
+ [llama.cpp](https://github.com/ggml-org/llama.cpp)、
+ [kTransformers](https://github.com/kvcache-ai/ktransformers)。
+
+エンジンはアイデアだけでなく、具体的なエンジニアリングの成果の上にも成り立っています。以下はいずれも
+現在ツリー内で使われているか、再実装されています:
+
+- [safetensors](https://github.com/huggingface/safetensors) — すべてのエンジンが読むコンテナ
+ (`c/st.h`)。fp8 と I64 の dtype を含みます。
+- [tiktoken](https://github.com/openai/tiktoken) — `c/tok.h` はその `byte_pair_encode` を
+ 正確に再実装しており、連結した結果の語彙 ID が最も小さい隣接ペアをマージするため、
+ tiktoken 由来の語彙にはマージリストが不要です。
+- [llama.cpp](https://github.com/ggml-org/llama.cpp) — `c/grammar.h` の GBNF 文法サブセットは
+ その構文とスタック集合による PDA に従っており、Metal の経路はその
+ `newBufferWithBytesNoCopy` による常駐テクニックを借用しています。
+- [vLLM](https://github.com/vllm-project/vllm) — エンジンが位置ごとに一致させている出力
+ セマンティクスのリファレンス(例: 最終ノルムが LM ヘッドに対してどこに入るか)。
+- [transformers](https://github.com/huggingface/transformers) — オラクル:
+ CI はランダム初期化モデルをこれに対してトークン単位で再現します。
+- [DietGPU](https://github.com/facebookresearch/dietgpu) — 実験的な圧縮エキスパートティア
+ (`COLI_ANS`)の背後にある GPU ANS コーデック。
+- [rocWMMA](https://github.com/ROCm/rocWMMA) — HIP バックエンドは CUDA の
+ `nvcuda::wmma` の fragment/mma_sync API をこれにマッピングしており(`c/backend_gpu_compat.h`)、
+ それによって 1 つの .cu ソースを両ベンダー向けにコンパイルできます。
+
+## ライセンス
+
+Apache 2.0。GLM-5.2 の重みは Z.ai により MIT ライセンスで公開されています。
diff --git a/README.md b/README.md
index f23129938..3a0ad7d9e 100644
--- a/README.md
+++ b/README.md
@@ -10,7 +10,7 @@
Website ·
Discord ·
- English · 简体中文 · 繁體中文 · Italiano
+ English · 简体中文 · 繁體中文 · Italiano · 日本語
**Tiny engine, immense model.** Run **frontier MoE models — 744B to 2.8T
@@ -40,7 +40,7 @@ may reduce speed; it must not quietly redefine the model.
```
$ ./coli chat
- 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
+ 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU
✓ ready in 32s · resident 9.9 GB
› ciao!
◆ Ciao! 😊 Come posso aiutarti oggi?
@@ -135,7 +135,7 @@ shows otherwise. These are the main questions now:
| hypothesis | evidence so far | experiment still needed |
|---|---|---|
| Routing history can place experts better than plain LRU | learned pins improve repeated workloads, but can overfit a prompt | held-out, cross-session A/Bs across coding, chat, multilingual, and long-context workloads |
-| Multiple SSDs can turn independent bandwidth into decode speed | weighted mirror/split routing is implemented and validated; the bandwidth model is sound | cold-cache one-drive vs two-drive GLM-5.2 runs on real, independent controllers |
+| Multiple SSDs can turn independent bandwidth into decode speed | two independent NVMe drives measured +37.5% decode; a slower third drive was neutral after weighted striping ([measurements](docs/multidisk.md#what-has-been-measured)) | reproduce across drive speeds, controller layouts, and cache states |
| A hardware-aware planner can approach each machine's best configuration automatically | RAM/VRAM budgets and several backends are detected today | compare the generated plan with a controlled parameter sweep across laptops, workstations, NUMA hosts, and multi-GPU systems |
| Lossless or quality-bounded representations can reduce weight movement enough to matter | format and quantization ablations exist, with correctness/quality gates | reproduce quality, bytes moved, latency, and cost per useful token together — not compression ratio alone |
| Routing-aware speculation can pay before near-full residency | MTP and grammar drafts work, but MTP has also measured a 32% loss around 85% expert hit | map the break-even surface across acceptance, expert hit rate, batch union, and draft depth |
@@ -232,21 +232,30 @@ precision are the same whether an expert answered from VRAM or from disk.
-### Dual-SSD: two copies of the model, twice the read bandwidth
+
-Decode is disk-bound on most machines, and expert reads are read-only — so if you have a **second SSD**, put a full copy of the model on it and let the engine stream from both drives at once:
+### Multiple SSDs: stream model copies from more than one drive
+
+When decode is disk-bound, a **second SSD** can help: put a copy of the model on
+it and let the engine read from both drives. For GLM-5.2, from `c/` in a source
+checkout (or from an unpacked release):
```bash
-COLI_MODEL=/fast/glm52_i4 COLI_MODEL_MIRROR=/second/glm52_i4 ./coli chat
-COLI_DISK_WEIGHTS=9,3 ... # optional: primary,mirror bandwidth ratio (else measured at startup)
+COLI_MODEL_MIRROR=/second/glm52_i4 python3 ./coli chat --model /fast/glm52_i4
```
-Each expert is routed to one drive by a deterministic hash, weighted by the two drives' measured (or declared) bandwidth, so readahead/PILOT prefetch and the demand read always hit the same drive and nothing is cached twice. The aggregate bandwidth is the sum of both drives — a 9 GB/s + 3 GB/s pair reads experts ~33% faster than the fast drive alone, and the OMP-parallel pin/warmup load streams from both. Details worth knowing:
+The engine measures the drives at startup to weight the read split. Buffered
+reads use deterministic expert routing; eligible direct reads can stripe one
+expert across replicas. Independent drives provide bandwidth headroom, not a
+guaranteed token-rate multiplier: shared controllers, cache hits, and compute
+can limit the gain. See the [multi-disk guide](docs/multidisk.md) for Bash and
+PowerShell examples, measured gains and limits, and a single-drive comparison.
+Details worth knowing:
-- the mirror is **validated at startup** (per-file size + safetensors header must be byte-identical to the primary); divergent or missing files silently stay on the primary, so a **partial mirror is fine** — a smaller second SSD holding only some shards still helps;
+- the mirror is **validated at startup** (per-file size + safetensors header must be byte-identical to the primary); divergent or missing files stay on the primary, so a **partial mirror is fine** — a smaller second SSD can serve the shards it holds;
- the mirror is **never written**: `.coli_usage`, `.coli_kv` and all sidecars stay on the primary;
- a read error on the mirror falls back to the primary (one warning, no crash), so unplugging the second drive mid-run degrades instead of killing the server;
-- routing never changes tokens — both copies are byte-identical, and the per-run `MIRROR:` stats line shows GB served per drive.
+- routing never changes tokens — both copies are byte-identical; enable `PROF=1` for the `MIRROR:` profile counters showing GB served per drive.
The same engine spans the whole range: on a 25 GB laptop everything streams from
disk (slow but correct); on a large host the entire expert set becomes resident
@@ -325,6 +334,15 @@ full forensic story). Grammar-forced drafts
constrained JSON output. Whether speculation is a net win depends on your
cache temperature — measure, and use `DRAFT=0` when it doesn't pay.
+Verify batches can also opt into an **exact attention core** with
+`COLI_EXACT_VERIFY=1` ([#689](https://github.com/JustVugg/colibri/issues/689)):
+the CPU MLA-absorb score and context dots accumulate integer products and round
+once, so a near-tie in a verify row resolves the same way on every host, at
+roughly 0.6x tok/s on a tiny oracle (the dot itself is ~5–7x the float loop).
+Two limits to know: with a quantised KV cache (`tq1`, TQ or int8 KV) the
+context dot keeps the float path, so exactness there is not provided; and a
+real near-tie flip has only been argued, not yet caught on GLM-5.2 at n=64.
+
## What it achieves
@@ -434,7 +452,7 @@ the model's `config.json`):
> | **Inkling** | ~469 GB | 25 GB with the int4 dense container, ~120 GB without | not needed |
> | **Kimi K3** | ~1.6 TB | 32 GB+ | not needed |
> | **DeepSeek V4 Flash** | ~167 GB (REAP 150B: ~85 GB) | 16 GB min, 32 GB comfortable | optional; any NVIDIA card from the GTX 10 series up (Pascal/Turing via `CUDA_ARCH=portable-pre-ampere NO_TC=1`, best on RTX 50) makes prefill 5-10x and decode ~2.5x faster |
-> | **Qwen3.8-Flash-Next** | ~185.5 GB (official FP8 checkpoint) | 16 GB min, 24 GB comfortable at the default context | not supported; CPU only |
+> | **Qwen3.8-Flash-Next** | ~185.5 GB (official FP8 checkpoint) | 16 GB min, 24 GB comfortable at the default context | optional; the CUDA VRAM expert tier with dense trunk quantized to int8 in VRAM |
> | **Qwen3.6-35B-A3B** | ~20 GB (int4-gs64 container) | 24 GB (needs full RAM residency) | optional; the CUDA VRAM expert tier measured **1.44 -> 10.05 tok/s (7.0x)** on two 8 GB cards, output bit-identical to CPU |
>
> A GPU only ever makes it faster. Speed is set by your disk, because the experts
@@ -449,7 +467,7 @@ the model's `config.json`):
| **Kimi K3** (Moonshot) | 2.8T / 104B | [`moonshotai/Kimi-K3`](https://huggingface.co/moonshotai/Kimi-K3) — original checkpoint, routed experts stay **native MXFP4** | `make -C c kimi_k3` | [kimi_k3.md](docs/kimi_k3.md) |
| **DeepSeek V4 Flash** | 284B / 13B | official sharded checkpoint — routed experts stay **native fp4**, dense stays fp8-e4m3; the **REAP-pruned 150B** ([`puwaer/DeepSeek-V4-Flash-0731-reap-150b`](https://huggingface.co/puwaer/DeepSeek-V4-Flash-0731-reap-150b), 85 GB, 132 of 256 experts) loads with the same engine and no conversion | `make -C c deepseek-v4` | [deepseek-v4.md](docs/deepseek-v4.md) |
| **DeepSeek V4.1 Flash** | 552B / 16B | official checkpoint, **no conversion**: experts are already fp4, dense is fp8-e4m3. 203 GB of it is an n-gram memory read from disk a few hundred bytes at a time, and the routed experts cost **4.5 GB per token** against GLM-5.2's 12.7. Vision, tool calling and the DSpark draft head are all on | `make -C c deepseek_v41` | [deepseek-v41.md](docs/deepseek-v41.md) |
-| **Qwen3.8-Flash-Next** (Alibaba) | 125B + 51B n-gram / 6B | [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) — original checkpoint; PLE stays pageable and experts stay **native block-FP8** | `make -C c qwen38` (CPU only) | [qwen38.md](docs/qwen38.md) |
+| **Qwen3.8-Flash-Next** (Alibaba) | 125B + 51B n-gram / 6B | [`Qwen/Qwen3.8-Flash-Next-FP8`](https://huggingface.co/Qwen/Qwen3.8-Flash-Next-FP8) — original checkpoint; PLE stays pageable and experts stay **native block-FP8** | `make -C c qwen38` (`CUDA=1` for the VRAM expert tier) | [qwen38.md](docs/qwen38.md) |
| **Qwen3.6** (Alibaba) | 35B / 3B | [`Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64`](https://huggingface.co/Kreuzzelg/qwen36-35b-a3b-colibri-i4-gs64) (~20 GB, **recommended**) — hybrid Gated Attention + Gated DeltaNet | `make -C c qwen36` (`CUDA=1` for the VRAM expert tier) | [qwen36.md](docs/qwen36.md) |
| **OLMoE** (AI2) | 7B / 1B | converted with `c/tools/convert_olmoe_merged.py` — **int8** container, ~7 GB | `make -C c olmoe` | — |
diff --git a/README.zh-CN.md b/README.zh-CN.md
index c15a7ea3d..7f9a9420d 100644
--- a/README.zh-CN.md
+++ b/README.zh-CN.md
@@ -4,7 +4,7 @@
Discord ·
- English · 简体中文 · 繁體中文 · Italiano
+ English · 简体中文 · 繁體中文 · Italiano · 日本語
**小巧引擎,庞大模型。**在消费级与异构硬件上运行**前沿 MoE 模型——从 744B 到
@@ -25,7 +25,7 @@ Colibrì 刻意用于验证激进的系统思路——因此**对速度不作 SL
```
$ ./coli chat
- 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
+ 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU
✓ ready in 32s · resident 9.9 GB
› ciao!
◆ Ciao! 😊 Come posso aiutarti oggi?
diff --git a/README.zh-TW.md b/README.zh-TW.md
index adfeb43ae..c208ac413 100644
--- a/README.zh-TW.md
+++ b/README.zh-TW.md
@@ -4,7 +4,7 @@
Discord ·
- English · 简体中文 · 繁體中文 · Italiano
+ English · 简体中文 · 繁體中文 · Italiano · 日本語
**小巧引擎,龐大模型。**在消費級與異質硬體上執行**前沿 MoE 模型——從 744B 到
@@ -25,7 +25,7 @@ Colibrì 刻意用於驗證激進的系統構想——因此**對速度不作 SL
```
$ ./coli chat
- 🐦 colibri v1.12.0 — GLM-5.2 · 744B MoE · int4 · streaming CPU
+ 🐦 colibri v1.12.1 — GLM-5.2 · 744B MoE · int4 · streaming CPU
✓ ready in 32s · resident 9.9 GB
› ciao!
◆ Ciao! 😊 Come posso aiutarti oggi?
diff --git a/c/.gitignore b/c/.gitignore
index 36d9678ca..ef2a31d61 100644
--- a/c/.gitignore
+++ b/c/.gitignore
@@ -21,6 +21,7 @@ tests/fuzz_rans
tests/bench_omp_grain
tests/mxfp4_ref.o
mxfp4_cuda_test
+mxfp4_expert_cuda_test
absorb_determinism_test
cuda_fmt_trap_test
tests/*.dSYM/
diff --git a/c/Makefile b/c/Makefile
index 2fe13efd5..58a1d00a5 100644
--- a/c/Makefile
+++ b/c/Makefile
@@ -491,16 +491,36 @@ endif
TEST_RULES := $(shell sed -n 's|^tests/\(test_[a-z0-9_]*\)\$$(EXE):.*|\1|p' $(firstword $(MAKEFILE_LIST)))
# test_uring is Linux-only. V4 engine tests are appended below only on supported
# x86-64 Linux/Windows and aarch64 Linux hosts; the V4 infrastructure tests have
-# unconditional rules and therefore remain portable gates.
+# unconditional rules and therefore remain portable gates. The six forced
+# SSE4.1 tests use -msse4.1/-mno-avx2/-mno-fma, which only exist as gcc/clang
+# flags on x86 -- unconditionally in TEST_BINS they take down `check` on
+# arm64 (e.g. macOS): "unsupported option '-msse4.1' for target arm64-...".
+# Appended back below only on x86-64 hosts, alongside the other conditional
+# platform tests already handled the same way.
+# test_e8x4g64_loader takes a minted container directory on argv and prints its
+# usage (exit 2) without one, so it is a harness rather than a gate: it has a
+# rule here so it compiles under the suite's own $(CFLAGS) and its warnings are
+# caught -- reachable via `make e8x4g64-loader-check`, kept out of `check` itself
+# since a bare invocation exits 2 -- and tests/test_e8x4g64_mint_load.py is what
+# actually drives it end to end.
# test_qwen38_tier_engine drives qwen38.c over the generated FP8 fixture
# (qwen38_tiny_fp8, gitignored); it runs from qwen38-tier-engine-check.
TEST_EXCLUDE = test_uring test_deepseek_v4 test_v4_ownership test_v4_serve_framing \
test_segment_adapters_registration test_segment_adapters_real \
test_edge_adapters_registration test_edge_adapters_real \
- test_qwen38_tier_engine
+ test_gsgemv_sse41 test_qgemv_sse41 test_olmoe_dot_i8_16_sse41 test_st_f16_bf16_simd_sse41 test_qwen38_tier_engine \
+ test_e8x4g64_loader \
+ test_i4_grouped_sse41 test_i4_grouped_sse41_o1 test_i4_grouped_sse41_no_contract
TEST_BINS = $(addprefix tests/,$(addsuffix $(EXE),$(filter-out $(TEST_EXCLUDE),$(TEST_RULES))))
ifneq (,$(LINUX))
TEST_BINS += tests/test_uring$(EXE)
+TEST_BINS += tests/test_exact_dot$(EXE)
+endif
+ifneq (,$(X86_64))
+TEST_BINS += tests/test_gsgemv_sse41$(EXE) tests/test_qgemv_sse41$(EXE) \
+ tests/test_olmoe_dot_i8_16_sse41$(EXE) tests/test_st_f16_bf16_simd_sse41$(EXE) \
+ tests/test_i4_grouped_sse41$(EXE) \
+ tests/test_i4_grouped_sse41_o1$(EXE) tests/test_i4_grouped_sse41_no_contract$(EXE)
endif
ifeq ($(COLI_V4_SUPPORTED),1)
TEST_BINS += tests/test_v4_hybrid_policy$(EXE) tests/test_k3_fill_budget$(EXE) tests/test_v4_bank_pair$(EXE)
@@ -700,6 +720,9 @@ deepseek-v4-tiny-check: deepseek-v4-tiny-generate
$(PYTHON) tests/test_deepseek_v4_prefix.py \
--binary $(CURDIR)/$(if $(IS_WIN),deepseek_v4.exe,deepseek_v4) \
--fixture $(CURDIR)/deepseek_v4_tiny
+ $(PYTHON) tests/test_deepseek_v4_brio.py \
+ --binary $(CURDIR)/$(if $(IS_WIN),deepseek_v4.exe,deepseek_v4) \
+ --fixture $(CURDIR)/deepseek_v4_tiny
deepseek-v4-oracle: deepseek-v4
@test -n "$(MODEL)" || { echo "usage: make deepseek-v4-oracle MODEL=/path/to/checkpoint" >&2; exit 2; }
@@ -749,11 +772,20 @@ ifneq "$(BUILD_CONFIG)" "$(BUILD_CONFIG_OLD)"
# cmd.exe, so `make glm.exe` from the VS Native Tools prompt or with scoop MinGW
# (neither ships sh.exe) failed with "'printf' is not recognized" and left the
# stamp stale. $(file ...) works regardless of the shell make falls back to. (#478)
+#
+# GNU Make 3.81 -- /usr/bin/make on macOS -- has no $(file ...): it expands to
+# nothing, the stamp is never written, and every build relinks (#1732). Make
+# 3.x is found there and not under cmd.exe (MSYS2 and MinGW ship 4.x), and
+# macOS always has a POSIX printf, so that one case keeps the shell write.
+ifneq ($(filter 3.%,$(MAKE_VERSION)),)
+$(shell printf '%s\n' '$(subst ','\'',$(BUILD_CONFIG))' > .build-config)
+else
$(file >.build-config,$(BUILD_CONFIG))
endif
+endif
.build-config: ;
-colibri$(EXE): colibri.c pin_pool.h cli_args.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h sample.h kv_persist.h telemetry.h route_trace.h omp_tune.h kv_fp8.h kv_tq.h abl.h backend_cuda.h backend_metal.h backend_vulkan.h decode_batch.h edge_adapters.h edge_runtime.h edge_tok_internal.h schema_gbnf.h segment_adapter_internal.h segment_adapters.h segment_runtime.h tier.h $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) .build-config
+colibri$(EXE): colibri.c sse41_kernels.h exact_dot.h oracle.h pin_pool.h cli_args.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h omp_tune.h kv_fp8.h kv_tq.h abl.h backend_cuda.h backend_metal.h backend_vulkan.h decode_batch.h edge_adapters.h edge_runtime.h edge_tok_internal.h schema_gbnf.h segment_adapter_internal.h segment_adapters.h segment_runtime.h tier.h $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) $(VK_SPV) .build-config
$(CC) $(CFLAGS) colibri.c $(CUDA_OBJ) $(METAL_OBJ) $(VK_OBJ) -o colibri$(EXE) $(LDFLAGS)
# Vulkan backend object (plain C + vulkan headers) and its SPIR-V shaders.
@@ -772,7 +804,7 @@ backend_loader.o: backend_loader.c backend_cuda.h compat.h .build-config
# shell that has the MSVC environment set (e.g. after vcvars64.bat, or from a
# "x64 Native Tools Command Prompt"). COLI_CUDA_BUILDING_DLL enables
# __declspec(dllexport) so the 15 API symbols are exported.
-cuda-dll: backend_cuda.cu backend_cuda.h
+cuda-dll: backend_cuda.cu backend_cuda.h fp8_format.h
@command -v "$(NVCC)" >/dev/null 2>&1 || { echo "nvcc not found: set CUDA_HOME or NVCC" >&2; exit 1; }
@command -v cl >/dev/null 2>&1 || { echo "cl.exe (MSVC) not in PATH — run vcvars64.bat first" >&2; exit 1; }
# The banner is localized ("for x64" / "per x64" / "pour x64"): match the arch token only (#1531).
@@ -802,7 +834,7 @@ cuda-dll: backend_cuda.cu backend_cuda.h
# survives. No path is hardcoded — see the HIP SDK contract above.
#
# make hip-dll HIP_DLL=1 HIP_SDK_ROOT= HIP_ARCH=gfx1151
-hip-dll: backend_cuda.cu backend_cuda.h backend_gpu_compat.h
+hip-dll: backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h
@test -x "$(HIPCC)" || command -v "$(HIPCC)" >/dev/null 2>&1 || { echo "hipcc not found at \"$(HIPCC)\": set HIP_SDK_ROOT=, HIP_BIN_DIR= or HIPCC=" >&2; exit 1; }
@test -d "$(HIP_INCLUDE_DIR)" || { echo "HIP include dir not found: \"$(HIP_INCLUDE_DIR)\" — set HIP_INCLUDE_DIR=" >&2; exit 1; }
@test -f "$(HIP_INCLUDE_DIR)/hip/hip_runtime.h" || { echo "hip/hip_runtime.h missing under \"$(HIP_INCLUDE_DIR)\" — set HIP_INCLUDE_DIR=" >&2; exit 1; }
@@ -821,7 +853,7 @@ backend_cuda_ink.o: backend_cuda_ink.cu backend_cuda_ink.h .build-config
@command -v "$(NVCC)" >/dev/null 2>&1 || { echo "nvcc not found: set CUDA_HOME or NVCC" >&2; exit 1; }
"$(NVCC)" $(NVCCFLAGS) -c backend_cuda_ink.cu -o $@
-backend_cuda.o: backend_cuda.cu backend_cuda.h backend_gpu_compat.h .build-config
+backend_cuda.o: backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h .build-config
@command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; }
"$(GPUCC)" $(GPUFLAGS) -c backend_cuda.cu -o $@
@@ -944,10 +976,12 @@ rans: $(RANSLIB)
$(RANSLIB): tools/rans_ctypes.c rans.h
$(CC) $(CFLAGS) -fPIC -shared $< -o $@ $(LDFLAGS)
-cuda-test: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu
+cuda-test: tests/test_mxfp4_expert_cuda.cu backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/test_backend_cuda.cu tests/test_ragged_attention.cu tests/test_absorb_determinism.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.c tests/test_fp8_warp_cuda.cu tests/test_fp8_cuda.cu tests/test_weights_owned_cuda.cu tests/test_cuda_fmt_trap_cuda.cu tests/test_alloc_footprint_cuda.cu tests/test_cuda_init_failure.cu
@command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; }
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_backend_cuda.cu -o backend_cuda_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
./backend_cuda_test$(EXE)
+ "$(GPUCC)" $(GPUFLAGS) tests/test_cuda_init_failure.cu -o cuda_init_failure_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
+ ./cuda_init_failure_test$(EXE)
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_ragged_attention.cu -o ragged_attention_test$(EXE) $(GPU_TEST_LIBS)
./ragged_attention_test$(EXE)
# weight_at's device-side refusal (the __trap()) on real silicon. The
@@ -1003,9 +1037,16 @@ cuda-test: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/test_backen
# Getting that wrong would leave OpenMP symbols in the object while the link
# below no longer pulls the runtime, and CI could not catch it because CI
# never runs cuda-test. -Xclang appears nowhere else in CFLAGS.
- $(CC) $(filter-out -Xclang -fopenmp,$(CFLAGS)) -fno-openmp -c tests/mxfp4_ref.c -o tests/mxfp4_ref.o
- "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_mxfp4_cuda.cu tests/mxfp4_ref.o -o mxfp4_cuda_test$(EXE) $(GPU_TEST_LIBS)
+ $(CC) $(filter-out -Xclang -fopenmp,$(CFLAGS)) -fno-openmp -fPIC -c tests/mxfp4_ref.c -o tests/mxfp4_ref.o
+ # -x none only under HIP: hip-test is `$(MAKE) cuda-test HIP=1`, so this
+ # line also runs under nvcc for CUDA users, and nvcc accepts only c, c++
+ # and cu for -x -- it fails the invocation on anything else. CI never
+ # executes cuda-test, so that breakage would surface only on the next
+ # CUDA box to run the suite.
+ "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_mxfp4_cuda.cu $(if $(filter 1,$(HIP)),-x none) tests/mxfp4_ref.o -o mxfp4_cuda_test$(EXE) $(GPU_TEST_LIBS)
./mxfp4_cuda_test$(EXE)
+ "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/test_mxfp4_expert_cuda.cu $(if $(filter 1,$(HIP)),-x none) tests/mxfp4_ref.o -o mxfp4_expert_cuda_test$(EXE) $(GPU_TEST_LIBS)
+ ./mxfp4_expert_cuda_test$(EXE)
# Allocator footprint (#687): what a cudaMalloc really takes off the card,
# which is what the expert tier has to be charged rather than the logical
# byte count. Runs LAST on purpose - it is the only test here that
@@ -1046,7 +1087,7 @@ gpu-compile: backend_cuda.o
"$(GPUCC)" $(GPUFLAGS) tests/test_weights_owned_cuda.cu -o weights_owned_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
"$(GPUCC)" $(GPUFLAGS) tests/test_cuda_fmt_trap_cuda.cu -o cuda_fmt_trap_test$(EXE) $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
-cuda-bench: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/bench_tensor_core.cu
+cuda-bench: backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/bench_tensor_core.cu
@command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; }
"$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/bench_tensor_core.cu -o backend_cuda_bench$(EXE) $(ANS_NVCC_LIBS)
./backend_cuda_bench$(EXE)
@@ -1054,19 +1095,23 @@ cuda-bench: backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/bench_tens
# fmt=8 kernel bench: old vs COLI_CUDA_F8_WARP kernels per decode path, census
# expert shapes, JSON on stdout (kernel time only). The build is a separate
# file target so tools/run_f8_bench.sh can keep stdout pure JSON.
-fp8_bench$(EXE): backend_cuda.cu backend_cuda.h backend_gpu_compat.h tests/bench_fp8_cuda.cu
+fp8_bench$(EXE): backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h tests/bench_fp8_cuda.cu
@command -v "$(GPUCC)" >/dev/null 2>&1 || { echo "$(GPUCC_NAME) not found" >&2; exit 1; }
"$(GPUCC)" $(GPUFLAGS) tests/bench_fp8_cuda.cu -o fp8_bench$(EXE) $(ANS_NVCC_LIBS)
fp8-bench: fp8_bench$(EXE)
./fp8_bench$(EXE)
+# Matched resident int8 GPU projection timings; run explicitly, not in CI.
+tests/bench_cuda_resident_batch$(EXE): tests/bench_cuda_resident_batch.cu backend_cuda.cu backend_cuda.h fp8_format.h backend_gpu_compat.h
+ "$(GPUCC)" $(GPUFLAGS) backend_cuda.cu tests/bench_cuda_resident_batch.cu -o $@ $(ANS_NVCC_LIBS) $(GPU_TEST_LIBS)
+
# NOCUDA_*: olmoe.c has no COLI_CUDA code at all, so CUDA=1 would otherwise
# hand it a define matching nothing and link a runtime it never calls -- a
# build whose compile line, libraries and exit status all claim "CUDA" while
# the GPU sits idle. Same guard #783 put on kimi_k3, which since gaining an
# MXFP4 expert path no longer needs it. tests/test_makefile_cuda_scope.py
# asserts this shape.
-olmoe$(EXE): olmoe.c cli_args.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h kv_prefix.h pin_pool.h route_trace.h serve_codec.h edge_adapters.h edge_runtime.h edge_tok_internal.h fused_simd.h segment_adapter_internal.h segment_adapters.h segment_runtime.h
+olmoe$(EXE): olmoe.c cli_args.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h kv_prefix.h pin_pool.h route_trace.h serve_codec.h serve_budget.h edge_adapters.h edge_runtime.h edge_tok_internal.h fused_simd.h segment_adapter_internal.h segment_adapters.h segment_runtime.h sse41_kernels.h
$(CC) $(NOCUDA_CFLAGS) olmoe.c -o olmoe$(EXE) $(NOCUDA_LDFLAGS)
# Qwen3.6-35B-A3B engine (hybrid Gated Attention + Gated DeltaNet + streaming
@@ -1092,13 +1137,21 @@ QWEN36_TIER_SRC =
QWEN36_CFLAGS = $(NOCUDA_CFLAGS)
QWEN36_LDFLAGS = $(NOCUDA_LDFLAGS)
endif
-qwen36$(EXE): qwen36.c decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+# On Windows, EXE=.exe. Keep a bare qwen36 target so GNU make does not
+# fall through to its implicit %: %.c rule and compile qwen36.c alone.
+ifneq ($(EXE),)
+.PHONY: qwen36
+qwen36: qwen36$(EXE)
+endif
+# Rebuild when CUDA_DLL changes; otherwise an existing CPU-only executable can
+# be reported as up to date despite selecting the Windows CUDA DLL tier.
+qwen36$(EXE): qwen36.c sse41_kernels.h decode_batch.h serve_poll.h cli_args.h qwen36_tier.h expert_ffn.h idot.h st.h json.h compat.h omp_tune.h kv_prefix.h pin_pool.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h gsgemv.h qgemv.h $(QWEN36_TIER_SRC) $(CUDA_OBJ) .build-config
$(CC) $(QWEN36_CFLAGS) qwen36.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen36$(EXE) $(QWEN36_LDFLAGS)
# DeepSeek V4.1 Flash: one file, like every other portable engine. The fp4 experts
# stream from the official checkpoint and quant.h's mxfp4 kernel reads them as they
# are, so there is no conversion target here to go with it.
-deepseek_v41$(EXE): deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h \
+deepseek_v41$(EXE): deepseek_v41.c sse41_kernels.h cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h fp8_format.h idot.h \
sparse_attn.h hyper_connections.h serve_codec.h serve_poll.h .build-config
$(CC) $(CFLAGS) deepseek_v41.c -o deepseek_v41$(EXE) $(LDFLAGS)
@@ -1106,7 +1159,7 @@ deepseek_v41$(EXE): deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h co
# loader and SERVE=1 protocol. With CUDA=1 it links the same expert tier as
# qwen36 (fp8 streaming mode: hot experts get VRAM copies, the RAM LRU stays);
# without it the tier header's inline stubs keep the build toolkit-free.
-qwen38$(EXE): qwen38.c pin_pool.h cli_args.h qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h omp_tune.h quant.h route_trace.h tok.h tok_unicode.h tok_unicode_o200k.h serve_codec.h edge_adapter_internal.h edge_adapters.h edge_runtime.h qwen38_vision.h segment_adapter_internal.h segment_adapters.h segment_runtime.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+qwen38$(EXE): qwen38.c sse41_kernels.h pin_pool.h cli_args.h qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h omp_tune.h quant.h fp8_format.h idot.h route_trace.h tok.h tok_unicode.h tok_unicode_o200k.h serve_codec.h edge_adapter_internal.h edge_adapters.h edge_runtime.h qwen38_vision.h segment_adapter_internal.h segment_adapters.h segment_runtime.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) qwen38.c $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o qwen38$(EXE) $(QWEN36_LDFLAGS)
.PHONY: qwen38-tiny-generate qwen38-tiny-check
@@ -1127,9 +1180,13 @@ qwen38-ple-prefetch-check: qwen38-tiny-generate qwen38$(EXE)
done; done; done; \
[ $$fail -eq 0 ] && echo "PLE prefetch: identical tokens on and off, 16 configurations" || exit 1
+# Q38_TRUNK_MIN_KB=0 puts every dense matrix of the fixture (all under the
+# default 1 MiB threshold) on the int8 trunk with integer dot products, the
+# path the released checkpoint takes; 1024 leaves them BF16.
qwen38-tiny-check: qwen38-tiny-generate qwen38$(EXE)
- @for batch in 0 1; do for bf16 in 0 1; do for cap in 1 4; do \
- Q38_PREFILL_BATCH=$$batch Q38_NATIVE_BF16=$$bf16 OMP_NUM_THREADS=2 SNAP=./qwen38_tiny ./qwen38$(EXE) $$cap 8 ./qwen38_tiny/ref.json || exit $$?; \
+ @for trunk in 1024 0; do for batch in 0 1; do for bf16 in 0 1; do for cap in 1 4; do \
+ Q38_TRUNK_MIN_KB=$$trunk Q38_PREFILL_BATCH=$$batch Q38_NATIVE_BF16=$$bf16 OMP_NUM_THREADS=2 SNAP=./qwen38_tiny ./qwen38$(EXE) $$cap 8 ./qwen38_tiny/ref.json || exit $$?; \
+ done; \
done; \
done; \
done
@@ -1143,8 +1200,9 @@ qwen38-tiny-fp8-generate:
$(PYTHON) tools/make_qwen38_tiny.py --out ./qwen38_tiny_fp8 --fp8-experts
qwen38-tiny-fp8-check: qwen38-tiny-fp8-generate qwen38$(EXE)
- @for native in 1 0; do for cap in 1 2; do \
- Q38_NATIVE_FP8=$$native OMP_NUM_THREADS=2 SNAP=./qwen38_tiny_fp8 ./qwen38$(EXE) $$cap 8 ./qwen38_tiny_fp8/ref.json || exit $$?; \
+ @for trunk in 1024 0; do for native in 1 0; do for cap in 1 2; do \
+ Q38_TRUNK_MIN_KB=$$trunk Q38_NATIVE_FP8=$$native OMP_NUM_THREADS=2 SNAP=./qwen38_tiny_fp8 ./qwen38$(EXE) $$cap 8 ./qwen38_tiny_fp8/ref.json || exit $$?; \
+ done; \
done; \
done
@@ -1156,30 +1214,51 @@ qwen38-tier-engine-check: qwen38-tiny-fp8-generate tests/test_qwen38_tier_engine
# Same tier sources as the engine: the test includes qwen36.c, so with CUDA=1
# it needs qwen36_tier.c and the backend object too (without CUDA, the header's
# inline stubs cover it and both are empty).
-tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/test_qwen36_ctx$(EXE): tests/test_qwen36_ctx.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+ $(CC) $(CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(LDFLAGS)
+
+# CACHE_ROUTE: route_select() against a residency table, no model needed.
+tests/test_qwen36_cache_route$(EXE): tests/test_qwen36_cache_route.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(LDFLAGS)
-tests/test_qwen36_dense_batch$(EXE): tests/test_qwen36_dense_batch.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/test_qwen36_dense_batch$(EXE): tests/test_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
-tests/test_qwen36_json_escape$(EXE): tests/test_qwen36_json_escape.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/test_qwen36_json_escape$(EXE): tests/test_qwen36_json_escape.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
# Both tokenizer.json merge spellings ("a b" strings and ["a","b"] pairs)
# must index the same merge table; the pair form is what Qwen3.6 ships.
-tests/test_qwen36_tok_merges$(EXE): tests/test_qwen36_tok_merges.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/test_qwen36_tok_merges$(EXE): tests/test_qwen36_tok_merges.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
# A byte-counted serving payload may end mid-character; the pre-tokenizer must
# not read past it. qwen38 already gates this (tests/test_qwen38_tokenizer.c).
-tests/test_qwen36_tok_truncated$(EXE): tests/test_qwen36_tok_truncated.c qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/test_qwen36_tok_truncated$(EXE): tests/test_qwen36_tok_truncated.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+ $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
+
+# push_id / bpe_piece must refuse loudly, not crash, on a failed realloc during growth.
+tests/test_qwen36_encode_oom$(EXE): tests/test_qwen36_encode_oom.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+ $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
+
+# slot_ensure_int8's wiring onto #1271's unpack_int4_to_int8: regression check
+# that its output matches the old separate scalar nibble-unpack loop it replaced.
+tests/test_qwen36_slot_int8$(EXE): tests/test_qwen36_slot_int8.c qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+ $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
+
+# the dense trunk's integer path: quantizer contract, int4 planar packing, dispatch.
+tests/test_qwen36_dense_idot$(EXE): tests/test_qwen36_dense_idot.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+ $(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
+
+# #1653: added tokens are split out before the regex pre-tokenizer, HF-style.
+tests/test_qwen36_tokenizer$(EXE): tests/test_qwen36_tokenizer.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
# Reproducible local timing evidence; intentionally not a noisy CI perf gate.
-tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/bench_qwen36_dense_batch$(EXE): tests/bench_qwen36_dense_batch.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c qwen36_tier.h st.h json.h compat.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
-inkling$(EXE): inkling.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda_ink.h backend_metal.h edge_adapters.h edge_runtime.h edge_tok_internal.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(INK_CUDA_OBJ) $(METAL_OBJ)
+inkling$(EXE): inkling.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h serve_budget.h backend_cuda_ink.h backend_metal.h edge_adapters.h edge_runtime.h edge_tok_internal.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(INK_CUDA_OBJ) $(METAL_OBJ)
$(CC) $(CFLAGS) inkling.c $(INK_CUDA_OBJ) $(METAL_OBJ) -o inkling$(EXE) $(LDFLAGS)
# ENGINES WITHOUT A CUDA BACKEND (#783).
@@ -1196,10 +1275,10 @@ NOCUDA_LDFLAGS = $(filter-out -lcudart -lstdc++ -lcuda -L$(CUDA_HOME)/lib64 \
# GLM-5.3-Flash: routed experts stream from the int4-gs64 container.
# METAL=1 accelerates resident matrices and routed MoE; CPU remains fallback.
-glm53$(EXE): glm53.c decode_batch.h pin_pool.h cli_args.h st.h json.h stop_ids.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h serve_poll.h route_trace.h quant.h hyper_connections.h delta_attention.h sparse_index.h vision_tower.h backend_metal.h backend_vulkan.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(METAL_OBJ) $(VK_OBJ) $(VK_SPV)
+glm53$(EXE): glm53.c sse41_kernels.h decode_batch.h pin_pool.h cli_args.h st.h json.h stop_ids.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h serve_poll.h route_trace.h quant.h fp8_format.h idot.h hyper_connections.h delta_attention.h sparse_index.h vision_tower.h backend_metal.h backend_vulkan.h edge_adapter_internal.h edge_adapters.h edge_runtime.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(METAL_OBJ) $(VK_OBJ) $(VK_SPV)
$(CC) $(CFLAGS) glm53.c $(METAL_OBJ) $(VK_OBJ) -o glm53$(EXE) $(LDFLAGS)
-kimi_k3$(EXE): kimi_k3.c cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda.h backend_metal.h backend_vulkan.h edge_adapters.h edge_runtime.h edge_tok_internal.h hybrid_split.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(CUDA_OBJ) $(VK_OBJ) $(VK_SPV) $(METAL_OBJ)
+kimi_k3$(EXE): kimi_k3.c sse41_kernels.h serve_budget.h cli_args.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h kv_prefix.h pin_pool.h serve_codec.h backend_cuda.h backend_metal.h backend_vulkan.h edge_adapters.h edge_runtime.h edge_tok_internal.h hybrid_split.h segment_adapter_internal.h segment_adapters.h segment_runtime.h $(CUDA_OBJ) $(VK_OBJ) $(VK_SPV) $(METAL_OBJ)
$(CC) $(CFLAGS) kimi_k3.c $(CUDA_OBJ) $(VK_OBJ) $(METAL_OBJ) -o kimi_k3$(EXE) $(LDFLAGS)
# Use a baseline that matches the compiler target. macOS already targets a
@@ -1227,8 +1306,8 @@ iobench$(EXE): iobench.c compat.h
tests/test_serve_sentinel$(EXE): tests/test_serve_sentinel.c compat.h serve_codec.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_cluster_protocol$(EXE): tests/test_cluster_protocol.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_cluster_protocol$(EXE): tests/test_cluster_protocol.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
tests/test_ue8m0$(EXE): tests/test_ue8m0.c st.h json.h compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1239,10 +1318,13 @@ tests/test_json$(EXE): tests/test_json.c json.h
tests/test_tok_o200k$(EXE): tests/test_tok_o200k.c tok.h tok_unicode.h tok_unicode_o200k.h json.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_k3_ram_budget$(EXE): tests/test_k3_ram_budget.c kimi_k3.c st.h tok.h quant.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h
+tests/test_tok_gpt2$(EXE): tests/test_tok_gpt2.c tok.h tok_unicode.h tok_unicode_o200k.h json.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_k3_mmap$(EXE): tests/test_k3_mmap.c kimi_k3.c st.h tok.h quant.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h
+tests/test_k3_ram_budget$(EXE): tests/test_k3_ram_budget.c sse41_kernels.h kimi_k3.c st.h tok.h quant.h fp8_format.h idot.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
+tests/test_k3_mmap$(EXE): tests/test_k3_mmap.c sse41_kernels.h kimi_k3.c st.h tok.h quant.h fp8_format.h idot.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(VK_OBJ)
$(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS)
tests/test_tok_kimi_tiny$(EXE): tests/test_tok_kimi_tiny.c tok.h tok_unicode.h tok_unicode_o200k.h json.h
@@ -1253,13 +1335,19 @@ tests/test_st_pread$(EXE): tests/test_st_pread.c st.h json.h compat.h
tests/test_st_slice$(EXE): tests/test_st_slice.c st.h json.h compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h edge_runtime.c edge_runtime.h edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h route_trace.h tok_unicode.h tok_unicode_o200k.h
+tests/test_qwen38_tokenizer$(EXE): tests/test_qwen38_tokenizer.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h edge_runtime.c edge_runtime.h edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h
$(CC) $(NOCUDA_CFLAGS) $< edge_runtime.c -o $@ $(NOCUDA_LDFLAGS)
-tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h tok_unicode.h tok_unicode_o200k.h
+tests/test_qwen38_config$(EXE): tests/test_qwen38_config.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h
+ $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
+
+tests/test_dsv41_serve_budget$(EXE): tests/test_dsv41_serve_budget.c sse41_kernels.h deepseek_v41.c cli_args.h st.h json.h tok.h tok_unicode.h compat.h omp_tune.h kv_prefix.h pin_pool.h quant.h fp8_format.h idot.h sparse_attn.h hyper_connections.h serve_codec.h serve_poll.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
-tests/test_qwen38_serve_framing$(EXE): tests/test_qwen38_serve_framing.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
+tests/test_qwen38_idot$(EXE): tests/test_qwen38_idot.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h tok_unicode.h tok_unicode_o200k.h
+ $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
+
+tests/test_qwen38_serve_framing$(EXE): tests/test_qwen38_serve_framing.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
tests/test_qwen38_vision$(EXE): tests/test_qwen38_vision.c qwen38_vision.h st.h json.h compat.h
@@ -1281,13 +1369,13 @@ qwen38-vision-serve-check: qwen38$(EXE)
$(PYTHON) tools/make_edge_tiny_tokenizer.py --vocab-size 64 ./qwen38_mm_tiny
$(PYTHON) tests/test_qwen38_vision_serve.py --binary ./qwen38$(EXE) --fixture ./qwen38_mm_tiny
-tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
+tests/test_qwen38_prefix$(EXE): tests/test_qwen38_prefix.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
-tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
+tests/test_qwen38_metrics$(EXE): tests/test_qwen38_metrics.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
-tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c qwen38.c qwen38_core.h qwen38_nfc.h qwen38_nfc_tables.h segment_runtime.c segment_runtime.h segment_adapters.h segment_adapter_internal.h st.h json.h compat.h quant.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
+tests/test_qwen38_native_weights$(EXE): tests/test_qwen38_native_weights.c sse41_kernels.h qwen38.c qwen38_core.h kv_prefix.h qwen38_nfc.h qwen38_nfc_tables.h segment_runtime.c segment_runtime.h segment_adapters.h segment_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h serve_codec.h tok_unicode.h tok_unicode_o200k.h
$(CC) $(NOCUDA_CFLAGS) $< segment_runtime.c -o $@ $(NOCUDA_LDFLAGS)
tests/test_st_map$(EXE): tests/test_st_map.c st.h json.h compat.h
@@ -1299,6 +1387,28 @@ tests/test_dup_name_refusal$(EXE): tests/test_dup_name_refusal.c st.h json.h com
tests/test_st_shape$(EXE): tests/test_st_shape.c st.h json.h compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+# Exhaustive bit-exact gate for st.h's bf16_to_f32_bulk/f16_to_f32_bulk AVX2 tier
+# (all 65536 patterns per format -- see the file for why no tolerance applies).
+tests/test_st_f16_bf16_simd$(EXE): tests/test_st_f16_bf16_simd.c st.h json.h compat.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
+# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march
+# so the SSE4.1 body in st.h is actually exercised even on a devbox that
+# would otherwise always pick the AVX2 tier.
+tests/test_st_f16_bf16_simd_sse41$(EXE): tests/test_st_f16_bf16_simd.c st.h json.h compat.h
+ $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS)
+
+# bf16_to_f32_bulk/f16_to_f32_bulk vs the scalar per-element reference, NOT a
+# test gate. Build on demand: make tests/bench_st_f16_bf16_simd ARCH=native
+tests/bench_st_f16_bf16_simd$(EXE): tests/bench_st_f16_bf16_simd.c st.h json.h compat.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
+# Synthetic peak-RSS proxy for qwen36.c's dense-int8-during-load change
+# (load_tq vs the old post-hoc qdw_register pass), NOT a test gate. Build on
+# demand: make tests/bench_load_tq_peak_rss
+tests/bench_load_tq_peak_rss$(EXE): tests/bench_load_tq_peak_rss.c
+ $(CC) -O3 -march=native $< -o $@ -lm
+
tests/test_st$(EXE): tests/test_st.c st.h json.h compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1333,25 +1443,30 @@ tests/test_grammar$(EXE): tests/test_grammar.c grammar.h
# equivalence under the shipping flags is covered by the differential dump
# (old vs new rope_interleave are byte-identical); this test guards the
# source-level formula equivalence and must not depend on contraction luck.
-tests/test_rope_invfreq$(EXE): tests/test_rope_invfreq.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) -ffp-contract=off $< -o $@ $(LDFLAGS)
+tests/test_rope_invfreq$(EXE): tests/test_rope_invfreq.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) -ffp-contract=off $< $(VK_OBJ) -o $@ $(LDFLAGS)
# schema->GBNF compile cache (#7): grammar_reset must equal a fresh setup, and
# the GrDraft.src ownership must not leak/double-free. Includes colibri.c.
-tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_grammar_cache$(EXE): tests/test_grammar_cache.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
# Standalone: drives a faithful miniature of moe()'s routing+accumulate and links
# the SAME abl.h the engine links -- no model/weights needed (the ablation-logic gate).
tests/test_ablate$(EXE): tests/test_ablate.c abl.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h json.h
+# Standalone: exercises the DEGRADE_ZERO miss-slot zero-fill logic extracted from
+# moe() -- no model/weights needed (issue #865).
+tests/test_degrade_zero$(EXE): tests/test_degrade_zero.c
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_spec_decode_state$(EXE): tests/test_spec_decode_state.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/test_schema_gbnf$(EXE): tests/test_schema_gbnf.c schema_gbnf.h grammar.h json.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_spec_decode_state$(EXE): tests/test_spec_decode_state.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
tests/test_decode_batch$(EXE): tests/test_decode_batch.c decode_batch.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1361,13 +1476,16 @@ tests/test_pin_pool$(EXE): tests/test_pin_pool.c pin_pool.h kv_prefix.h
tests/test_serve_codec$(EXE): tests/test_serve_codec.c serve_codec.h compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_serve_budget$(EXE): tests/test_serve_budget.c serve_budget.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
tests/test_segment_runtime$(EXE): tests/test_segment_runtime.c segment_runtime.c segment_runtime.h
$(CC) $(CFLAGS) tests/test_segment_runtime.c segment_runtime.c -o $@ $(LDFLAGS)
tests/test_segment_conformance$(EXE): tests/test_segment_conformance.c tests/segment_conformance_fixtures.c tests/segment_conformance_fixtures.h segment_runtime.c segment_runtime.h
$(CC) $(CFLAGS) tests/test_segment_conformance.c tests/segment_conformance_fixtures.c segment_runtime.c -o $@ $(LDFLAGS)
-tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ)
+tests/test_inkling_serve_framing$(EXE): tests/test_inkling_serve_framing.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h serve_budget.h $(INK_CUDA_OBJ) $(METAL_OBJ)
$(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS)
tests/test_inkling_shared_batch$(EXE): tests/test_inkling_shared_batch.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ)
@@ -1381,58 +1499,68 @@ tests/bench_inkling_shared_batch$(EXE): tests/bench_inkling_shared_batch.c inkli
tests/test_inkling_cache_index$(EXE): tests/test_inkling_cache_index.c inkling.c st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h omp_tune.h route_trace.h kv_prefix.h serve_codec.h $(INK_CUDA_OBJ) $(METAL_OBJ)
$(CC) $(CFLAGS) $< $(INK_CUDA_OBJ) $(METAL_OBJ) -o $@ $(LDFLAGS)
-tests/test_kimi_serve_framing$(EXE): tests/test_kimi_serve_framing.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h
+tests/test_kimi_serve_framing$(EXE): tests/test_kimi_serve_framing.c sse41_kernels.h serve_budget.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ)
$(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS)
-tests/test_kimi_cache_index$(EXE): tests/test_kimi_cache_index.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h
+tests/test_kimi_cache_index$(EXE): tests/test_kimi_cache_index.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ)
$(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS)
-tests/test_k3_chat_tools$(EXE): tests/test_k3_chat_tools.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h
+tests/test_k3_chat_tools$(EXE): tests/test_k3_chat_tools.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ)
$(CC) $(NOCUDA_CFLAGS) $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS)
# olmoe's matmul_q, not colibri's: compares the IDOT path against FP32
# ACTIVATIONS rather than an integer reference. NOCUDA_* for the same reason
# the olmoe target uses it -- olmoe.c contains no COLI_CUDA code.
-tests/test_olmoe_matmul_q$(EXE): tests/test_olmoe_matmul_q.c olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
+tests/test_olmoe_matmul_q$(EXE): tests/test_olmoe_matmul_q.c sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
-tests/test_olmoe_serve_framing$(EXE): tests/test_olmoe_serve_framing.c olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
+tests/test_olmoe_serve_framing$(EXE): tests/test_olmoe_serve_framing.c serve_budget.h sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
-tests/test_olmoe_cache_index$(EXE): tests/test_olmoe_cache_index.c olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
+tests/test_olmoe_cache_index$(EXE): tests/test_olmoe_cache_index.c sse41_kernels.h olmoe.c st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
$(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
-tests/test_qwen36_cache_index$(EXE): tests/test_qwen36_cache_index.c qwen36.c expert_ffn.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
+tests/test_qwen36_cache_index$(EXE): tests/test_qwen36_cache_index.c gsgemv.h qgemv.h sse41_kernels.h qwen36.c expert_ffn.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h qwen36_tier.h $(QWEN36_TIER_SRC) $(CUDA_OBJ)
$(CC) $(QWEN36_CFLAGS) $< $(QWEN36_TIER_SRC) $(CUDA_OBJ) -o $@ $(QWEN36_LDFLAGS)
-tests/test_idot$(EXE): tests/test_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_idot$(EXE): tests/test_idot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_i4_grouped$(EXE): tests/test_i4_grouped.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) -DCOLI_HAVE_GROUPED_PAIR $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_stops$(EXE): tests/test_stops.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+# Force the grouped-int4 oracle onto the SSE4.1 tier even on an AVX2 host.
+tests/test_i4_grouped_sse41$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) -O3 -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_cfg_topk$(EXE): tests/test_cfg_topk.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_i4_grouped_sse41_o1$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) -O1 -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_topp$(EXE): tests/test_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_i4_grouped_sse41_no_contract$(EXE): tests/test_i4_grouped.c colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h idot.h sse41_kernels.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) -O3 -ffp-contract=off -DCOLI_HAVE_GROUPED_PAIR -DCOLI_I4_GROUPED_SCALAR_EXACT -msse4.1 -mno-avx2 -mno-fma $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
+tests/test_stops$(EXE): tests/test_stops.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
+tests/test_cfg_topk$(EXE): tests/test_cfg_topk.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
+tests/test_topp$(EXE): tests/test_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
# bench_topp is a microbenchmark (old qsort vs new heap partial-select, #335), NOT a test
# gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_topp
-tests/bench_topp$(EXE): tests/bench_topp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/bench_topp$(EXE): tests/bench_topp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_sample_nan$(EXE): tests/test_sample_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_sample_nan$(EXE): tests/test_sample_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_temp_env$(EXE): tests/test_temp_env.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_temp_env$(EXE): tests/test_temp_env.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c colibri.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_kv_alloc$(EXE): tests/test_kv_alloc.c sse41_kernels.h colibri.c oracle.h st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
tests/test_kv_fp8$(EXE): tests/test_kv_fp8.c kv_fp8.h decode_batch.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1440,15 +1568,15 @@ tests/test_kv_fp8$(EXE): tests/test_kv_fp8.c kv_fp8.h decode_batch.h
tests/test_kv_tq$(EXE): tests/test_kv_tq.c kv_tq.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_kv_disk$(EXE): tests/test_kv_disk.c colibri.c st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_kv_disk$(EXE): tests/test_kv_disk.c sse41_kernels.h colibri.c oracle.h st.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h kv_fp8.h kv_tq.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
# fmt=6 kernel oracle: needs the generated grid table, and its fixture comes from
# the reference codec (tools/make_e8_fixture.py) — regenerate if the layout moves.
-tests/test_e8_kernel$(EXE): tests/test_e8_kernel.c quant.h
+tests/test_e8_kernel$(EXE): tests/test_e8_kernel.c sse41_kernels.h quant.h fp8_format.h idot.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_e4m3_vector$(EXE): tests/test_e4m3_vector.c quant.h
+tests/test_e4m3_vector$(EXE): tests/test_e4m3_vector.c sse41_kernels.h quant.h fp8_format.h idot.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
tests/test_stop_ids$(EXE): tests/test_stop_ids.c stop_ids.h json.h
@@ -1475,13 +1603,13 @@ fuzz-rans: tests/fuzz_rans.c rans.h
tests/fuzz_rans.c -o tests/fuzz_rans -lm
./tests/fuzz_rans
-tests/test_int3$(EXE): tests/test_int3.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_int3$(EXE): tests/test_int3.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_int3_load$(EXE): tests/test_int3_load.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_int3_load$(EXE): tests/test_int3_load.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c quant.h
+tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c sse41_kernels.h quant.h fp8_format.h idot.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
# Host-only: backend_cuda.h's format predicate is plain C, so its truth table is
@@ -1490,17 +1618,24 @@ tests/test_fp8_passthrough$(EXE): tests/test_fp8_passthrough.c quant.h
tests/test_cuda_fmt_guard$(EXE): tests/test_cuda_fmt_guard.c backend_cuda.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_fp8_load$(EXE): tests/test_fp8_load.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
+# The fmt=8 LUT-gate state machine. Like test_cuda_fmt_guard it links no CUDA
+# object and needs no toolchain: the two decisions live as pure predicates in
+# backend_cuda.h and backend_cuda.cu calls them, so this pins the engine's own
+# logic on a plain CPU build.
+tests/test_cuda_lut_gate$(EXE): tests/test_cuda_lut_gate.c backend_cuda.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_qt_addrow$(EXE): tests/test_qt_addrow.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_fp8_load$(EXE): tests/test_fp8_load.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_logit_nan$(EXE): tests/test_logit_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_qt_addrow$(EXE): tests/test_qt_addrow.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_router_nan$(EXE): tests/test_router_nan.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_logit_nan$(EXE): tests/test_logit_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
+tests/test_router_nan$(EXE): tests/test_router_nan.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
tests/test_i4_acc512$(EXE): tests/test_i4_acc512.c
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1511,7 +1646,7 @@ tests/test_compat_direct$(EXE): tests/test_compat_direct.c compat.h
tests/test_expert_store_ops$(EXE): tests/test_expert_store_ops.c expert_store.h tensor.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_native_quant$(EXE): tests/test_native_quant.c deepseek_v4.c native_quant.h tensor.h quant.h
+tests/test_native_quant$(EXE): tests/test_native_quant.c sse41_kernels.h deepseek_v4.c native_quant.h tensor.h quant.h fp8_format.h idot.h
$(CC) $(CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT deepseek_v4.c $< -o $@ $(LDFLAGS)
tests/test_edge_runtime$(EXE): tests/test_edge_runtime.c edge_runtime.c edge_runtime.h
@@ -1549,14 +1684,14 @@ SEGMENT_RUNTIME_LIB = $(SEGMENT_BUILD_DIR)/libcolibri_segment_edge.a
$(SEGMENT_BUILD_DIR):
mkdir -p $@
-$(SEGMENT_BUILD_DIR)/glm.o: colibri.c segment_runtime.h edge_runtime.h \
+$(SEGMENT_BUILD_DIR)/glm.o: colibri.c sse41_kernels.h oracle.h segment_runtime.h edge_runtime.h \
segment_adapters.h edge_adapters.h segment_adapter_internal.h \
edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DCOLIBRI_NO_MAIN -c colibri.c -o $@
-$(SEGMENT_BUILD_DIR)/glm53.o: glm53.c segment_runtime.h edge_runtime.h \
+$(SEGMENT_BUILD_DIR)/glm53.o: glm53.c sse41_kernels.h segment_runtime.h edge_runtime.h \
segment_adapters.h edge_adapters.h segment_adapter_internal.h \
- edge_adapter_internal.h st.h quant.h tok.h hyper_connections.h \
+ edge_adapter_internal.h st.h quant.h fp8_format.h tok.h hyper_connections.h \
delta_attention.h sparse_index.h vision_tower.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DGLM53_NO_MAIN -c glm53.c -o $@
@@ -1565,29 +1700,29 @@ $(SEGMENT_BUILD_DIR)/inkling.o: inkling.c segment_runtime.h edge_runtime.h \
edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DINKLING_NO_MAIN -c inkling.c -o $@
-$(SEGMENT_BUILD_DIR)/kimi.o: kimi_k3.c segment_runtime.h edge_runtime.h \
+$(SEGMENT_BUILD_DIR)/kimi.o: kimi_k3.c sse41_kernels.h segment_runtime.h edge_runtime.h \
segment_adapters.h edge_adapters.h segment_adapter_internal.h \
edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DKIMI_K3_NO_MAIN -c kimi_k3.c -o $@
-$(SEGMENT_BUILD_DIR)/olmoe.o: olmoe.c segment_runtime.h edge_runtime.h \
+$(SEGMENT_BUILD_DIR)/olmoe.o: olmoe.c sse41_kernels.h segment_runtime.h edge_runtime.h \
segment_adapters.h edge_adapters.h segment_adapter_internal.h \
edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DOLMOE_NO_MAIN -c olmoe.c -o $@
-$(SEGMENT_BUILD_DIR)/qwen36.o: qwen36.c segment_runtime.h edge_runtime.h \
+$(SEGMENT_BUILD_DIR)/qwen36.o: qwen36.c gsgemv.h qgemv.h sse41_kernels.h segment_runtime.h edge_runtime.h \
segment_adapters.h edge_adapters.h segment_adapter_internal.h \
edge_adapter_internal.h st.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DQWEN36_NO_MAIN -c qwen36.c -o $@
-$(SEGMENT_BUILD_DIR)/qwen38.o: qwen38.c qwen38_core.h segment_runtime.h \
+$(SEGMENT_BUILD_DIR)/qwen38.o: qwen38.c sse41_kernels.h qwen38_core.h kv_prefix.h segment_runtime.h \
segment_adapters.h segment_adapter_internal.h edge_runtime.h \
- edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h \
+ edge_adapters.h edge_adapter_internal.h st.h json.h compat.h quant.h fp8_format.h \
qwen38_nfc.h qwen38_nfc_tables.h route_trace.h tok_unicode.h \
tok_unicode_o200k.h | $(SEGMENT_BUILD_DIR)
$(CC) $(SEGMENT_CPU_CFLAGS) -DQWEN38_NO_MAIN -c qwen38.c -o $@
-$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: deepseek_v4.c deepseek_v4.h \
+$(SEGMENT_V4_OBJS): $(SEGMENT_BUILD_DIR)/%.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h \
deepseek_v4_internal.h segment_runtime.h segment_adapters.h \
edge_runtime.h edge_adapters.h segment_adapter_internal.h \
edge_adapter_internal.h edge_tok_internal.h st.h | $(SEGMENT_BUILD_DIR)
@@ -1714,7 +1849,7 @@ V4_TEST_LINK_OBJS += COLI_V4_UNIT_GPU.o backend_loader_dsv4.o
V4_TEST_EXTRA_TARGETS = COLI_V4_UNIT_GPU.o backend_loader_dsv4.o
endif
-tests/test_deepseek_v4$(EXE): tests/test_deepseek_v4.c deepseek_v4.c deepseek_v4.h compat.h \
+tests/test_deepseek_v4$(EXE): tests/test_deepseek_v4.c sse41_kernels.h deepseek_v4.c deepseek_v4.h compat.h \
Makefile.deepseek-v4.units Makefile.deepseek-v4
$(MAKE) -f Makefile.deepseek-v4 ARCH=$(ARCH) deepseek-v4-test-objs \
COLI_V4_UNIT_ST.o COLI_V4_UNIT_CONFIG.o COLI_V4_UNIT_MATH.o COLI_V4_UNIT_SPARSE_ATTENTION.o \
@@ -1739,16 +1874,16 @@ V4_OWNERSHIP_OBJS = \
$(V4_OWN_DIR):
mkdir -p $(V4_OWN_DIR)
-$(V4_OWN_DIR)/COLI_V4_UNIT_RUNTIME.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR)
+$(V4_OWN_DIR)/COLI_V4_UNIT_RUNTIME.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR)
$(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_RUNTIME -c deepseek_v4.c -o $@
-$(V4_OWN_DIR)/COLI_V4_UNIT_CONFIG.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR)
+$(V4_OWN_DIR)/COLI_V4_UNIT_CONFIG.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h | $(V4_OWN_DIR)
$(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_CONFIG -c deepseek_v4.c -o $@
-$(V4_OWN_DIR)/COLI_V4_UNIT_ST.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h st.h | $(V4_OWN_DIR)
+$(V4_OWN_DIR)/COLI_V4_UNIT_ST.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h st.h | $(V4_OWN_DIR)
$(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_ST -c deepseek_v4.c -o $@
-$(V4_OWN_DIR)/COLI_V4_UNIT_NATIVE_QUANT.o: deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h quant.h | $(V4_OWN_DIR)
+$(V4_OWN_DIR)/COLI_V4_UNIT_NATIVE_QUANT.o: deepseek_v4.c sse41_kernels.h deepseek_v4.h deepseek_v4_internal.h quant.h fp8_format.h idot.h | $(V4_OWN_DIR)
$(CC) $(V4_OWN_CFLAGS) -DCOLI_V4_UNIT_NATIVE_QUANT -c deepseek_v4.c -o $@
# The engine (RUNTIME unit) opens its expert store through the pluggable
@@ -1760,7 +1895,7 @@ $(V4_OWN_DIR)/expert_store_registry.o: expert_store_registry.c expert_store_regi
tests/test_v4_ownership$(EXE): tests/test_v4_ownership.c $(V4_OWNERSHIP_OBJS)
$(CC) $(V4_OWN_CFLAGS) $< $(V4_OWNERSHIP_OBJS) -o $@ -pthread $(LDFLAGS)
-tests/test_v4_serve_framing$(EXE): tests/test_v4_serve_framing.c deepseek_v4.c \
+tests/test_v4_serve_framing$(EXE): tests/test_v4_serve_framing.c sse41_kernels.h deepseek_v4.c \
deepseek_v4.h deepseek_v4_internal.h serve_codec.h Makefile.deepseek-v4 Makefile.deepseek-v4.units
$(MAKE) -f Makefile.deepseek-v4 ARCH=$(ARCH) $@
@@ -1772,6 +1907,9 @@ tests/test_route_trace$(EXE): tests/test_route_trace.c route_trace.h compat.h
tests/test_cli_args$(EXE): tests/test_cli_args.c cli_args.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_oracle$(EXE): tests/test_oracle.c oracle.h json.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
# Il tier CUDA con un backend finto: gira SENZA GPU perche' il test definisce i
# coli_cuda_* e registra cosa riceve. E' il solo modo di provare in CI che un
# esperto arriva davvero in VRAM e nel formato giusto (#1331) -- un test che si
@@ -1785,12 +1923,20 @@ tests/test_qwen36_tier_int8$(EXE): tests/test_qwen36_tier_int8.c tests/qwen36_fa
tests/test_qwen36_tier_multidev$(EXE): tests/test_qwen36_tier_multidev.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+# Budget separate gate/up/down scale allocations at their own size classes.
+tests/test_qwen36_tier_scale_budget$(EXE): tests/test_qwen36_tier_scale_budget.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
# Same fake backend, single device: proves qt_shutdown returns (under an
# alarm(10) watchdog) instead of hanging forever when a group is still open
# and an LFRU swap is parked waiting for cv_take (#1340).
tests/test_qwen36_tier_shutdown$(EXE): tests/test_qwen36_tier_shutdown.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+# Release owned projections and experts before CUDA teardown, then reopen safely.
+tests/test_qwen36_tier_release$(EXE): tests/test_qwen36_tier_release.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
# Same fake backend, but driving the ENGINE: the warmstart in qwen36.c hands the
# tier raw pointers into the RAM expert slots, so only a test that includes
# qwen36.c can prove the weights behind them are still there afterwards -- on an
@@ -1806,6 +1952,10 @@ tests/test_qwen36_tier_shutdown$(EXE): tests/test_qwen36_tier_shutdown.c tests/q
tests/test_qwen36_tier_invariants$(EXE): tests/test_qwen36_tier_invariants.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+# Async collection errors must drain all devices and reject partial output.
+tests/test_qwen36_tier_take_error$(EXE): tests/test_qwen36_tier_take_error.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
# Same fake backend, with uploads that take time: qt_fill_wait must not return
# until the last enqueued expert is RESIDENT, not merely dequeued -- the engine
# frees the RAM int8 copies right after it (#1360 saw the gap one run in
@@ -1821,6 +1971,10 @@ tests/test_qwen36_tier_fill_wait$(EXE): tests/test_qwen36_tier_fill_wait.c tests
tests/test_qwen36_tier_autoplace$(EXE): tests/test_qwen36_tier_autoplace.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+# Failed tier startup must release host storage and initialized synchronization.
+tests/test_qwen36_tier_init_failure$(EXE): tests/test_qwen36_tier_init_failure.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
# Same fake backend: the fp8 streaming mode a model whose experts do not fit
# in RAM needs (Qwen3.8) -- cap < n_experts accepted, fmt=8 uploads with
# 128x128 block scales, bytes staged unchanged, no pointer kept into the
@@ -1828,21 +1982,37 @@ tests/test_qwen36_tier_autoplace$(EXE): tests/test_qwen36_tier_autoplace.c tests
tests/test_qwen36_tier_fp8$(EXE): tests/test_qwen36_tier_fp8.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+# Failed gate/up/down uploads must release every unpublished CUDA tensor.
+tests/test_qwen36_tier_rollback$(EXE): tests/test_qwen36_tier_rollback.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
# Generic resident dense matrices (the Qwen3.8 trunk) and per-offer placement.
tests/test_qwen36_tier_dense$(EXE): tests/test_qwen36_tier_dense.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
-tests/test_qwen36_tier_int8_engine$(EXE): tests/test_qwen36_tier_int8_engine.c tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h
+# DeltaNet input projection batching, state parity and block failure fallback.
+tests/test_qwen36_dnproj_batch$(EXE): tests/test_qwen36_dnproj_batch.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c qwen36_tier.c qwen36_tier.h backend_cuda.h compat.h quant.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
+# the automatic trunk placement can be withdrawn after the startup probe (fake backend).
+tests/test_qwen36_tier_withdraw$(EXE): tests/test_qwen36_tier_withdraw.c tests/qwen36_fake_cuda.h qwen36_tier.c qwen36_tier.h backend_cuda.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
+tests/test_qwen36_tier_int8_engine$(EXE): tests/test_qwen36_tier_int8_engine.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h
+ $(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
+
+# dnout / attnproj / shexp offered, placed and served from VRAM (fake backend).
+tests/test_qwen36_trunk_dense$(EXE): tests/test_qwen36_trunk_dense.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
# #1391: il decode path deve offrire gli esperti int8 al tier, non solo il warmstart
-tests/test_qwen36_tier_int8_decode$(EXE): tests/test_qwen36_tier_int8_decode.c tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h
+tests/test_qwen36_tier_int8_decode$(EXE): tests/test_qwen36_tier_int8_decode.c gsgemv.h qgemv.h sse41_kernels.h tests/qwen36_fake_cuda.h qwen36.c expert_ffn.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
# The qwen38 engine through its own main() on the fake backend: tier start
# from the FP8 fixture, reduced CPU list, qt_note on recycled slots, oracle
# tokens unchanged. Needs the fixture (qwen38-tiny-fp8-generate).
-tests/test_qwen38_tier_engine$(EXE): tests/test_qwen38_tier_engine.c tests/qwen36_fake_cuda.h qwen38.c qwen38_core.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h quant.h route_trace.h
+tests/test_qwen38_tier_engine$(EXE): tests/test_qwen38_tier_engine.c sse41_kernels.h tests/qwen36_fake_cuda.h qwen38.c qwen38_core.h kv_prefix.h qwen36_tier.c qwen36_tier.h backend_cuda.h cli_args.h st.h json.h compat.h quant.h fp8_format.h idot.h route_trace.h
$(CC) $(CFLAGS) -DCOLI_CUDA $< -o $@ $(LDFLAGS)
tests/test_serve_poll$(EXE): tests/test_serve_poll.c serve_poll.h
@@ -1858,19 +2028,22 @@ tests/test_rss_anon$(EXE): tests/test_rss_anon.c
tests/test_mem_available$(EXE): tests/test_mem_available.c compat.h
$(CC) $(CFLAGS) tests/test_mem_available.c -o $@ $(LDFLAGS)
+tests/test_exact_dot$(EXE): tests/test_exact_dot.c exact_dot.h
+ $(CC) $(CFLAGS) tests/test_exact_dot.c -o tests/test_exact_dot$(EXE) $(LDFLAGS)
+
tests/test_798_guards$(EXE): tests/test_798_guards.c st.h json.h compat.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_dsa_select$(EXE): tests/test_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_dsa_select$(EXE): tests/test_dsa_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_corpus_draft$(EXE): tests/test_corpus_draft.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_corpus_draft$(EXE): tests/test_corpus_draft.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_cap_precedence$(EXE): tests/test_cap_precedence.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_mirror_stripe_split$(EXE): tests/test_mirror_stripe_split.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_cap_precedence$(EXE): tests/test_cap_precedence.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+tests/test_mirror_stripe_split$(EXE): tests/test_mirror_stripe_split.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
tests/test_v4_hybrid_policy$(EXE): tests/test_v4_hybrid_policy.c deepseek_v4_hybrid.h hybrid_split.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1880,29 +2053,29 @@ tests/test_k3_fill_budget$(EXE): tests/test_k3_fill_budget.c hybrid_split.h
tests/test_v4_bank_pair$(EXE): tests/test_v4_bank_pair.c deepseek_v4_bank_pair.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_ram_clamp$(EXE): tests/test_ram_clamp.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_ram_clamp$(EXE): tests/test_ram_clamp.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_cap_mixed_width$(EXE): tests/test_cap_mixed_width.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_cap_mixed_width$(EXE): tests/test_cap_mixed_width.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_eslot_inflight$(EXE): tests/test_eslot_inflight.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_eslot_inflight$(EXE): tests/test_eslot_inflight.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_glm_cache_index$(EXE): tests/test_glm_cache_index.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_glm_cache_index$(EXE): tests/test_glm_cache_index.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_ssd_probe$(EXE): tests/test_ssd_probe.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
# bench_dsa_select is a microbenchmark (old qsort vs new quickselect partial-select, #356),
# NOT a test gate -- intentionally absent from TEST_BINS. Build on demand: make tests/bench_dsa_select
-tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/bench_dsa_select$(EXE): tests/bench_dsa_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
# bench_router_select is a microbenchmark (duplicate-prefix scan vs marked-score scan),
# not a test gate. Build on demand: make tests/bench_router_select.
-tests/bench_router_select$(EXE): tests/bench_router_select.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/bench_router_select$(EXE): tests/bench_router_select.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
# bench_indexer_allocations: microbenchmark (DeepSeek V4 indexer malloc vs persistent arena scratch), NOT a test gate.
@@ -1912,30 +2085,35 @@ tests/bench_indexer_allocations$(EXE): tests/bench_indexer_allocations.c
# bench_idot: microbenchmark (single-acc vs independent-acc AVX-VNNI idot), NOT a test gate.
# Build on demand on an AVX-VNNI CPU: make tests/bench_idot ARCH=native
-tests/bench_idot$(EXE): tests/bench_idot.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/bench_idot$(EXE): tests/bench_idot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
+# bench_i4p_gidot: microbenchmark (per-row vs multi-row/AMX K1b grouped planar IDOT), NOT a test gate.
+# Build on demand: make tests/bench_i4p_gidot ARCH=native
+tests/bench_i4p_gidot$(EXE): tests/bench_i4p_gidot.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
# bench_gemv_stream: microbenchmark (decode-regime GEMV bandwidth vs the read ceiling;
# frozen-baseline + deinterleaved-x candidate A/B), NOT a test gate.
# Build on demand: make tests/bench_gemv_stream ARCH=native
-tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/bench_gemv_stream$(EXE): tests/bench_gemv_stream.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
# bench_mla_simd: microbenchmark (scalar vs AVX2/NEON MLA-absorb reductions, #442),
# NOT a test gate. Build on demand: make tests/bench_mla_simd
-tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
+tests/bench_mla_simd$(EXE): tests/bench_mla_simd.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_uring$(EXE): tests/test_uring.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h route_trace.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_uring$(EXE): tests/test_uring.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h route_trace.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_pipe_block$(EXE): tests/test_pipe_block.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_pipe_block$(EXE): tests/test_pipe_block.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
-tests/test_pilot_ring$(EXE): tests/test_pilot_ring.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_moe_gs_guard$(EXE): tests/test_moe_gs_guard.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
- $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+tests/test_pilot_ring$(EXE): tests/test_pilot_ring.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+tests/test_moe_gs_guard$(EXE): tests/test_moe_gs_guard.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
tests/test_omp_tune$(EXE): tests/test_omp_tune.c omp_tune.h compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
@@ -1943,9 +2121,68 @@ tests/test_omp_tune$(EXE): tests/test_omp_tune.c omp_tune.h compat.h
tests/test_compat_env$(EXE): tests/test_compat_env.c compat.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
-tests/test_kvb_notice$(EXE): tests/test_kvb_notice.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h sample.h kv_persist.h telemetry.h
+tests/test_kvb_notice$(EXE): tests/test_kvb_notice.c sse41_kernels.h colibri.c oracle.h st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h idot.h sample.h kv_persist.h telemetry.h $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) -o $@ $(LDFLAGS)
+
+# Standalone: proves the group-scaled int8 GEMV keeps every row's float
+# operations in their original order -- the assumption qwen36's byte-identical
+# output rests on. Links the SAME gsgemv.h the engine links.
+tests/test_gsgemv$(EXE): tests/test_gsgemv.c gsgemv.h sse41_kernels.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
+# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march
+# so the SSE4.1 body in gsgemv.h is actually exercised even on a devbox that
+# would otherwise always pick the AVX2 tier.
+tests/test_gsgemv_sse41$(EXE): tests/test_gsgemv.c gsgemv.h sse41_kernels.h
+ $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS)
+
+# Standalone: proves the plain (non-group-scaled) int8 GEMV keeps its exact
+# sequence of float operations -- the assumption qwen36's byte-identical
+# output rests on. Links the SAME qgemv.h the engine links.
+tests/test_qgemv$(EXE): tests/test_qgemv.c qgemv.h sse41_kernels.h
$(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march
+# so the SSE4.1 body in qgemv.h is actually exercised even on a devbox that
+# would otherwise always pick the AVX2 tier.
+tests/test_qgemv_sse41$(EXE): tests/test_qgemv.c qgemv.h sse41_kernels.h
+ $(CC) $(CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(LDFLAGS)
+
+# olmoe's dot_i8_16 (ARM NEON/AVX2/SSE4.1 variants): must be bit-exact against
+# a scalar int8 reference -- pure integer arithmetic, no tolerance needed.
+# NOCUDA_* for the same reason the other olmoe.c-including test rules use it.
+tests/test_olmoe_dot_i8_16$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c sse41_kernels.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
+ $(CC) $(NOCUDA_CFLAGS) $< -o $@ $(NOCUDA_LDFLAGS)
+
+# Same gate, forced onto the SSE4.1 tier: overrides the host's default -march
+# so the SSE4.1 body in olmoe.c's dot_i8_16 is actually exercised even on a
+# devbox that would otherwise always pick the AVX2 tier.
+tests/test_olmoe_dot_i8_16_sse41$(EXE): tests/test_olmoe_dot_i8_16.c olmoe.c sse41_kernels.h st.h json.h compat.h sample.h tok.h tok_unicode.h tok_unicode_o200k.h omp_tune.h route_trace.h serve_codec.h
+ $(CC) $(NOCUDA_CFLAGS) -msse4.1 -mno-avx2 -mno-fma $< -o $@ $(NOCUDA_LDFLAGS)
+
+# layer_cuda_shard_kvb's format-allowlist refusal (see the test's file header). The
+# function only exists under -DCOLI_CUDA, so on a default (CPU) build the binary is a
+# loud SKIP; built with CUDA=1 it links backend_cuda.o ($(CUDA_OBJ)) and exercises the
+# real guard -- no GPU needed, every probed path returns before any device context.
+tests/test_shard_kvb_refuse$(EXE): tests/test_shard_kvb_refuse.c colibri.c st.h uring.h json.h tok.h tok_unicode.h compat.h grammar.h tier.h quant.h fp8_format.h sample.h kv_persist.h telemetry.h $(CUDA_OBJ) $(VK_OBJ)
+ $(CC) $(CFLAGS) $< $(VK_OBJ) $(CUDA_OBJ) -o $@ $(LDFLAGS)
+
+# The e8x4g64 container loader harness. It takes a minted directory on argv and
+# is driven by tests/test_e8x4g64_mint_load.py, which builds it too -- this rule
+# exists so it also compiles under the suite's own $(CFLAGS) rather than only
+# under the driver's hand-copied flag list, and so a warning regression in it
+# fails the normal build.
+tests/test_e8x4g64_loader$(EXE): tests/test_e8x4g64_loader.c st.h quant.h fp8_format.h compat.h
+ $(CC) $(CFLAGS) $< -o $@ $(LDFLAGS)
+
+# Reachable build-only entry point for the harness above: `make check` never
+# calls this (it stays out of TEST_BINS/test-c, per TEST_EXCLUDE), so this adds
+# no time to `check`. It exists so the suite's own $(CFLAGS) and warnings are
+# actually exercised on demand, the way qwen38-tier-engine-check does for
+# test_qwen38_tier_engine below.
+.PHONY: e8x4g64-loader-check
+e8x4g64-loader-check: tests/test_e8x4g64_loader$(EXE)
+
test-c: $(TEST_BINS)
$(PYTHON) tools/run_tests.py $(TEST_BINS)
@@ -2045,7 +2282,7 @@ install: colibri$(EXE) glm53$(EXE) inkling$(EXE) kimi_k3$(EXE) olmoe$(EXE) qwen3
$(INSTALL) -m 755 deepseek_v4$(EXE) $(DESTDIR)$(LIBEXECDIR)/deepseek_v4$(EXE); \
fi
$(INSTALL) -m 644 family_registry.py resource_plan.py doctor.py autotune.py \
- openai_server.py cluster.py v4_dsml.py version.py $(DESTDIR)$(LIBEXECDIR)/
+ openai_server.py cluster.py v4_dsml.py v41_dsml.py version.py $(DESTDIR)$(LIBEXECDIR)/
$(INSTALL) -m 644 tools/*.py $(DESTDIR)$(LIBEXECDIR)/tools/
@# The dashboard is an optional build artifact (cd web && npm run build), so install
@# it only when it exists. It goes NEXT TO openai_server.py, which probes ./web/dist.
@@ -2080,5 +2317,9 @@ bench: iobench$(EXE)
tests/test_kv_prefix$(EXE): tests/test_kv_prefix.c kv_prefix.h
$(CC) $(CFLAGS) tests/test_kv_prefix.c -o tests/test_kv_prefix$(EXE) $(LDFLAGS)
-tests/test_kimi_request_state$(EXE): tests/test_kimi_request_state.c kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h omp_tune.h route_trace.h
+tests/test_kimi_request_state$(EXE): tests/test_kimi_request_state.c sse41_kernels.h kimi_k3.c kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ)
$(CC) $(NOCUDA_CFLAGS) tests/test_kimi_request_state.c $(VK_OBJ) -o tests/test_kimi_request_state$(EXE) $(NOCUDA_LDFLAGS)
+
+# Kimi CUDA dispatch/fallback with a fake backend; no CUDA toolkit required.
+tests/test_kimi_cuda_expert$(EXE): tests/test_kimi_cuda_expert.c sse41_kernels.h kimi_k3.c backend_cuda.h kv_prefix.h serve_codec.h st.h json.h tok.h tok_unicode.h tok_unicode_o200k.h compat.h quant.h fp8_format.h idot.h omp_tune.h route_trace.h $(VK_OBJ)
+ $(CC) $(NOCUDA_CFLAGS) -DCOLI_CUDA $< $(VK_OBJ) -o $@ $(NOCUDA_LDFLAGS)
diff --git a/c/Makefile.deepseek-v4 b/c/Makefile.deepseek-v4
index a7858363f..fccd0bdae 100644
--- a/c/Makefile.deepseek-v4
+++ b/c/Makefile.deepseek-v4
@@ -223,8 +223,48 @@ REGISTRY_OBJ = expert_store_registry.o
V4_SERVE_TEST := tests/test_v4_serve_framing$(if $(IS_WIN),.exe,)
V4_SERVE_TEST_OBJS := $(filter-out COLI_V4_UNIT_GENERATE_STATS.o,$(V4_OBJS))
+# The objects are named after the unit, not after the flags they were built
+# with, and three builds share them: this engine (CUDA=1 adds
+# -DCOLI_V4_GPU_TIER), the parent's test-c (no GPU tier) and its test-asan
+# (EXTRA_CFLAGS). Timestamps cannot tell those apart, so make would report
+# "is up to date" and link one build's objects into another (#1702). The
+# compiler line is recorded here and rewritten only when it changes; every
+# object compiled with $(CFLAGS) depends on it. Same scheme as git's GIT-CFLAGS.
+V4_FLAGS_STAMP = deepseek_v4.cflags
+V4_TRACK_FLAGS = $(subst ','\'',$(CC) $(CFLAGS))
+$(V4_FLAGS_STAMP): FORCE
+ @flags='$(V4_TRACK_FLAGS)'; \
+ if test x"$$flags" != x"`cat $@ 2>/dev/null`"; then \
+ test -f $@ && echo "deepseek-v4: build flags changed, rebuilding the units" >&2; \
+ echo "$$flags" > $@; \
+ fi
+FORCE:
+
+# The same defect one level down. backend_cuda_dsv4.o is named after its source,
+# but what it contains comes from $(NVCC) $(V4_NVCCFLAGS): -arch=$(CUDA_ARCH)
+# (or the -gencode preset), the -DCOLI_DSV4_NO_TC guard and the DeepGEMM defines
+# all arrive through it. $(CFLAGS) above does not cover it, so without this:
+#
+# make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 CUDA_ARCH=sm_86
+# make -f Makefile.deepseek-v4 deepseek-v4 CUDA=1 CUDA_ARCH=sm_80
+# -> no output at all, exit 0
+#
+# and the engine that asked for sm_80 links the sm_86 object, in silence. The
+# parent Makefile keeps the same class of state for the same reason, CUDA_ARCH
+# included, in .build-config (#306). Same scheme as $(V4_FLAGS_STAMP) above, so
+# a dry run (make -n) or a clean writes nothing: the file is only touched by the
+# recipe, and only when the command really changed.
+V4_CUDA_FLAGS_STAMP = deepseek_v4.cudaflags
+V4_TRACK_CUDA_FLAGS = $(subst ','\'',$(NVCC) $(V4_NVCCFLAGS))
+$(V4_CUDA_FLAGS_STAMP): FORCE
+ @flags='$(V4_TRACK_CUDA_FLAGS)'; \
+ if test x"$$flags" != x"`cat $@ 2>/dev/null`"; then \
+ test -f $@ && echo "deepseek-v4: CUDA build flags changed, rebuilding backend_cuda_dsv4.o" >&2; \
+ echo "$$flags" > $@; \
+ fi
+
.PHONY: deepseek-v4 deepseek-v4-objs deepseek-v4-clean deepseek-v4-test-objs \
- deepseek-v4-test-registry print-v4-objs
+ deepseek-v4-test-registry print-v4-objs FORCE
deepseek-v4: $(V4_BINARY)
# Build just the amalgamation unit + registry objects (no link). External
@@ -242,24 +282,25 @@ $(V4_BINARY): $(V4_OBJS)
# Pluggable expert-store backend registry (standalone; not a deepseek_v4.c unit).
# Linked into the binary so the engine can dispatch COLI_EXPERT_STORE backends.
-$(REGISTRY_OBJ): expert_store_registry.c expert_store_registry.h expert_store.h
+$(REGISTRY_OBJ): expert_store_registry.c expert_store_registry.h expert_store.h \
+ $(V4_FLAGS_STAMP)
$(CC) $(CFLAGS) -c expert_store_registry.c -o $@
$(TARGET_OBJS) $(TEST_UNIT_OBJS): %.o: deepseek_v4.c deepseek_v4.h \
- deepseek_v4_internal.h deepseek_v4_dspark.inc st.h json.h compat.h tensor.h quant.h \
+ deepseek_v4_internal.h deepseek_v4_dspark.inc st.h json.h compat.h tensor.h quant.h fp8_format.h sse41_kernels.h \
route_trace.h \
native_quant.h native_quant_batch.h native_quant_dual.h \
- native_quant_fp4_rows16.h expert_store_registry.h
+ native_quant_fp4_rows16.h expert_store_registry.h $(V4_FLAGS_STAMP)
$(CC) $(CFLAGS) -D$* -c deepseek_v4.c -o $@
$(V4_HOT_TEST_OBJ): deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h \
- st.h json.h compat.h tensor.h quant.h route_trace.h \
- native_quant.h native_quant_fp4_rows16.h expert_store_registry.h
+ st.h json.h compat.h tensor.h quant.h fp8_format.h route_trace.h sse41_kernels.h \
+ native_quant.h native_quant_fp4_rows16.h expert_store_registry.h $(V4_FLAGS_STAMP)
$(CC) $(CFLAGS) -DCOLI_V4_TEST_HOOKS \
-DCOLI_V4_UNIT_EXPERT_STORE_HOT_ROWS16 -c deepseek_v4.c -o $@
$(V4_BATCH_TEST_OBJ): deepseek_v4.c deepseek_v4.h deepseek_v4_internal.h \
- tensor.h quant.h native_quant.h native_quant_batch.h
+ tensor.h quant.h fp8_format.h native_quant.h native_quant_batch.h $(V4_FLAGS_STAMP) sse41_kernels.h
$(CC) $(CFLAGS) -DCOLI_V4_TEST_HOOKS \
-DCOLI_V4_UNIT_NATIVE_QUANT_BATCH -c deepseek_v4.c -o $@
@@ -272,11 +313,12 @@ $(V4_SERVE_TEST): tests/test_v4_serve_framing.c deepseek_v4.c deepseek_v4.h \
# The CUDA tier loader object. Compiled only on Windows, where it is appended
# to V4_OBJS and resolves the MSVC-built coli_cuda_dsv4.dll at runtime. On
# Linux it is empty so the engine links without it.
-backend_loader_dsv4.o: backend_loader_dsv4.c backend_cuda_dsv4.h
+backend_loader_dsv4.o: backend_loader_dsv4.c backend_cuda_dsv4.h $(V4_FLAGS_STAMP)
$(CC) $(CFLAGS) -c backend_loader_dsv4.c -o $@
# Linux/macOS CUDA=1: the tier compiled straight into the engine.
-backend_cuda_dsv4.o: backend_cuda_dsv4.cu backend_cuda_dsv4.h $(V4_DEEPGEMM_DEP)
+backend_cuda_dsv4.o: backend_cuda_dsv4.cu backend_cuda_dsv4.h $(V4_DEEPGEMM_DEP) \
+ $(V4_CUDA_FLAGS_STAMP)
"$(NVCC)" $(V4_NVCCFLAGS) -c backend_cuda_dsv4.cu -o $@
ifneq ($(DEEPGEMM_STAMP),)
@@ -297,4 +339,5 @@ test_expert_store_registry: test_expert_store_registry.c expert_store_registry.c
deepseek-v4-clean:
rm -f $(V4_BINARY) $(V4_OBJS) $(TEST_UNIT_OBJS) $(V4_HOT_TEST_OBJ) \
$(V4_BATCH_TEST_OBJ) \
- $(V4_SERVE_TEST) test_expert_store_registry
+ $(V4_SERVE_TEST) test_expert_store_registry $(V4_FLAGS_STAMP) \
+ $(V4_CUDA_FLAGS_STAMP)
diff --git a/c/autotune.py b/c/autotune.py
index b8f64e8ba..31d853f71 100644
--- a/c/autotune.py
+++ b/c/autotune.py
@@ -466,6 +466,16 @@ def run_once(name, overlay, launch_cap):
if proc.returncode:
raise RuntimeError(f"{name} failed ({proc.returncode})\n{output[-2000:]}")
return parse_replay(output)
+
+ def measure(name, overlay, launch_cap):
+ samples = []
+ for repeat in range(repeats):
+ progress(f"{name} cap={cap} ({repeat + 1}/{repeats})")
+ sample = run_once(name, overlay, launch_cap)
+ sample["ttft_s"] = None
+ samples.append(sample)
+ recorded_cap = launch_cap if arch in CAP_ARCHES else None
+ return _summarize_measurement(name, overlay, samples, recorded_cap)
else:
if engine_cls is None:
from openai_server import Engine as engine_cls
@@ -533,17 +543,6 @@ def collect(piece):
recorded_cap = launch_cap if arch in CAP_ARCHES else None
return _summarize_measurement(name, overlay, samples, recorded_cap)
- if arch == "glm":
- def measure(name, overlay, launch_cap):
- samples = []
- for repeat in range(repeats):
- progress(f"{name} cap={cap} ({repeat + 1}/{repeats})")
- sample = run_once(name, overlay, launch_cap)
- sample["ttft_s"] = None
- samples.append(sample)
- recorded_cap = launch_cap if arch in CAP_ARCHES else None
- return _summarize_measurement(name, overlay, samples, recorded_cap)
-
baseline = measure("baseline", {}, cap)
winner = baseline
accumulated = {}
diff --git a/c/backend_cuda.cu b/c/backend_cuda.cu
index 07d5dfc42..b7930a0f7 100644
--- a/c/backend_cuda.cu
+++ b/c/backend_cuda.cu
@@ -1,7 +1,12 @@
#include "backend_cuda.h"
+#include "fp8_format.h" /* FP8_BLOCK: the shared fmt=8 scale-block edge (see that header) */
#include "backend_gpu_compat.h"
+static_assert(FP8_BLOCK == 128, "fmt=8 on-disk containers carry ceil(dim/128)-edged scale "
+ "grids (mint tool, docs/FORMATS.md); FP8_BLOCK is container format, not a "
+ "tunable -- an edit here is a format change");
+
/* Optional fmt=8 decode candidate (COLI_CUDA_F8_WARP=2): cuda_fp8.h maps
* __nv_cvt_fp8_to_halfraw to an sm_89+ cvt instruction, with a bit-manip
* fallback below 890. CUDA-only; the HIP build keeps the LUT decode. */
@@ -63,7 +68,7 @@ struct ColiCudaTensor {
int fmt, I, O, device;
int gs; /* quant group size; 0 = per-row scales (#334) */
int ng; /* number of scale groups per row = ceil(I/gs) for fmt=4 */
- size_t scale_count; /* floats in `scales`: O per-row, O*ng grouped */
+ size_t scale_count; /* scale elements: ue8m0 bytes for fmt=7, floats otherwise */
int tracked;
int weights_owned;
#ifdef COLI_ANS
@@ -74,6 +79,11 @@ struct ColiCudaTensor {
int ragged_count;
};
+static size_t tensor_scale_bytes(const ColiCudaTensor *t) {
+ if (!t->fmt || t->fmt == 6) return 0;
+ return t->scale_count * (t->fmt == 7 ? sizeof(uint8_t) : sizeof(float));
+}
+
#ifdef COLI_ANS
struct AnsArenaChunk { uint8_t *p; size_t used,cap; };
#endif
@@ -82,6 +92,11 @@ typedef struct {
int compute_major,compute_minor;
float *x, *y, *gate, *up;
size_t x_cap, y_cap, gate_cap, up_cap;
+ /* Streaming MXFP4 weights are refreshed on every call; only storage is reused. */
+ void *mxfp4_weights, *mxfp4_scales;
+ size_t mxfp4_weights_cap, mxfp4_scales_cap;
+ void *mxfp4_expert_weights, *mxfp4_expert_scales;
+ size_t mxfp4_expert_weights_cap, mxfp4_expert_scales_cap;
/* Staging of the resident dense matvec (coli_cuda_matmul), apart from
* x/y: the expert group (coli_cuda_expert_group_issue) runs on ctx->stream
* asynchronously while the engine's thread keeps computing -- qwen38's
@@ -277,13 +292,14 @@ __device__ static inline float mx4_weight_at(const uint8_t *q, int i) {
* branch and the fall-through is a refusal.
*
* It used to be the other way round: int2 was the fall-through, so every format
- * this function does not decode -- fmt=5 (int3-g64), fmt=6 (E8/IQ3), fmt=8
- * (fp8-e4m3), and anything added later -- was read as 2-bit values and returned
- * numbers. Meanwhile the CPU functions doing the same job on the same tensor,
- * qt_addrow and qt_matvec_rows (colibri.c), both exit(1) naming the function and
- * the fmt. Two backends, identical unsupported input, one refusing and one
- * fabricating: that asymmetry is the defect, independent of any particular
- * format's arrival.
+ * this function does not decode -- fmt=5 (int3-g64), fmt=6 (E8/IQ3), and
+ * anything added later -- was read as 2-bit values and returned numbers.
+ * (fmt=8 was in that misread set too, then refused, until it gained its own
+ * explicit branch below for the absorb path.) Meanwhile the CPU functions
+ * doing the same job on the same tensor, qt_addrow and qt_matvec_rows
+ * (colibri.c), both exit(1) naming the function and the fmt. Two backends,
+ * identical unsupported input, one refusing and one fabricating: that
+ * asymmetry is the defect, independent of any particular format's arrival.
*
* WHY __trap() AND NOT A DIAGNOSTIC. This is device code inside a running
* kernel; there is no stderr to name the tensor on and no way to unwind. __trap
@@ -312,6 +328,15 @@ __device__ static float weight_at(const void *weights, int fmt, size_t row, int
const uint8_t *base = static_cast(weights) + row;
if (fmt == 0) return reinterpret_cast(base)[i];
if (fmt == 1) return static_cast(reinterpret_cast(base)[i]);
+ /* fmt=8 (fp8-e4m3): raw byte, same layout as fmt=1 (row_bytes(8,I)==I), decoded
+ * through the shared c_e4m3 LUT (same table quant_matmul's fmt==8 branch reads,
+ * uploaded once by coli_cuda_fp8_set_lut). Callers gate on the LUT being live
+ * before a fmt=8 tensor ever reaches this function (coli_cuda_tensor_upload
+ * refuses the upload otherwise), so the table is always populated here. Returns
+ * the decoded WEIGHT only, unscaled -- absorb_scale below applies the
+ * per-128x128-block scale, exactly like every other quantized fmt returns
+ * unscaled through this function. */
+ if (fmt == 8) return c_e4m3[base[i]];
const uint8_t *q = base;
if (fmt == 2 || fmt == 4) { /* fmt=4: same nibble layout */
uint8_t v = q[i >> 1];
@@ -325,13 +350,34 @@ __device__ static float weight_at(const void *weights, int fmt, size_t row, int
return 0.0f; /* not reached: __trap() does not return */
}
-/* Scale for output `row`, input element `k`. fmt=4 (grouped int4) stores ng
- * scales per row at scales[row*ng + k/gs]; every other quantized format has
- * one scale per row at scales[row]. Mirrors quant_matmul's fmt==4 branch so the
+/* Scale for output `row`, input element `k`. Three layouts reach this: fmt=4
+ * (grouped int4) stores ng scales per row at scales[row*ng + k/gs]; fmt=8
+ * (fp8-e4m3) stores one scale per 128x128 BLOCK, block-row-major, and is
+ * handled by its own branch below; every OTHER quantized format has one scale
+ * per row at scales[row]. Mirrors quant_matmul's fmt==4 branch so the
* attention absorb kernels apply per-group scales instead of the per-row
* (fmt=2) semantic that crashed #298's g64 kv_b. */
__device__ static float absorb_scale(const float *wscale, int fmt, int gs, int ng, int row, int k) {
if (!fmt) return 1.f;
+ if (fmt == 8) {
+ /* fp8-e4m3: one f32 scale per 128x128 BLOCK, block-row-major
+ * ([ceil(O/128), ceil(I/128)]), exactly quant_matmul's fmt==8 indexing
+ * (scl[i >> 7] on a scale row selected by o >> 7) and matmul_fp8's CPU
+ * reference (quant.h). `ng` here is coli_cuda_tensor_upload's t->ng,
+ * which for fmt=8 is set to ceil(I/128) specifically (not the fmt=4
+ * group count) -- see the upload-time assignment there. `gs` is unused
+ * for fmt=8 (always 0, only fmt=4 sets it), so the block edge is the
+ * fixed FP8_BLOCK constant (fp8_format.h, shared with the CPU side),
+ * not a caller-supplied group size. Rounding note: the GEOMETRY here
+ * matches quant_matmul_f8w/matmul_fp8, but their fp8 accumulation
+ * convention (f32 partial per block, scale once per partial, double
+ * across blocks) is NOT carried into the absorb kernels -- they apply
+ * the scale per element into a float accumulator, matching their own
+ * fmt=4 arm's long-standing behavior; CPU-vs-CUDA absorb divergence
+ * is an accepted, documented class (#510). */
+ int rowBlk = row / FP8_BLOCK, colBlk = k / FP8_BLOCK;
+ return wscale[(size_t)rowBlk * ng + colBlk];
+ }
if (fmt != 4) return wscale[row];
int g = k / gs; if (g >= ng) g = ng - 1; /* tail of the last (partial) group */
return wscale[(size_t)row * ng + g];
@@ -548,9 +594,9 @@ __global__ static void quant_matmul(float *y, const float *x, const void *weight
* the ORIGINAL dense path, kept for COLI_CUDA_F8_WARP=0; the default
* routes fmt=8 to quant_matmul_f8w instead (quant_matmul_launch). */
const uint8_t *wrow = static_cast(weights) + row;
- const float *scl = scales + (size_t)(o >> 7) * (size_t)((I + 127) >> 7);
+ const float *scl = scales + (size_t)(o / FP8_BLOCK) * (size_t)((I + FP8_BLOCK - 1) / FP8_BLOCK);
for (int i = threadIdx.x; i < I; i += blockDim.x)
- sum += xs[i] * c_e4m3[wrow[i]] * scl[i >> 7];
+ sum += xs[i] * c_e4m3[wrow[i]] * scl[i / FP8_BLOCK];
} else {
for (int i = threadIdx.x; i < I; i += blockDim.x)
sum += xs[i] * weight_at(weights, fmt, row, i);
@@ -634,6 +680,14 @@ __global__ static void silu_mul(float *gate, const float *up, size_t n) {
}
}
+__global__ static void situ_mul(float *gate, const float *up, size_t n, float b1, float b2) {
+ size_t i = (size_t)blockIdx.x * blockDim.x + threadIdx.x;
+ if (i < n) {
+ float g = gate[i], u = up[i];
+ gate[i] = b1 * tanhf(g / b1) * (1.f / (1.f + expf(-g))) * b2 * tanhf(u / b2);
+ }
+}
+
/* Four warps share one A tile and compute 16x64 outputs. This matters for
* prefill: the first prototype reloaded/converter A once per 16 output cols. */
__global__ static void w4a16_matmul(float *y,const float *x,const uint8_t *w,
@@ -1242,28 +1296,39 @@ extern "C" int coli_cuda_init(const int *devices, int count) {
int available = 0;
if (!devices || count < 1 || count > COLI_CUDA_MAX_DEVICES) return 0;
if (!cuda_ok(cudaGetDeviceCount(&available), "device discovery")) return 0;
- g_nctx = 0;
+ /* Validate the whole list before creating resources or replacing state. */
for (int i = 0; i < count; i++) {
- int device = devices[i];
- if (device < 0 || device >= available) {
- std::fprintf(stderr, "[CUDA] invalid device %d (available: 0..%d)\n", device, available - 1);
- g_nctx = 0;
+ if (devices[i] < 0 || devices[i] >= available) {
+ std::fprintf(stderr, "[CUDA] invalid device %d (available: 0..%d)\n", devices[i], available - 1);
return 0;
}
- if (find_ctx(device)) {
- std::fprintf(stderr, "[CUDA] duplicate device %d\n", device);
- g_nctx = 0;
+ for (int j = 0; j < i; j++) if (devices[j] == devices[i]) {
+ std::fprintf(stderr, "[CUDA] duplicate device %d\n", devices[i]);
return 0;
}
+ }
+ if (g_nctx) {
+ /* Same decision as before, routed through the shared predicate in
+ * backend_cuda.h so a host-side test can pin it without nvcc; the
+ * return value is unchanged (1 for the same set, 0 otherwise). */
+ int live[COLI_CUDA_MAX_DEVICES];
+ for (int i = 0; i < g_nctx; i++) live[i] = g_ctx[i].device;
+ int d = coli_cuda_init_disposition(g_nctx, count, devices, live);
+ if (d == COLI_CUDA_INIT_REFUSE)
+ std::fprintf(stderr, "[CUDA] device list change requires shutdown first\n");
+ return d == COLI_CUDA_INIT_ACCEPT;
+ }
+ for (int i = 0; i < count; i++) {
+ int device = devices[i];
DeviceContext *ctx = &g_ctx[g_nctx];
*ctx = {};
ctx->device = device;
- if (!select_ctx(ctx)) { g_nctx = 0; return 0; }
+ if (!select_ctx(ctx)) { coli_cuda_shutdown(); return 0; }
cudaDeviceProp prop{};
- if (!cuda_ok(cudaGetDeviceProperties(&prop, device), "device properties")) { g_nctx = 0; return 0; }
+ if (!cuda_ok(cudaGetDeviceProperties(&prop, device), "device properties")) { coli_cuda_shutdown(); return 0; }
ctx->compute_major=prop.major;ctx->compute_minor=prop.minor;
if(!cuda_ok(cudaStreamCreateWithFlags(&ctx->stream,cudaStreamNonBlocking),"stream creation")){
- g_nctx=0;return 0;
+ coli_cuda_shutdown();return 0;
}
#ifdef COLI_ANS
if(std::getenv("CUDA_RAW_EXPERTS")){
@@ -1288,6 +1353,10 @@ extern "C" void coli_cuda_shutdown(void) {
for (int i = 0; i < g_nctx; i++) {
DeviceContext *ctx = &g_ctx[i];
if (!select_ctx(ctx)) continue;
+ if (ctx->mxfp4_weights) cudaFree(ctx->mxfp4_weights);
+ if (ctx->mxfp4_scales) cudaFree(ctx->mxfp4_scales);
+ if (ctx->mxfp4_expert_weights) cudaFree(ctx->mxfp4_expert_weights);
+ if (ctx->mxfp4_expert_scales) cudaFree(ctx->mxfp4_expert_scales);
if (ctx->x) cudaFree(ctx->x);
if (ctx->y) cudaFree(ctx->y);
if (ctx->dx) cudaFree(ctx->dx);
@@ -1312,6 +1381,10 @@ extern "C" void coli_cuda_shutdown(void) {
ctx->ans_scratch=nullptr;ctx->ans_chunks=nullptr;ctx->ans_raw=nullptr;ctx->ans_raw_cap=0;
ctx->ans_host=nullptr;ctx->ans_host_cap=0;ctx->ans_copy_pending=0;
#endif
+ ctx->mxfp4_weights = ctx->mxfp4_scales = nullptr;
+ ctx->mxfp4_weights_cap = ctx->mxfp4_scales_cap = 0;
+ ctx->mxfp4_expert_weights = ctx->mxfp4_expert_scales = nullptr;
+ ctx->mxfp4_expert_weights_cap = ctx->mxfp4_expert_scales_cap = 0;
ctx->x = ctx->y = ctx->gate = ctx->up = nullptr;
ctx->dx = ctx->dy = nullptr; ctx->dx_cap = ctx->dy_cap = 0;
ctx->qx=nullptr; ctx->qscale=nullptr;
@@ -1324,6 +1397,22 @@ extern "C" void coli_cuda_shutdown(void) {
ctx->group_desc=nullptr; ctx->group_desc_cap=0;
}
g_nctx = 0;
+ /* g_fp8_lut_ready is PROCESS-WIDE while the e4m3 table (c_e4m3, a
+ * __constant__ device symbol whose lifetime is the CUDA primary context,
+ * not this file's host-side DeviceContext structs) is PER-DEVICE. A later
+ * coli_cuda_init may select a device the previous span never published to;
+ * without this reset the upload gate (g_fp8_lut_ready, checked in
+ * coli_cuda_tensor_upload) would still be satisfied from the PREVIOUS
+ * boot and admit fmt=8 tensors whose kernels there decode against an
+ * unwritten (zero) table: silent all-zero weights, the exact
+ * fabricated-numbers failure mode the format gates exist to refuse.
+ * Reset so every boot must publish its own LUT (coli_cuda_fp8_set_lut)
+ * before any fmt=8 upload. Shutdown is the ONLY site that needs to clear
+ * the flag: coli_cuda_init refuses a re-init that names a different
+ * device set (it returns early while g_nctx is non-zero, leaving the
+ * existing contexts and their published table untouched), so the device
+ * set can only WIDEN by passing through here first. */
+ g_fp8_lut_ready = 0;
#ifdef COLI_ANS
if(g_ans_sidecar){std::fclose(g_ans_sidecar);g_ans_sidecar=nullptr;}
#if defined(__linux__)
@@ -1412,16 +1501,22 @@ extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor,
/* fmt=6 keeps its scales inside each 98-byte block, so it is the one
* quantized format that legitimately arrives with scales == NULL. */
if (!rb || (fmt && fmt != 6 && !scales)) return 0;
- if (fmt == 8 && !g_fp8_lut_ready) return 0; /* kernels would read a zero LUT */
+ /* kernels would read a zero LUT; shared predicate, pinned by
+ * tests/test_cuda_lut_gate.c without a CUDA toolchain */
+ if (!coli_cuda_fp8_gate_admits(fmt, g_fp8_lut_ready)) return 0;
ColiCudaTensor *t = static_cast(std::calloc(1, sizeof(*t)));
if (!t) return 0;
t->fmt = fmt; t->I = I; t->O = O; t->device = device; t->weight_bytes = rb * (size_t)O;
t->gs = (fmt==4 && g_upload_gs>0) ? g_upload_gs : 0;
t->ng = t->gs ? (I + t->gs - 1) / t->gs : 1;
t->scale_count = t->gs ? (size_t)O * (size_t)t->ng : (size_t)O;
- if (fmt == 8) { /* per-128x128-block scales: [ceil(O/128), ceil(I/128)] */
- t->ng = (I + 127) / 128;
- t->scale_count = (size_t)((O + 127) / 128) * (size_t)t->ng;
+ if (fmt == 7) {
+ t->ng = (I + 31) / 32;
+ t->scale_count = (size_t)O * t->ng;
+ }
+ if (fmt == 8) { /* per-block scales: [ceil(O/FP8_BLOCK), ceil(I/FP8_BLOCK)] (fp8_format.h) */
+ t->ng = (int)fp8_nblk(I);
+ t->scale_count = (size_t)fp8_nblk(O) * (size_t)t->ng;
}
if (!cuda_ok(cudaMalloc(&t->weights, t->weight_bytes), "tensor allocation")) {
coli_cuda_tensor_free(t);
@@ -1439,8 +1534,8 @@ extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor,
offset_to_signed_s4<<<(unsigned)((t->weight_bytes+255)/256),256>>>((uint8_t*)t->weights,t->weight_bytes);
if(!cuda_ok(cudaGetLastError(),"int4 weight conversion")){coli_cuda_tensor_free(t);return 0;}}
if (fmt && fmt != 6) {
- if (!cuda_ok(cudaMalloc(&t->scales, t->scale_count * sizeof(float)), "scale allocation") ||
- !cuda_ok(cudaMemcpy(t->scales, scales, t->scale_count * sizeof(float), cudaMemcpyHostToDevice), "scale upload")) {
+ if (!cuda_ok(cudaMalloc(&t->scales, tensor_scale_bytes(t)), "scale allocation") ||
+ !cuda_ok(cudaMemcpy(t->scales, scales, tensor_scale_bytes(t), cudaMemcpyHostToDevice), "scale upload")) {
coli_cuda_tensor_free(t);
return 0;
}
@@ -1448,7 +1543,7 @@ extern "C" int coli_cuda_tensor_upload(ColiCudaTensor **tensor,
if (fmt == 6) t->scale_count = 0; /* in-block scales: nothing separate to track */
t->tracked = 1;
ctx->tensor_count++;
- ctx->tensor_bytes += t->weight_bytes + ((fmt && fmt != 6) ? t->scale_count * sizeof(float) : 0);
+ ctx->tensor_bytes += t->weight_bytes + tensor_scale_bytes(t);
*tensor = t;
return 1;
}
@@ -1634,10 +1729,9 @@ extern "C" int coli_cuda_tensor_update(ColiCudaTensor *tensor,
(uint8_t*)tensor->weights,tensor->weight_bytes);
if(!cuda_ok(cudaGetLastError(),"int4 weight refresh")) return 0;
}
- /* fmt=6 has no scale buffer at all (scales live in-block, scale_count 0), and
- * the fallback below would otherwise copy O floats out of a NULL host pointer. */
+ /* fmt=6 stores scales in-block; fmt=7 stores byte exponents separately. */
return !tensor->fmt || tensor->fmt==6 || cuda_ok(cudaMemcpy(tensor->scales,scales,
- (tensor->scale_count?tensor->scale_count:(size_t)tensor->O)*sizeof(float),
+ tensor_scale_bytes(tensor),
cudaMemcpyHostToDevice),"scale refresh");
}
@@ -1728,9 +1822,10 @@ extern "C" int coli_cuda_matmul_mxfp4(float *y, const float *x,
size_t wb = (size_t)O * rb, sb = (size_t)O * ng;
size_t xb = (size_t)S * I * sizeof(float), yb = (size_t)S * O * sizeof(float);
- uint8_t *dw = nullptr, *ds = nullptr;
- if (!cuda_ok(cudaMalloc(&dw, wb), "mxfp4 weight alloc")) return 0;
- if (!cuda_ok(cudaMalloc(&ds, sb), "mxfp4 scale alloc")) { cudaFree(dw); return 0; }
+ if (!reserve_bytes(&ctx->mxfp4_weights, &ctx->mxfp4_weights_cap, wb) ||
+ !reserve_bytes(&ctx->mxfp4_scales, &ctx->mxfp4_scales_cap, sb)) return 0;
+ uint8_t *dw = static_cast(ctx->mxfp4_weights);
+ uint8_t *ds = static_cast(ctx->mxfp4_scales);
int ok = reserve(&ctx->x, &ctx->x_cap, xb) && reserve(&ctx->y, &ctx->y_cap, yb) &&
cuda_ok(cudaMemcpy(dw, q4, wb, cudaMemcpyHostToDevice), "mxfp4 weight upload") &&
@@ -1743,8 +1838,52 @@ extern "C" int coli_cuda_matmul_mxfp4(float *y, const float *x,
ok = cuda_ok(cudaGetLastError(), "mxfp4 launch") &&
cuda_ok(cudaMemcpy(y, ctx->y, yb, cudaMemcpyDeviceToHost), "mxfp4 output download");
}
- cudaFree(dw);
- cudaFree(ds);
+ return ok;
+}
+
+/* Reuse one weight/scale staging allocation across the three projections.
+ * Default-stream copies are ordered after the previous projection's reads. */
+static int mxfp4_project(float *y, const float *x, uint8_t *dw, uint8_t *ds,
+ const uint8_t *w, const uint8_t *sc, int S, int I, int O) {
+ size_t rb = ((size_t)I + 1) / 2, ng = ((size_t)I + 31) / 32;
+ if (!cuda_ok(cudaMemcpy(dw, w, (size_t)O * rb, cudaMemcpyHostToDevice), "expert weight upload") ||
+ !cuda_ok(cudaMemcpy(ds, sc, (size_t)O * ng, cudaMemcpyHostToDevice), "expert scale upload")) return 0;
+ quant_matmul<<>>(y, x, dw, reinterpret_cast(ds),
+ 7, S, I, O, rb, 32, (int)ng);
+ return cuda_ok(cudaGetLastError(), "MXFP4 expert projection");
+}
+
+extern "C" int coli_cuda_expert_mxfp4(float *y, const float *x,
+ const unsigned char *gate_w, const unsigned char *gate_s,
+ const unsigned char *up_w, const unsigned char *up_s,
+ const unsigned char *down_w, const unsigned char *down_s,
+ int S, int D, int I, float b1, float b2) {
+ if (fault_injected() || !x || !y || !gate_w || !gate_s || !up_w || !up_s ||
+ !down_w || !down_s || S < 1 || S > 65535 || D < 1 || I < 1 ||
+ !(b1 > 0.f) || !(b2 > 0.f) || !std::isfinite(b1) || !std::isfinite(b2)) return 0;
+ DeviceContext *ctx = find_ctx(0);
+ if (!select_ctx(ctx)) return 0;
+ size_t xb = (size_t)S * D * sizeof(float), ib = (size_t)S * I * sizeof(float);
+ if (!reserve(&ctx->x, &ctx->x_cap, xb) || !reserve(&ctx->y, &ctx->y_cap, xb) ||
+ !reserve(&ctx->gate, &ctx->gate_cap, ib) || !reserve(&ctx->up, &ctx->up_cap, ib)) return 0;
+ size_t gw = (size_t)I * (((size_t)D + 1) / 2), dwb = (size_t)D * (((size_t)I + 1) / 2);
+ size_t gs = (size_t)I * (((size_t)D + 31) / 32), dsb = (size_t)D * (((size_t)I + 31) / 32);
+ /* Grow to the largest projection seen, then reuse across routed experts.
+ * Slot identity is irrelevant: every call refreshes all weight bytes. */
+ if (!reserve_bytes(&ctx->mxfp4_expert_weights, &ctx->mxfp4_expert_weights_cap, gw > dwb ? gw : dwb) ||
+ !reserve_bytes(&ctx->mxfp4_expert_scales, &ctx->mxfp4_expert_scales_cap, gs > dsb ? gs : dsb)) return 0;
+ uint8_t *dw = static_cast(ctx->mxfp4_expert_weights);
+ uint8_t *ds = static_cast(ctx->mxfp4_expert_scales);
+ int ok = cuda_ok(cudaMemcpy(ctx->x, x, xb, cudaMemcpyHostToDevice), "expert input upload") &&
+ mxfp4_project(ctx->gate, ctx->x, dw, ds, gate_w, gate_s, S, D, I) &&
+ mxfp4_project(ctx->up, ctx->x, dw, ds, up_w, up_s, S, D, I);
+ if (ok) {
+ size_t n = (size_t)S * I;
+ situ_mul<<<(unsigned)((n + 255) / 256), 256>>>(ctx->gate, ctx->up, n, b1, b2);
+ ok = cuda_ok(cudaGetLastError(), "SiTU-GLU launch") &&
+ mxfp4_project(ctx->y, ctx->gate, dw, ds, down_w, down_s, S, I, D) &&
+ cuda_ok(cudaMemcpy(y, ctx->y, xb, cudaMemcpyDeviceToHost), "expert output download");
+ }
return ok;
}
@@ -2198,18 +2337,20 @@ extern "C" const float *coli_cuda_expert_group_take(int device) {
/* The absorb kernels decode `w` through weight_at + absorb_scale, which know
- * per-row and fmt=4 group scales only. Refuse anything else (fmt=5/6/8) rather
- * than mis-decode it — the caller keeps its CPU attention path. (`proj`
- * tensors are exempt: they run through quant_matmul, which dispatches every
- * format it uploads.) A dedicated block-scale absorb for fmt=8 is follow-up
- * work, same shape as routing fmt=4 through the grouped kernels was.
+ * per-row scales, fmt=4 group scales, and fmt=8 per-128x128-block scales.
+ * Refuse anything else (fmt=5/6/7) rather than mis-decode it — the caller
+ * keeps its CPU attention path. (`proj` tensors are exempt: they run through
+ * quant_matmul, which dispatches every format it uploads.) fmt=8 support
+ * funnels through this one predicate for all the absorb host wrappers below,
+ * so none of them needed a separate change.
*
* The admissible set is weight_at's own, taken from the shared predicate rather
- * than restated as `fmt <= 4`: this gate and weight_at's device-side backstop
- * must not be able to drift apart, and the old inequality also admitted
- * NEGATIVE fmt values, which weight_at would then have fallen through on. Same
- * truth table for every fmt a container can actually carry (0..8), so no
- * existing container changes behaviour here. */
+ * than restated as an inequality: this gate and weight_at's device-side
+ * backstop must not be able to drift apart, and the old `fmt <= 4` also
+ * admitted NEGATIVE fmt values, which weight_at would then have fallen through
+ * on. A fmt=8 tensor implies a live e4m3 LUT (upload refuses it otherwise --
+ * see the predicate's caveat note in backend_cuda.h), so no extra gate is
+ * needed here. */
static int absorb_fmt_ok(const ColiCudaTensor *w){
return w && coli_cuda_weight_at_supported(w->fmt);
}
@@ -2386,19 +2527,14 @@ extern "C" void coli_cuda_tensor_free(ColiCudaTensor *tensor) {
DeviceContext *ctx = find_ctx(tensor->device);
if (ctx) select_ctx(ctx);
if (tensor->tracked && ctx) {
- /* Must mirror the upload's accounting exactly -- literally the same
- * expression upload uses to charge (scale_count * sizeof(float), gated
- * on fmt=6 never having a separate scale buffer), so the two can no
- * longer drift independently. Over-subtracting here trips the >= guard
- * below, which silently leaves the tensor's bytes on the device counter
- * forever. */
+ /* Charge and release the same format-specific scale storage. */
size_t storage_bytes =
#ifdef COLI_ANS
tensor->compressed ? tensor->archive_bytes :
#endif
tensor->weight_bytes;
size_t bytes = storage_bytes +
- ((tensor->fmt && tensor->fmt != 6) ? tensor->scale_count * sizeof(float) : 0);
+ tensor_scale_bytes(tensor);
if (ctx->tensor_count) ctx->tensor_count--;
if (ctx->tensor_bytes >= bytes) ctx->tensor_bytes -= bytes;
}
@@ -2410,19 +2546,14 @@ extern "C" void coli_cuda_tensor_free(ColiCudaTensor *tensor) {
extern "C" size_t coli_cuda_tensor_bytes(const ColiCudaTensor *tensor) {
if (!tensor) return 0;
- /* Must mirror upload's and free's accounting exactly -- literally the same
- * expression they use (scale_count * sizeof(float), gated on fmt=6 never
- * having a separate scale buffer) -- so all three can no longer drift
- * independently. The prior `O * ng` shape over-reported for fmt=8 (real
- * footprint is (O+127)/128 * ng block scales, not O * ng) and for fmt=6
- * (which has no separate scale buffer at all). */
+ /* Logical size uses the same scale layout as upload and free. */
size_t storage_bytes =
#ifdef COLI_ANS
tensor->compressed ? tensor->archive_bytes :
#endif
tensor->weight_bytes;
return storage_bytes +
- ((tensor->fmt && tensor->fmt != 6) ? tensor->scale_count * sizeof(float) : 0);
+ tensor_scale_bytes(tensor);
}
/* What a cudaMalloc of `bytes` actually takes off the card.
@@ -2536,7 +2667,7 @@ extern "C" size_t coli_cuda_tensor_vram(const ColiCudaTensor *tensor) {
tensor->weight_bytes;
size_t total = coli_cuda_alloc_footprint(storage_bytes);
if (tensor->fmt && tensor->fmt != 6)
- total += coli_cuda_alloc_footprint(tensor->scale_count * sizeof(float));
+ total += coli_cuda_alloc_footprint(tensor_scale_bytes(tensor));
return total;
}
diff --git a/c/backend_cuda.h b/c/backend_cuda.h
index 953e8b39f..e58dbf95f 100644
--- a/c/backend_cuda.h
+++ b/c/backend_cuda.h
@@ -23,7 +23,9 @@ extern "C" {
/* Weight formats the generic per-element device decoder (weight_at,
* backend_cuda.cu) can actually decode: f32, int8-row, int4 nibbles (fmt=2 and
- * the grouped fmt=4, same packing), and int2. Nothing else.
+ * the grouped fmt=4, same packing), int2, and fmt=8 (fp8-e4m3 raw bytes,
+ * decoded through the c_e4m3 LUT -- absorb-path support; absorb_scale supplies
+ * its per-128x128-block scale). Nothing else.
*
* WHY THIS IS A PREDICATE AND NOT A COMMENT. weight_at used to END in the int2
* decode as an unguarded fall-through, so ANY other format handed to it -- a
@@ -42,18 +44,69 @@ extern "C" {
* same arrangement colibri.c uses for metal_fused_fmt_ok.
*
* NOT a statement about which formats the CUDA BACKEND supports: quant_matmul
- * has its own explicit branches for fmt=6 (E8/IQ3), fmt=7 (MXFP4) and fmt=8
- * (fp8-e4m3) that never route through weight_at. This predicate is scoped to
- * weight_at's own dispatch, which is what the absorb and grouped-expert kernels
- * decode through. */
+ * has its own explicit branches for fmt=6 (E8/IQ3) and fmt=7 (MXFP4) that
+ * never route through weight_at (and its own fmt=8 branch for the dense path
+ * -- weight_at's fmt=8 branch serves the absorb kernels, which share the same
+ * c_e4m3 LUT). This predicate is scoped to weight_at's own dispatch, which is
+ * what the absorb and grouped-expert kernels decode through.
+ *
+ * fmt=8 CAVEAT, stated because the truth table alone cannot carry it: a fmt=8
+ * decode additionally requires the e4m3 LUT to have been published to the
+ * configured devices (coli_cuda_fp8_set_lut). The exact mechanism, so the
+ * claim cannot outrun it: coli_cuda_fp8_set_lut copies the table into every
+ * context live AT CALL TIME and sets a process-wide flag; the flag gates
+ * fmt=8 uploads (coli_cuda_tensor_upload refuses until it is set).
+ * coli_cuda_shutdown clears the flag; coli_cuda_init never writes it. That is
+ * enough because init will not rebuild contexts underneath a live set: a
+ * re-init naming the same device set returns success and leaves the contexts,
+ * and the table published to them, untouched, while one naming a different
+ * device set is refused before any context is rebuilt. So the device set
+ * cannot widen past what the last publish covered without going through
+ * shutdown, and no fmt=8 ColiCudaTensor can reach a kernel whose device has
+ * an unwritten table. This predicate
+ * deliberately does not restate that gate: it answers "does weight_at have a
+ * decode branch for this fmt", which is the question the launch-site gates
+ * and the device-side __trap() backstop share. */
static inline int coli_cuda_weight_at_supported(int fmt) {
- return fmt == 0 || fmt == 1 || fmt == 2 || fmt == 3 || fmt == 4;
+ return fmt == 0 || fmt == 1 || fmt == 2 || fmt == 3 || fmt == 4 || fmt == 8;
+}
+
+/* The two decisions the fmt=8 LUT gate rests on, as pure predicates. They live
+ * here rather than inline in backend_cuda.cu so a host-side test can pin them
+ * with no CUDA toolchain and no GPU (tests/test_cuda_lut_gate.c). backend_cuda.cu
+ * calls BOTH at the real decision sites, so the test pins the engine's own
+ * logic rather than a second copy that could drift from it -- which is the
+ * failure this factoring exists to prevent, the gate having no CI reach
+ * otherwise. */
+
+/* Does the upload gate admit this tensor? Only fmt=8 needs the published
+ * table; every other format decodes without one. */
+static inline int coli_cuda_fp8_gate_admits(int fmt, int lut_ready) {
+ return fmt != 8 || lut_ready != 0;
+}
+
+/* What coli_cuda_init must do with a request while a device set may be live.
+ * BUILD: nothing is live, build the contexts. ACCEPT: the same set is already
+ * live -- return success and touch nothing, so the table published to those
+ * contexts stays valid. REFUSE: a different set is live -- refuse before
+ * rebuilding anything, so the set cannot widen past the last publish. */
+enum { COLI_CUDA_INIT_BUILD = 0, COLI_CUDA_INIT_ACCEPT = 1, COLI_CUDA_INIT_REFUSE = -1 };
+static inline int coli_cuda_init_disposition(int nctx, int count,
+ const int *want, const int *live) {
+ int i;
+ if (nctx <= 0) return COLI_CUDA_INIT_BUILD;
+ if (count != nctx) return COLI_CUDA_INIT_REFUSE;
+ for (i = 0; i < count; i++) if (want[i] != live[i]) return COLI_CUDA_INIT_REFUSE;
+ return COLI_CUDA_INIT_ACCEPT;
}
/* Opaque, persistent device copy of one resident quantized tensor. */
typedef struct ColiCudaTensor ColiCudaTensor;
-/* Devices are CUDA ordinals, not positions in the input list. */
+/* Devices are CUDA ordinals, not positions in the input list.
+ * Repeating the same ordered list preserves active contexts. Changing an
+ * active list returns 0 without replacing it; release tensors and shut down
+ * before selecting a different list. Init/shutdown require caller serialization. */
COLI_CUDA_DLLEXPORT int coli_cuda_init(const int *devices, int count);
COLI_CUDA_DLLEXPORT void coli_cuda_shutdown(void);
/* Number of CUDA devices visible to this process, before a device list is
@@ -115,6 +168,15 @@ COLI_CUDA_DLLEXPORT int coli_cuda_matmul_mxfp4(float *y, const float *x,
const unsigned char *e8s,
int S, int I, int O);
+/* Streaming Kimi expert: down(SiTU(gate(x), up(x))). Weights are MXFP4
+ * host buffers; intermediate activations remain on device. No weight cache.
+ * Returns 0 on failure; callers must accumulate y only after success. */
+COLI_CUDA_DLLEXPORT int coli_cuda_expert_mxfp4(float *y, const float *x,
+ const unsigned char *gate_w, const unsigned char *gate_s,
+ const unsigned char *up_w, const unsigned char *up_s,
+ const unsigned char *down_w, const unsigned char *down_s,
+ int S, int D, int I, float b1, float b2);
+
COLI_CUDA_DLLEXPORT int coli_cuda_matmul(ColiCudaTensor **tensor,
float *y, const float *x,
const void *weights, const float *scales,
diff --git a/c/backend_gpu_compat.h b/c/backend_gpu_compat.h
index dbbd07ff9..f3a3508f3 100644
--- a/c/backend_gpu_compat.h
+++ b/c/backend_gpu_compat.h
@@ -60,6 +60,7 @@ namespace nvcuda { namespace wmma = ::rocwmma; }
#endif
#define cudaError_t hipError_t
#define cudaSuccess hipSuccess
+#define cudaErrorMemoryAllocation hipErrorOutOfMemory
#define cudaGetErrorString hipGetErrorString
#define cudaGetLastError hipGetLastError
#define cudaSetDevice hipSetDevice
@@ -67,6 +68,7 @@ namespace nvcuda { namespace wmma = ::rocwmma; }
#define cudaDeviceProp hipDeviceProp_t
#define cudaGetDeviceProperties hipGetDeviceProperties
#define cudaMalloc hipMalloc
+#define cudaMallocManaged hipMallocManaged
#define cudaFree hipFree
#define cudaMemcpy hipMemcpy
#define cudaMemcpy2D hipMemcpy2D
diff --git a/c/backend_loader.c b/c/backend_loader.c
index d877553a8..a3d4db200 100644
--- a/c/backend_loader.c
+++ b/c/backend_loader.c
@@ -102,6 +102,11 @@ typedef int (*fn_matmul)(ColiCudaTensor **tensor, float *y, const flo
int fmt, int S, int I, int O, int device, int gs);
typedef int (*fn_matmul_mxfp4)(float *y, const float *x, const unsigned char *q4,
const unsigned char *e8s, int S, int I, int O);
+typedef int (*fn_expert_mxfp4)(float *y, const float *x,
+ const unsigned char *gate_w, const unsigned char *gate_s,
+ const unsigned char *up_w, const unsigned char *up_s,
+ const unsigned char *down_w, const unsigned char *down_s,
+ int S, int D, int I, float b1, float b2);
typedef void (*fn_tensor_free)(ColiCudaTensor *tensor);
typedef size_t (*fn_tensor_bytes)(const ColiCudaTensor *tensor);
typedef size_t (*fn_tensor_vram)(const ColiCudaTensor *tensor);
@@ -179,6 +184,7 @@ static struct {
fn_fp8_set_lut fp8_set_lut;
fn_matmul matmul;
fn_matmul_mxfp4 matmul_mxfp4;
+ fn_expert_mxfp4 expert_mxfp4;
fn_tensor_free tensor_free;
fn_tensor_bytes tensor_bytes;
fn_tensor_vram tensor_vram;
@@ -1433,6 +1439,7 @@ static int coli_cuda_load(void){
* nothing by this name, and the wrapper's 0 is the engine's own "fall back
* to CPU" result, so an older DLL still serves GLM and Qwen3.6 (#1405). */
RESOLVE_OPT(matmul_mxfp4, fn_matmul_mxfp4)
+ RESOLVE_OPT(expert_mxfp4, fn_expert_mxfp4)
RESOLVE_OPT(available_device_count, fn_available_device_count) /* qwen36 tier (#1533); older DLLs fall back to device_count */
RESOLVE(tensor_free, fn_tensor_free)
RESOLVE(tensor_bytes, fn_tensor_bytes)
@@ -1656,6 +1663,15 @@ int coli_cuda_matmul_mxfp4(float *y, const float *x, const unsigned char *q4,
return g_cuda.matmul_mxfp4(y, x, q4, e8s, S, I, O);
}
+int coli_cuda_expert_mxfp4(float *y, const float *x,
+ const unsigned char *gate_w, const unsigned char *gate_s,
+ const unsigned char *up_w, const unsigned char *up_s,
+ const unsigned char *down_w, const unsigned char *down_s,
+ int S, int D, int I, float b1, float b2) {
+ if (!g_cuda.available || !g_cuda.expert_mxfp4) return 0;
+ return g_cuda.expert_mxfp4(y, x, gate_w, gate_s, up_w, up_s, down_w, down_s, S, D, I, b1, b2);
+}
+
void coli_cuda_tensor_free(ColiCudaTensor *tensor){
if(g_cuda.available && g_cuda.tensor_free) g_cuda.tensor_free(tensor);
}
diff --git a/c/backend_metal.mm b/c/backend_metal.mm
index e3497320c..3b767e2b3 100644
--- a/c/backend_metal.mm
+++ b/c/backend_metal.mm
@@ -767,7 +767,7 @@ static size_t fmt_bytes(int fmt, int I, int O) {
// Grouped-int4 (fmt=4) scale-array size: one f32 per gsz-element group, per row -> O*ceil(I/gsz).
// fp8 (fmt=8) scale-array size: one f32 per 128x128 BLOCK -> ceil(O/128)*ceil(I/128) (2D,
// not per-row -- quant.h isn't included here, so the ceil-div is inlined rather than sharing
-// colibri.c's qt_scale_bytes/quant.h's fp8_nblk). The block is a fixed 128x128, so gs is
+// colibri.c's qt_scale_bytes/fp8_format.h's fp8_nblk). The block is a fixed 128x128, so gs is
// ignored for fmt==8. f32 is this build's implemented scale
// ENCODING for fmt=8 (see quant.h/colibri.c) -- this file has no reason to know that a
// UE8M0 encoding exists at all: qt_resolve_fmt refuses it on the CPU read path before any
diff --git a/c/coli b/c/coli
index bdc1072e2..3f081a627 100755
--- a/c/coli
+++ b/c/coli
@@ -32,9 +32,9 @@ import os, sys, subprocess, argparse, json, time, signal, shutil, threading, re,
# and the console provides its own editing; on FreeBSD Python links libedit,
# which this activates the same way.
try:
- import readline # noqa: F401 — importing it is the activation
+ import readline
except ImportError:
- pass
+ readline = None
# The engine mmaps every shard (144+ files); macOS default RLIMIT_NOFILE is 256.
if sys.platform != "win32":
@@ -53,7 +53,10 @@ if sys.platform == "win32":
try: s.reconfigure(encoding="utf-8")
except (AttributeError, OSError): pass
-HERE = os.path.dirname(os.path.abspath(__file__))
+# realpath, not abspath: on a merged-/usr system /bin and /sbin are symlinks
+# to /usr/bin, and an installed launcher invoked as /bin/coli would otherwise
+# derive /libexec/colibri instead of /usr/libexec/colibri (#1689).
+HERE = os.path.dirname(os.path.realpath(__file__))
sys.path.insert(0, HERE)
# version.py sits next to this script in a source checkout, but an installed
# layout puts the launcher in $(PREFIX)/bin while the support modules live in
@@ -477,9 +480,10 @@ def env_for_engine(a, arch, plan=None):
# sets it for GLM and openai_server.py sets it for the gateway, so `coli
# chat` and `coli serve` worked while `coli run` handed the sister engines
# an environment without it: OLMoE exited with "started without a model"
- # (#1501). Set it here, once, for all of them.
+ # (#1501, #1600). Set it here, once, for all of them. --model is the
+ # directory, same as env_for(): a leftover SNAP must not load another model.
if getattr(a, "model", None):
- env.setdefault("SNAP", os.path.abspath(a.model))
+ env["SNAP"] = os.path.abspath(a.model)
if arch == "olmoe":
env["CHAT"] = "1"
env["MAX_NEW"] = str(ngen_for(a, family=arch))
@@ -491,7 +495,7 @@ def env_for_engine(a, arch, plan=None):
# #855; before that `grep -c RAM_GB c/kimi_k3.c` returned 0, so `coli chat
# --ram 242` on Kimi K3 set an environment variable nobody looked at and the
# user's session ran itself out of memory with the flag apparently set.
- if arch in ("deepseek_v4", "kimi", "glm53", "olmoe"):
+ if arch in ("deepseek_v4", "deepseek_v41", "kimi", "glm53", "olmoe"):
if a.ram: env["RAM_GB"] = str(a.ram)
if arch == "glm53":
# I densi vanno a int4 di default: sono 9,7 B parametri su 321, e in
@@ -606,6 +610,14 @@ def env_for_engine(a, arch, plan=None):
gain=100.0*profile["gain"]
print(f" {C.dim}[TUNE] applied measured profile · +{gain:.1f}% "
f"calibration throughput{C.r}",file=sys.stderr)
+ # Kimi's CUDA path is gated on K3_CUDA, not COLI_CUDA. --gpu / --vram /
+ # auto-tier write COLI_CUDA=1 after proving the build; without this copy
+ # the flag was accepted and the experts still stayed on the CPU.
+ if arch == "kimi":
+ if env.get("COLI_CUDA") == "0":
+ env["K3_CUDA"] = "0"
+ elif env.get("COLI_CUDA") == "1" and "K3_CUDA" not in explicit_env:
+ env["K3_CUDA"] = "1"
return env
def dsv4_cuda_available(model=None):
@@ -672,15 +684,20 @@ def cuda_binary(engine=None):
for line in linked.stdout.splitlines())
except (OSError,subprocess.SubprocessError): return False
if sys.platform == "win32":
- # Windows CUDA_DLL=1 builds never link libcudart directly: glm.exe loads
- # coli_cuda.dll at runtime via LoadLibrary (backend_loader.c), so there's no
- # import-table entry for ldd/dumpbin to see. Detect the COLI_CUDA build via a
- # marker string baked into glm.c's #ifdef COLI_CUDA block instead, and require
- # coli_cuda.dll to actually sit next to glm.exe (else CUDA init fails at startup).
+ # Windows CUDA_DLL/HIP_DLL hosts never link the GPU runtime directly:
+ # the engine LoadLibrary's its backend (backend_loader.c). The host
+ # compiles exactly one basename -- coli_hip.dll or coli_cuda.dll --
+ # and that is the file that must sit next to it. The GLM/Qwen banner
+ # is not a GPU-build marker: Kimi K3 CUDA_DLL builds link the same
+ # loader without printing it, and a HIP host that only had coli_hip.dll
+ # used to be refused as CPU-only.
try:
- with open(engine,"rb") as f: built=b"[CUDA] mode: routed experts" in f.read()
+ with open(engine,"rb") as f: image=f.read()
except OSError: return False
- return built and os.path.exists(os.path.join(os.path.dirname(engine),"coli_cuda.dll"))
+ from doctor import windows_backend_dll
+ expected=windows_backend_dll(image)
+ if not expected: return False
+ return os.path.exists(os.path.join(os.path.dirname(engine),expected))
return False
def resource_request(a, env):
@@ -1191,7 +1208,7 @@ def cmd_info(a):
try: n_shards=len([x for x in os.listdir(a.model) if x.endswith('.safetensors')])
except OSError: n_shards=0
print(f" {C.yel}config.json is missing{C.r}: coli picks the engine from it, so nothing can run here yet.")
- print(f" Copy the checkpoint's config.json (with tokenizer.json and model.safetensors.index.json)")
+ print(" Copy the checkpoint's config.json (with tokenizer.json and model.safetensors.index.json)")
print(f" from the model repo next to the {n_shards} shard(s) found here, then run coli info again.")
try:
mi=open('/proc/meminfo').read()
@@ -1511,9 +1528,7 @@ def chat_commands_help():
def install_chat_completer():
"""TAB completa i comandi. Senza readline (Windows) si perde solo il TAB."""
- try:
- import readline
- except ImportError:
+ if readline is None:
return
names = [p + n for n in CHAT_COMMANDS for p in ("/", ":")]
def complete(text, state):
@@ -1624,7 +1639,10 @@ def chat_attached(a, base, model_id):
else:
print(f" {C.yel} /brio needs the options: /brio merge | request changes | close{C.r}\n")
continue
- brio_options=[o.strip() for o in spec.split("|") if o.strip()]
+ # dict.fromkeys deduplica tenendo l'ordine: il server rifiuta due
+ # opzioni uguali con un 400, e "merge | merge | close" e' un refuso
+ # facile da fare a mano, non una domanda a tre opzioni.
+ brio_options=list(dict.fromkeys(o.strip() for o in spec.split("|") if o.strip()))
if len(brio_options)<2:
brio_options=[]
print(f" {C.yel} at least two options are needed, separated by |{C.r}\n")
@@ -2252,9 +2270,21 @@ def cmd_stop(a):
print(f" nothing running — no serve on port {a.port}, no SERVE engines"); return
for pid,desc in targets.items(): print(f" {'would stop' if a.dry_run else 'stopping'} {pid}: {desc}")
if a.dry_run: return
- for pid in targets:
+ # Graceful first: only the launcher's Engine.close drains the engine
+ # (stdin EOF -> atexit -> HEAT_FILE save). SIGTERM the launchers and
+ # give them the drain window to exit on their own; engine pids left
+ # behind (orphaned, no launcher) get the old treatment after.
+ launchers = [pid for pid, d in targets.items() if d.startswith("coli serve")]
+ strays = [pid for pid, d in targets.items() if not d.startswith("coli serve")]
+ deadline = time.time() + 30.0
+ for pid in launchers:
try: os.kill(pid, signal.SIGTERM)
except OSError: pass
+ while time.time() < deadline and any(_pid_alive(p) for p in launchers):
+ time.sleep(0.25)
+ for pid in strays:
+ try: os.kill(pid, signal.SIGTERM) # normally already gone with the launcher
+ except OSError: pass
time.sleep(2.0)
for pid in targets:
# signal.SIGKILL does not exist on win32 and AttributeError is not OSError,
diff --git a/c/colibri.c b/c/colibri.c
index 348d4e1af..678a30f23 100644
--- a/c/colibri.c
+++ b/c/colibri.c
@@ -60,6 +60,7 @@
#include /* hwinfo_emit: CPU brand string senza /proc */
#endif
#include "cli_args.h"
+#include "oracle.h"
#include "st.h"
#ifdef __linux__
#include "uring.h"
@@ -285,9 +286,10 @@ static int64_t qt_bytes(const QT *t){ /* byte residenti del tensore */
return (int64_t)t->O*(((int64_t)t->I+255)/256)*98 + 4;
if(t->fmt==8){ /* fp8-e4m3 passthrough: O*I raw e4m3 bytes (n, byte-identical layout
* to fmt=1's weight bytes) + one f32 scale per 128x128 block
- * (FP8_BLOCK in quant.h, included below qt_bytes -- keep the
- * arithmetic literal here, same discipline as fmt=5's comment
- * above). Missing this branch would fall through to the fmt=2
+ * (FP8_BLOCK in fp8_format.h via quant.h, included below
+ * qt_bytes -- keep the arithmetic literal here, same
+ * discipline as fmt=5's comment above). Missing this branch would
+ * fall through to the fmt=2
* default below (packed-nibble formula, ~half the real weight
* bytes) and undercount a resident fp8 tensor's byte footprint --
* feeds AUTOPIN/RAM-budget math, so this branch is load-bearing
@@ -546,6 +548,19 @@ typedef struct {
* than in quant.h: that header is shared by standalone kernel tests and
* sibling engines, where translation-unit-local copies are unused and trip
* -Wunused-variable. */
+#include "exact_dot.h"
+/* COLI_EXACT_VERIFY=1 (opt-in, #689): during draft+verify forwards (g_spec_live) the CPU
+ * MLA-absorb attention core accumulates its score and context dots EXACTLY (integer products,
+ * one rounding per dot; exact_dot.h). No summation order, SIMD width or contraction flag can
+ * change those bits, so a verify row decides near-ties the same way on every host. Off by
+ * default: it is an integer path (~7x the float loop on the dot itself at -O3). The default paths
+ * are untouched. */
+static int g_exact_verify=-1;
+static int exact_verify_on(void){
+ if(g_exact_verify<0){ const char *e=getenv("COLI_EXACT_VERIFY"); g_exact_verify=(e&&atoi(e))?1:0;
+ if(g_exact_verify) fprintf(stderr,"[EXACT_VERIFY] draft+verify attention core on the exact (order-independent) dot (#689; COLI_EXACT_VERIFY=0 to disable)\n"); }
+ return g_exact_verify;
+}
static int g_idot=1;
#if defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD)
static int g_i4s=1;
@@ -902,7 +917,9 @@ static double edisk_s(void){ return atomic_load_explicit(&g_edisk_ns,memory_orde
* served that gets labeled cold overstates the cold class, the bucket this line exists to
* size). */
static uint32_t g_direct_heat_ticks=0;
+#ifndef COLIBRI_NO_MAIN
static int g_direct_heat_explicit=0; /* 1 if COLI_DISKCLASS_WINDOW was set (skip the auto-derive) */
+#endif
#define DC_COLD 0
#define DC_WARM 1
static _Atomic uint64_t g_dc_n[2]; /* [DC_COLD]/[DC_WARM]: loads classified */
@@ -1012,8 +1029,16 @@ static void matmul_i4_grouped_pair(float *yg, float *yu, const float *x,
const uint8_t *qu, const float *su,
int S, int I, int O, int gs){
int rb=(I+1)/2; int ng=(I+gs-1)/gs;
+ int o0=0;
+#if defined(__SSE4_1__) && !defined(__AVX2__)
+ if(!(gs&1)){
+ o0=O&~3;
+ if(o0) matmul_i4_grouped_pair_sse41_rows4(yg,yu,x,qg,sg,qu,su,S,I,O,gs,rb,ng,o0);
+ if(o0==O) return;
+ }
+#endif
#pragma omp parallel for schedule(static)
- for(int o=0;oq4,t->O,t->I); t->planar=1; return;
}
if(t->fmt!=2) return;
+#if defined(__AVX512F__)&&defined(__AVX512BW__)
+ /* fmt=2 stays a coppie qui: il gemello f32 planare replica l'ordine di
+ * accumulo AVX2, non quello del ramo dot_i4f_avx512 a 512 bit — il claim
+ * bit-identico della famiglia f32 vale solo dove i gemelli coincidono.
+ * EN: fmt=2 keeps the pair layout on AVX-512 builds; the f32 planar twin
+ * mirrors the AVX2 accumulation order, not the 512-bit f32 arm's. */
+ return;
+#endif
planarize_i4(t->q4,t->O,t->I); t->planar=1;
if(atomic_fetch_add_explicit(&g_planar_n,1,memory_order_relaxed)==0)
fprintf(stderr,"[K1] planar int4 layout active (PLANAR=0 disables)\n");
@@ -1301,6 +1339,13 @@ static int g_expert_budget=0; /* EXPERT_BUDGET=N -> cap distinct experts loaded
* (arXiv 2602.16052): top-32 of 64 capture 93% routing weight. */
static int64_t g_budget_dropped=0; /* total experts dropped by EXPERT_BUDGET across all layers */
static int64_t g_budget_rescued=0; /* experts re-kept because a position would have been left with zero */
+static int g_degrade_zero=0; /* DEGRADE_ZERO=1: zero-fill miss slots whose per-position gate weight
+ * is below DEGRADE_TAU instead of blocking on a demand-load.
+ * Opt-in only; changes output. Decode-only (S<=4 guard in moe()). */
+static float g_degrade_tau=0.03f; /* DEGRADE_TAU=: gate weight threshold (default 0.03).
+ * Issue #865: tau=0.03 zeroes 21.8% of slots for +2.9% perplexity. */
+static int64_t g_degrade_dropped=0; /* cumulative miss slots zeroed by DEGRADE_ZERO across all layers */
+static int64_t g_degrade_dropped_by_layer[512]; /* per-layer miss slots zeroed (for footer breakdown) */
/* CACHE_ROUTE (paper 2412.00099 max-rank): opt-in only. Keep true top-J always;
* fill remaining slots preferring pin∪LRU experts ranked within top-M (or mass ROUTE_P). */
static int g_cache_route=0;
@@ -1642,7 +1687,8 @@ static void rope_interleave(float *v, int pos, const Cfg *c){
* unverified mirrors (see qt_check_fmt threat model); an unbounded ftell->malloc
* gave a hostile file a load-time OOM or, on malloc failure, a NULL deref via
* b[got]=0. Cap the size, NULL-check the alloc, require a full read. Returns a
- * malloc'd NUL-terminated buffer, or NULL on any failure. Mirrors tok.h tk_read_file. */
+ * malloc'd NUL-terminated buffer, or NULL on any failure. Embedded NUL bytes are
+ * invalid JSON and must not hide an unchecked suffix. Mirrors tok.h tk_read_file. */
#define CFG_MAX_BYTES (256ll<<20) /* config/oracle JSON is KB-MB in practice */
static char* cfg_slurp(const char *path){
FILE *f=fopen(path,"rb"); if(!f) return NULL;
@@ -1650,7 +1696,7 @@ static char* cfg_slurp(const char *path){
if(n<0 || (long long)n>CFG_MAX_BYTES){ fclose(f); return NULL; }
char *b=malloc((size_t)n+1); if(!b){ fclose(f); return NULL; }
size_t got=fread(b,1,(size_t)n,f); fclose(f);
- if((long)got!=n){ free(b); return NULL; }
+ if((long)got!=n || memchr(b,'\0',got)){ free(b); return NULL; }
b[got]=0; return b;
}
static jval* cfg_root(const char *snap, char **arena){
@@ -2241,6 +2287,42 @@ static void qt_cuda_colocate(QT *dst,const QT *src){
}
static void layer_cuda_shard_kvb(Layer *l,int H,int Q,int V){
if(!g_cuda_enabled||!g_cuda_dense||g_cuda_ndev<2||l->kv_b.fmt==0)return;
+ /* SHARD FORMAT ALLOWLIST (explicit refusal; this was an ACCIDENTAL fail-safe): the
+ * rb/weights/scale arithmetic below is written for exactly fmt=1 (int8, per-row
+ * scale), fmt=2 (int4 per-row), fmt=3 (int2 per-row) and fmt=4 (int4 grouped).
+ * Any other fmt reaching it computes a wrong row-byte stride, takes l->kv_b.q4 as
+ * the weight pointer (NULL for fmt=8, whose raw e4m3 bytes live in q8 -- see the
+ * QT struct comment), and slices l->kv_b.s with per-row/per-group geometry that
+ * fmt=8's per-128x128-BLOCK scales (and fmt=6's single 4-byte tag) simply do not
+ * have. fmt=8 only ever "worked" here by accident: q4==NULL made
+ * coli_cuda_tensor_upload_g's !weights check reject the upload before anything
+ * dereferenced it -- silent, unnamed, and one refactor away from a misread.
+ * Refuse BY NAME instead, BEFORE any pointer/stride use, and say what happens
+ * instead: the un-sharded kv_b stays whole on its layer home device, where fmt=8
+ * kv_b decode runs the absorb path (qt_addrow/qt_matvec_rows' fmt=8 branches, or
+ * the CUDA absorb kernels via absorb_fmt_ok) -- COLI_CUDA_ATTN_SHARD is a no-op
+ * for it. Same "refuse rather than misread" discipline as qt_addrow/
+ * qt_matvec_rows' guards; notice only (no exit): sharding is an opt-in
+ * optimization and skipping it is the correct, working behavior. Bounded once
+ * per process per fmt, never per layer (metal_fmt_gate_notice, the precedent
+ * for bounded notices, is coarser still: one line per tensor KIND, naming only
+ * the first offending fmt). */
+ if(l->kv_b.fmt!=1&&l->kv_b.fmt!=2&&l->kv_b.fmt!=3&&l->kv_b.fmt!=4){
+ static int refused_fmt[32];
+ if(!refused_fmt[l->kv_b.fmt&31]){ refused_fmt[l->kv_b.fmt&31]=1;
+ if(l->kv_b.fmt==8)
+ fprintf(stderr,"layer_cuda_shard_kvb: kv_b fmt=8 (fp8-e4m3, per-128x128-block "
+ "scales) has no head-shard layout here -- refusing the shard; fmt=8 kv_b "
+ "runs the absorb path on the layer home device instead, so "
+ "COLI_CUDA_ATTN_SHARD is a no-op for it (applies to every layer)\n");
+ else
+ fprintf(stderr,"layer_cuda_shard_kvb: unsupported kv_b fmt=%d for the head-shard "
+ "upload (only fmt 1/2/3/4 match the per-row byte/scale strides computed "
+ "here) -- refusing the shard; kv_b stays whole on its layer home device "
+ "(applies to every layer)\n",l->kv_b.fmt);
+ }
+ return;
+ }
int rb=l->kv_b.fmt==1?l->kv_b.I:
(l->kv_b.fmt==2||l->kv_b.fmt==4)?(l->kv_b.I+1)/2:(l->kv_b.I+3)/4;
const uint8_t *weights=l->kv_b.fmt==1?(const uint8_t*)l->kv_b.q8:l->kv_b.q4;
@@ -3833,6 +3915,32 @@ static void expert_prefetch(Model *m, int layer, int eid){
/* ---- helper per l'ABSORPTION: accesso per-riga ai QT quantizzati ---- */
/* acc[0..I) += coef * W[row,:] (dequant al volo) */
+/* One fmt=8 block scale, checked before it multiplies anything.
+ *
+ * A NaN scale has no safe interpretation: it poisons the whole block and every
+ * accumulator downstream of it, and this function cannot repair it, so it is
+ * refused by name with the block that carried it -- the same "refuse rather
+ * than misread" discipline the format guards below apply.
+ *
+ * A ZERO scale is NOT refused: it is valid data. Block scales are amax/448, so
+ * a genuinely all-zero block (padding, an unused slice) legitimately produces
+ * zero, and decoding it as zeros is the correct answer. It cannot be confused
+ * with a decode against an unwritten table the way it can on the GPU, because
+ * there is no table here to be unwritten -- the CPU decoder reads its e4m3
+ * values from a compile-time constant table, which is why the LUT-ready gate
+ * exists only on the CUDA side. Do not re-add a zero refusal: it would reject
+ * valid checkpoints.
+ *
+ * The check is per BLOCK, not per element: one branch per FP8_BLOCK columns. */
+static float fp8_block_scale(float sc, int64_t blkO, int64_t bi, const char *who){
+ if(isnan(sc)){
+ fprintf(stderr,"%s: fmt=8 scale block [%lld,%lld] is NaN -- refusing rather than "
+ "propagate it through the absorb accumulator\n",who,(long long)blkO,(long long)bi);
+ exit(1);
+ }
+ return sc;
+}
+
static void qt_addrow(const QT *t, int row, float coef, float *acc){
int I=t->I;
if(t->fmt==0){ const float *w=t->qf+(int64_t)row*I; for(int i=0;i>2]>>((k&3)*2))&3)|(((hi[k>>3]>>(k&7))&1)<<2);
acc[base+k]+=cg*(float)((int)u-4); } }
return; }
+ /* fmt=8 (fp8-e4m3-b128, absorb-path support added here): t->s holds ONE f32 scale
+ * per 128x128 BLOCK (ceil(O/128)*ceil(I/128) entries, block-row-major), not O, and
+ * t->q4 is NULL for this format -- raw e4m3 bytes live in t->q8 instead, same
+ * convention as fmt=1 (see the QT struct comment). Mirrors matmul_fp8's (quant.h)
+ * block-scale indexing exactly: blkO=row/FP8_BLOCK selects the scale row, then one
+ * scale per FP8_BLOCK-wide slice of I. This is the branch that used to be missing
+ * -- see the guard below's history note. */
+ if(t->fmt==8){ const uint8_t *w=(const uint8_t*)t->q8+(int64_t)row*I;
+ int64_t nblkI=fp8_nblk(I), blkO=(int64_t)row/FP8_BLOCK;
+ const float *scl=t->s+blkO*nblkI;
+ for(int64_t bi=0; bi*FP8_BLOCKI) blen=I-base;
+ float sc=coef*fp8_block_scale(scl[bi],blkO,bi,"qt_addrow");
+ for(int i=base;is[row]) followed by fmt=1 (int8, explicit
* branch), fmt=2 (int4 packed, explicit branch), or the tail's own IMPLICIT fmt=3
* (int2 packed, the final unconditional block) -- there was no guard stopping any
@@ -3866,20 +3989,16 @@ static void qt_addrow(const QT *t, int row, float coef, float *acc){
* a heap OVERREAD, and the untouched fall-through then misreads t->q4's real E8
* lattice bytes as int2-packed data (same bug SHAPE as #298's CUDA absorb-kernel
* fix, and the same one this file's own fmt=4/5 branches above were added to
- * dodge -- fmt=6 was simply missed). fmt=8 (fp8-e4m3-b128): t->s holds
- * ceil(O/128)*ceil(I/128) per-block floats, not O -- t->s[row] overreads for
- * row>=nblk (e.g. a [130,130] tensor has nblk=4, so every row past 3 already reads
- * out of bounds), AND t->q4 is NULL for fmt=8 (raw bytes live in t->q8 instead,
- * same convention as fmt=1 -- see the QT struct comment), so the fall-through's
- * `t->q4+(int64_t)row*((I+3)/4)` dereferences NULL-plus-offset: SIGSEGV,
- * reproduced (see the report's proof-of-bite transcript). Refuse loudly instead --
- * this function has no byte-count context of its own to validate against (it only
- * ever sees an already-resolved QT), so "unsupported fmt" is the only check
- * available, same "refuse rather than misread" discipline qt_resolve_fmt applies
- * at load time. */
+ * dodge -- fmt=6 was simply missed). fmt=8 previously landed here too (t->s[row]
+ * overreads past row>=nblk, t->q4 is NULL -> SIGSEGV via the int2 fall-through);
+ * it now returns above via its own branch and never reaches this guard. Refuse
+ * loudly instead for anything else -- this function has no byte-count context of
+ * its own to validate against (it only ever sees an already-resolved QT), so
+ * "unsupported fmt" is the only check available, same "refuse rather than
+ * misread" discipline qt_resolve_fmt applies at load time. */
if(t->fmt!=1 && t->fmt!=2 && t->fmt!=3){
fprintf(stderr,"qt_addrow: unsupported fmt=%d for the per-row-scale absorb path "
- "(only fmt 1/2/3 reach this point; fmt 0/4/5 are handled above and return "
+ "(only fmt 1/2/3 reach this point; fmt 0/4/5/8 are handled above and return "
"before it) -- refusing rather than misread t->s[row]/t->q4\n", t->fmt);
exit(1);
}
@@ -3929,18 +4048,35 @@ static void qt_matvec_rows(const QT *t, int r0, int n, const float *x, float *y)
for(int k=0;k>2]>>((k&3)*2))&3)|(((hi[k>>3]>>(k&7))&1)<<2);
acc+=(float)((int)u-4)*x[base+k]; }
a+=(double)(acc*sr[g]); } }
+ /* fmt=8 (fp8-e4m3-b128, absorb-path support added here): per-128x128-BLOCK f32
+ * scale, block-row-major (ceil(O/128)*ceil(I/128) entries), raw bytes in t->q8
+ * (t->q4 is NULL for this format). Same block-scale indexing as matmul_fp8
+ * (quant.h) and qt_addrow's fmt=8 branch above: blkO=row/FP8_BLOCK picks the
+ * scale row, one scale per FP8_BLOCK-wide slice of I, double-accumulated
+ * across blocks like matmul_fp8 to avoid unfairly penalizing cross-block
+ * cancellation (widen-then-multiply, a+=(double)acc*sc -- the same rounding
+ * as matmul_fp8 and this function's grouped fmt=4 arm; fmt=5's arm rounds
+ * differently, multiplying in float before widening). */
+ else if(t->fmt==8){ const uint8_t *w=(const uint8_t*)t->q8+(int64_t)row*I;
+ int64_t nblkI=fp8_nblk(I), blkO=(int64_t)row/FP8_BLOCK;
+ const float *scl=t->s+blkO*nblkI;
+ for(int64_t bi=0; bi*FP8_BLOCKI) blen=I-base;
+ float sc=fp8_block_scale(scl[bi],blkO,bi,"qt_matvec_rows"); float acc=0;
+ for(int i=base;is is a fixed 4-byte tag (t->s[row] overreads for row>0),
- * fmt=8's t->s holds per-128x128-block floats (t->s[row] overreads for
- * row>=nblk) and t->q4 is NULL for fmt=8 -- both would have silently misread or
- * crashed here exactly like qt_addrow did before its own fix; refuse instead. */
+ * defect): fmt=6's t->s is a fixed 4-byte tag (t->s[row] overreads for row>0) --
+ * it would silently misread or crash here exactly like qt_addrow did before its
+ * fix; refuse instead. fmt=8 previously fell into this same trap and now has its
+ * own branch above instead. */
else if(t->fmt==3){ const uint8_t *w=t->q4+(int64_t)row*((I+3)/4); float s=t->s[row]; float acc=0;
for(int i=0;i>2]; acc+=((int)((b>>((i&3)*2))&3)-2)*x[i]; } a=acc*s; }
else {
fprintf(stderr,"qt_matvec_rows: unsupported fmt=%d for the per-row-scale absorb "
- "path (only fmt 0/1/2/3/4/5 are handled) -- refusing rather than misread "
+ "path (only fmt 0/1/2/3/4/5/8 are handled) -- refusing rather than misread "
"t->s[row]/t->q4\n", t->fmt);
exit(1);
}
@@ -3948,7 +4084,9 @@ static void qt_matvec_rows(const QT *t, int r0, int n, const float *x, float *y)
}
}
static int g_absorb=-1;
+#if defined(COLI_METAL) || !defined(COLIBRI_NO_MAIN)
static int g_metal_prefill=0; /* default 0: S>4 prefill attention stays on the CPU (bit-exact). COLI_METAL_PREFILL=1 opts it onto the GPU (~4x, near-tie divergence — see docs/metal.md, #622) */
+#endif
/* KV8=1: cache latente Lc/Rc in fp8 e4m3 + scala f32 per riga (~4x meno RAM del f32).
* CPU-only in this PR — sui percorsi CUDA/Metal che leggono righe f32 si spegne da
* solo (guardie !g_kv8), e forza COLI_CUDA_PIPE=0 (il pipe-prefill legge righe f32).
@@ -4776,6 +4914,13 @@ static void attention_rows(Model *m, Layer *l, int layer, float *x, int S, int p
} else {
const float *Lt=coli_kv_row(ks->Lc[layer],t,kvl);
const float *kr=coli_kv_row(ks->Rc[layer],t,c->qk_rope);
+ if(exact_verify_on()&&g_spec_live){
+ /* #689 exact verify: one exact accumulator over BOTH partial dots, rounded once */
+ exd_acc ea; exd_init(&ea);
+ for(int i=0;iqk_rope;d++) exd_add_ff(&ea,qr[d],kr[d]);
+ a=exd_finish(&ea);
+ } else {
/* MLA-absorb score: dot(qabs, Lt) + dot(qr, kr). #442: the qabs·Lt
* reduction is the hot f32 dot at this site (kvl=512 on GLM-5.2,
* runs nt times per (s,h), grows with context). SIMD-ify under
@@ -4798,10 +4943,21 @@ static void attention_rows(Model *m, Layer *l, int layer, float *x, int S, int p
for(;iqk_rope;d++) a+=qr[d]*kr[d];
}
+ }
sc[jj]=a*c->attn_scale;
}
softmax(sc,nt);
float clat[512]; memset(clat,0,kvl*sizeof(float));
+ if(exact_verify_on()&&g_spec_live&&!tq1&&!g_tq&&!g_kv8){
+ /* #689 exact verify: clat[i] = sum_t sc[t]*Lt[i] as an exact dot over t per column
+ * (transposed walk: cache-unfriendly, verify rows only). NOT taken on the quantised
+ * KV paths (tq1 / TQ / kv8): those keep the float context dot, so COLI_EXACT_VERIFY
+ * does not provide exactness for the context dot with a quantised cache (README). */
+ for(int i=0;iLc[layer],t,kvl)[i]); }
+ clat[i]=exd_finish(&ea); }
+ } else
for(int jj=0;jjpin[layer];
+ for(int z=0;znpin[layer];z++) if(P[z].eid==eid){ resident=1; break; }
+ if(!resident){ ESlot *Sl=m->ecache[layer]; int nn=m->ecn[layer];
+ for(int z=0;z= tau. Each position's weight is tested independently — this
+ * is the gate the +2.9% ppl measurement was taken under. */
+ for(int s=0;s=g_degrade_tau){
+ int e=idxs[(int64_t)s*K+kk];
+ for(int j=0;jbw){ bw=wv; be=idxs[(int64_t)s*K+kk]; }
+ }
+ if(be<0) be=idxs[(int64_t)s*K];
+ seen[be]=1;
+ for(int j=0;jdst[e->n++]=t; }
+typedef struct { EmitStore tokens; int vocab, finite; } OracleEmit;
+static void emit_oracle(int t, const float *lo, void *ud){
+ OracleEmit *e=(OracleEmit*)ud;
+ if(!oracle_logits_finite(lo,e->vocab)){
+ fprintf(stderr,"[ORACLE] non-finite logits at generated token %d\n",e->tokens.n);
+ e->finite=0;
+ }
+ emit_store(t,lo,&e->tokens);
+}
/* emit callback: detokenizza e stampa in streaming (chat/run), con heartbeat */
typedef struct { Tok *T; Model *m; double t0; int count; int quiet; } EmitStream;
static void emit_stream(int t, const float *lo, void *ud){
@@ -7827,8 +8065,9 @@ static void dump_top5_logits(int pos, const float *lo, int V, int expected, int
for(int k=0;k<5&&idx[k]>=0;k++) fprintf(stderr," %d:%.5f", idx[k], (double)val[k]);
fprintf(stderr,"\n");
}
-static void forward_all(Model *m, const int *ids, int S, int *pred, const int *ref){
+static int forward_all(Model *m, const int *ids, int S, int *pred, const int *ref){
Cfg *c=&m->c; int D=c->hidden;
+ int finite=1;
int dbg = ref && getenv("DEBUG_LOGITS");
kv_alloc(m,S);
float *x=falloc((int64_t)S*D);
@@ -7839,11 +8078,16 @@ static void forward_all(Model *m, const int *ids, int S, int *pred, const int *r
for(int s=0;sfinal_norm, D, c->eps); /* heap row (#183) */
matmul_qt(lo, row, &m->lm_head, 1);
+ if(!oracle_logits_finite(lo,c->vocab)){
+ fprintf(stderr,"[ORACLE] non-finite logits at teacher-forcing position %d\n",s);
+ finite=0; pred[s]=-1; continue;
+ }
int best=0; float bv=lo[0]; for(int i=1;ivocab;i++) if(lo[i]>bv){bv=lo[i];best=i;}
pred[s]=best;
if(dbg && pred[s]!=ref[s]) dump_top5_logits(s, lo, c->vocab, ref[s], pred[s]);
}
free(x); free(lo); free(row);
+ return finite;
}
/* log-prob (log-softmax) del token target dato il vettore di logit; *am=1 se e' l'argmax */
@@ -8007,12 +8251,14 @@ static void run_ablate_score(Model *m, const char *path){
free(ln); free(ids); free(x); free(lo); free(row); fclose(f);
}
-static void generate(Model *m, const int *prompt, int np, int n_new, int *out){
+static int generate(Model *m, const int *prompt, int np, int n_new, int *out, int *finite){
kv_alloc(m,np+n_new+g_draft+2);
for(int i=0;ic.vocab,1};
+ int emitted=spec_decode(m,out,np,n_new,-1,logit,emit_oracle,&es,NULL,NULL);
+ *finite=es.finite;
+ return emitted;
}
static void profile_print(Model *m, double elapsed){
@@ -8389,6 +8635,22 @@ static void run_text(Model *m, const char *snap, const char *prompt, int ngen){
printf(" | EXPERT_BUDGET=%d (dropped %lld experts, ~%.1f GB I/O saved)", g_expert_budget, (long long)g_budget_dropped, g_budget_dropped*18.9e6/1e9);
if(g_budget_rescued) printf(" [%lld rescued: budget too tight, position would have had 0 routed experts]", (long long)g_budget_rescued);
}
+ if(g_degrade_zero){
+ printf(" | DEGRADE_ZERO tau=%.3f (zeroed %lld miss slots", g_degrade_tau, (long long)g_degrade_dropped);
+ if(g_degrade_dropped>0){
+ /* top-3 layers by drop count */
+ int top[3]={-1,-1,-1}; int64_t tv[3]={0,0,0};
+ for(int i=0;i<512;i++){
+ int64_t v=g_degrade_dropped_by_layer[i]; if(!v) continue;
+ if(v>tv[0]){tv[2]=tv[1];top[2]=top[1];tv[1]=tv[0];top[1]=top[0];tv[0]=v;top[0]=i;}
+ else if(v>tv[1]){tv[2]=tv[1];top[2]=top[1];tv[1]=v;top[1]=i;}
+ else if(v>tv[2]){tv[2]=v;top[2]=i;}
+ }
+ printf("; top layers:");
+ for(int k=0;k<3&&top[k]>=0;k++) printf(" L%d:%lld",top[k],(long long)tv[k]);
+ }
+ printf(")");
+ }
printf("\n");
printf("speculation: %.2f tokens/forward (%llu forwards per %llu tokens) | MTP acceptance %.0f%% (%llu/%llu)\n",
m->n_fw?(double)m->n_emit/m->n_fw:1.0, (unsigned long long)m->n_fw, (unsigned long long)m->n_emit,
@@ -9501,13 +9763,6 @@ static void run_serve(Model *m, const char *snap){
free(ctx); m->kv=NULL; m->Lc=m->Rc=m->Ic=NULL; m->Lc8=m->Rc8=NULL; m->Lsc=m->Rsc=NULL; m->kv_start=NULL; m->max_t=0;
}
-static int *read_arr(jval*o,const char*k,int*n){
- jval*a=json_get(o,k);
- if(!a){ *n=0; return NULL; }
- int*r=malloc(a->len*sizeof(int));
- if(!r){ fprintf(stderr,"OOM read_arr\n"); exit(1); }
- for(int i=0;ilen;i++) r[i]=(int)a->kids[i]->num; *n=a->len; return r; }
-
/* telemetry, stats, usage persistence — moved to telemetry.h */
#ifdef COLI_VULKAN
@@ -10908,6 +11163,19 @@ static int coli_env_on(const char *name)
#ifndef COLIBRI_NO_MAIN
int main(int argc, char **argv){
+ int strict=coli_env_on("ORACLE_STRICT");
+ if(strict){
+ const char *modes[]={"REPLAY","CONSIST","SERVE","SCORE","ABLATE_SCORE","EXPERT_WORKER",
+ "I4_ACC512_TEST","I3_AVX512_TEST"};
+ for(size_t i=0;i16) g_pilot_nw=16;
g_pilot_evict_guard = getenv("PILOT_EVICT_GUARD")?atoi(getenv("PILOT_EVICT_GUARD")):1; /* 0 = old LRU eviction (A/B) */
+ g_degrade_zero = getenv("DEGRADE_ZERO")?atoi(getenv("DEGRADE_ZERO")):0;
+ g_degrade_tau = getenv("DEGRADE_TAU") ?atof(getenv("DEGRADE_TAU")) :0.03f;
+ if(g_degrade_tau<=0.f||g_degrade_tau>1.f) g_degrade_tau=0.03f; /* clamp to sane range */
+ if(g_degrade_zero)
+ fprintf(stderr,"[DEGRADE] zero-fill ON, tau=%.3f (approximate mode: miss slots with per-position gate weight < tau are never loaded)\n",g_degrade_tau);
g_disk_split = getenv("DISK_SPLIT")?atoi(getenv("DISK_SPLIT")):0; /* 1 = split dei disk load nelle stats */
g_pipe = getenv("PIPE")?atoi(getenv("PIPE")):
#ifdef _WIN32
@@ -11684,13 +11957,23 @@ int main(int argc, char **argv){
}
/* altrimenti: validazione contro l'oracolo (ref_glm.json) */
+ /* Diagnostic modes take precedence over TF and do not read predictions. */
+ int teacher_forcing=getenv("TF")!=NULL && !getenv("REPLAY") && !getenv("CONSIST");
const char *refpath=getenv("REF")?getenv("REF"):"ref_glm.json";
char *b=cfg_slurp(refpath);
- if(!b){ fprintf(stderr,"%s: cannot read oracle file (missing, unreadable, short, or > %lld bytes)\n",refpath,(long long)CFG_MAX_BYTES); return 1; }
- char *ar=NULL; jval *ref=json_parse(b,&ar);
- int np=0,nfull=0; int *prompt=read_arr(ref,"prompt_ids",&np); int *full=read_arr(ref,"full_ids",&nfull);
- if(!prompt||!full||np<1||nfull %lld bytes)\n",refpath,(long long)CFG_MAX_BYTES); return 1; }
+ OracleRef ref;
+ int valid=oracle_ref_parse(b,m.c.vocab,teacher_forcing,&ref);
+ free(b);
+ if(!valid) return 1;
+ int np=ref.np,nfull=ref.nfull; int *prompt=ref.prompt,*full=ref.full;
int n_new=nfull-np;
+ int tf_allowed=0;
+ if(strict && teacher_forcing &&
+ !oracle_tf_allowance(getenv("ORACLE_TF_MAX_MISMATCHES"),nfull,&tf_allowed)){
+ fprintf(stderr,"[ORACLE] ORACLE_TF_MAX_MISMATCHES must be an integer in [0,%d)\n",nfull);
+ oracle_ref_free(&ref); return 1;
+ }
/* L'oracolo (ref_glm.json in repo) e' del modello TINY: contro il 744B da' 0/20
* garantito su OGNI piattaforma (prompt-token tiny = spazzatura per il modello vero).
* Non e' un bug del motore — vedi #76. */
@@ -11707,25 +11990,29 @@ int main(int argc, char **argv){
" Nessun PROMPT: modo auto-validazione, ma ref_glm.json e' l'oracolo del modello TINY\n"
" (token max %d, il tuo vocab e' %d). Usa PROMPT=... per generare davvero (vedi sopra).\n",
maxid, m.c.vocab, maxid, m.c.vocab);
- return 1;
+ oracle_ref_free(&ref); return 1;
} }
if(getenv("REPLAY")){
run_replay(&m,full,nfull,np);
if(stats) stats_dump(&m,stats);
+ oracle_ref_free(&ref);
return 0;
}
if(getenv("CONSIST")){
run_consist(&m,full,nfull,np);
if(stats) stats_dump(&m,stats);
+ oracle_ref_free(&ref);
return 0;
}
- if(getenv("TF")){
- int *tf=read_arr(ref,"tf_pred",&(int){0});
- int *pred=malloc(nfull*sizeof(int)); double tt=now_s();
- forward_all(&m, full, nfull, pred, tf); double tdt=now_s()-tt;
+ if(teacher_forcing){
+ int *tf=ref.tf;
+ int *pred=malloc((size_t)nfull*sizeof(int));
+ if(!pred){ oracle_ref_free(&ref); return 1; }
+ double tt=now_s();
+ int finite=forward_all(&m, full, nfull, pred, tf); double tdt=now_s()-tt;
int ok=0; for(int i=0;itf_allowed);
}
- int *out=malloc((np+n_new)*sizeof(int));
+ int *out=malloc((size_t)nfull*sizeof(int));
+ if(!out){ oracle_ref_free(&ref); return 1; }
ProfBase pb; prof_base(&m,&pb);
- double t=now_s(); generate(&m,prompt,np,n_new,out); double dt=now_s()-t;
+ int finite=1;
+ double t=now_s(); int emitted=generate(&m,prompt,np,n_new,out,&finite); double dt=now_s()-t;
int match=0;
printf("\nReference (oracle): "); for(int i=np;i
+#include
#endif
#include
-static inline double compat_mem_available_gb(void){
+
+/* Total AND available in one call, one pass.
+ *
+ * "Available" alone was consolidated here by #1375 because there were two
+ * copies of it and one was wrong. "Total" is now in the same position: every
+ * caller that wants it hand-rolls its own #ifdef ladder (telemetry.h uses
+ * sysctl hw.memsize on macOS, olmoe.c uses sysconf(_SC_PHYS_PAGES), glm53.c
+ * read /proc/meminfo unconditionally), and those definitions do not agree.
+ * A budget computed from a total and an available that came from two
+ * different definitions is not a budget, it is a coincidence.
+ *
+ * One pass also matters on Linux specifically: MemTotal and MemAvailable are
+ * two lines of the same file, and reading it twice to get them is both a
+ * second open and a second chance to read a file that changed underneath.
+ *
+ * 0 means "not measurable" for either field; the caller decides the fallback. */
+static inline void compat_meminfo_gb(double *total_gb, double *avail_gb){
+ double total = 0, avail = 0;
#ifdef __APPLE__
+ uint64_t memsize = 0; size_t len = sizeof memsize;
+ if(sysctlbyname("hw.memsize", &memsize, &len, NULL, 0) == 0) total = (double)memsize / 1e9;
mach_msg_type_number_t cnt = HOST_VM_INFO64_COUNT;
vm_statistics64_data_t vm;
- if(host_statistics64(mach_host_self(), HOST_VM_INFO64, (host_info64_t)&vm, &cnt) != KERN_SUCCESS) return 0;
- return ((double)vm.free_count + (double)vm.inactive_count + (double)vm.purgeable_count)
- * (double)sysconf(_SC_PAGESIZE) / 1e9;
+ if(host_statistics64(mach_host_self(), HOST_VM_INFO64, (host_info64_t)&vm, &cnt) == KERN_SUCCESS)
+ avail = ((double)vm.free_count + (double)vm.inactive_count + (double)vm.purgeable_count)
+ * (double)sysconf(_SC_PAGESIZE) / 1e9;
#elif defined(_WIN32)
MEMORYSTATUSEX msx = {0};
msx.dwLength = sizeof(msx);
- if(!GlobalMemoryStatusEx(&msx)) return 0;
- double phys = (double)msx.ullAvailPhys / 1e9;
- double commit = (double)msx.ullAvailPageFile / 1e9;
- return commit > 0 && commit < phys ? commit : phys;
+ if(GlobalMemoryStatusEx(&msx)){
+ total = (double)msx.ullTotalPhys / 1e9;
+ double phys = (double)msx.ullAvailPhys / 1e9;
+ double commit = (double)msx.ullAvailPageFile / 1e9;
+ avail = commit > 0 && commit < phys ? commit : phys;
+ }
#else
- FILE *f = fopen("/proc/meminfo", "r"); if(!f) return 0;
- char ln[256]; double kb = 0;
- while(fgets(ln, sizeof ln, f)) if(sscanf(ln, "MemAvailable: %lf", &kb) == 1) break;
- fclose(f); return kb / 1e6;
+ FILE *f = fopen("/proc/meminfo", "r");
+ if(f){
+ char ln[256]; double kb;
+ /* /proc/meminfo's "kB" is KiB (1024 B), so a GB is kb*1024/1e9, not
+ * kb/1e6. The old kb/1e6 understated by 2.3% -- harmless while the
+ * number was only ever compared against itself, but glm53 now weighs
+ * it against a model size computed from byte counts (/1e9, true GB),
+ * and a budget that subtracts true GB from understated GB is wrong in
+ * the direction that matters: it hands back less than it should. */
+ /* MemTotal precedes MemAvailable in /proc/meminfo, but do not rely on
+ * the order: stop only once both have been seen. */
+ while((total == 0 || avail == 0) && fgets(ln, sizeof ln, f)){
+ if(total == 0 && sscanf(ln, "MemTotal: %lf", &kb) == 1){ total = kb * 1024.0 / 1e9; continue; }
+ if(avail == 0 && sscanf(ln, "MemAvailable: %lf", &kb) == 1) avail = kb * 1024.0 / 1e9;
+ }
+ fclose(f);
+ }
#endif
+ if(total_gb) *total_gb = total;
+ if(avail_gb) *avail_gb = avail;
+}
+
+static inline double compat_mem_available_gb(void){
+ double avail = 0;
+ compat_meminfo_gb(NULL, &avail);
+ return avail;
}
#endif /* COMPAT_H */
diff --git a/c/deepseek_v4.c b/c/deepseek_v4.c
index 66a831415..bc8f38622 100644
--- a/c/deepseek_v4.c
+++ b/c/deepseek_v4.c
@@ -1368,7 +1368,7 @@ static int build_runtime_plan(ColiV4Engine *engine,
uint64_t maximum_layer = 0, dense_total = 0;
for (int layer = 0; layer < config->num_hidden_layers; layer++) {
ColiDeepSeekV4LayerPlan layer_plan;
- ColiDeepSeekV4LayerStats stats;
+ ColiDeepSeekV4LayerStats stats = {0};
if (coli_v4_layer_plan(&layer_plan, config, layer,
error, error_size) ||
coli_v4_layer_validate(&layer_plan, index, &stats,
@@ -8241,8 +8241,9 @@ static size_t hot_slot_index(const V4ExpertStoreState *state,
*
* FLOCK-packed checkpoints store [scales][weights] contiguously and need one
* request. Standard HF checkpoints keep the ranges apart: weights use direct
- * I/O while the much smaller scales use buffered pread. Any direct-I/O error
- * falls back to the exact buffered path. */
+ * I/O while the much smaller scales use buffered pread. REAP-style
+ * per_matrix records issue one window per scale/weight segment. Any
+ * direct-I/O error falls back to the exact buffered path. */
static uint64_t v4_direct_reads;
static uint64_t v4_direct_flock_reads;
static uint64_t v4_direct_payload_bytes;
@@ -8298,33 +8299,87 @@ static int v4_read_direct_window(const V4ExpertStoreState *state, int shard,
return 0;
}
+/* Direct window into an interior slab offset. v4_read_direct_window bounces
+ * at slab[0], so a second per_matrix segment would clobber earlier bytes. */
+static int v4_read_direct_copy(const V4ExpertStoreState *state, int shard,
+ int rep, unsigned char *destination,
+ uint64_t offset, size_t length) {
+ if (!destination) return -1;
+ if (!length) return 0;
+ if (length > SIZE_MAX - 8192u) return -1;
+ unsigned char *bounce = NULL;
+ if (posix_memalign((void **)&bounce, 4096, length + 8192u)) return -1;
+ int result = v4_read_direct_window(state, shard, rep, bounce, offset,
+ length, 0);
+ if (!result) memcpy(destination, bounce, length);
+ compat_aligned_free(bounce);
+ return result;
+}
+
+static int v4_try_direct_segment(V4ExpertStoreState *state, int shard, int rep,
+ V4ExpertSlot *slot, uint64_t dest,
+ uint64_t offset, uint64_t bytes) {
+ if (!slot->aligned_slab ||
+ !coli_st_streaming_direct_available_rep(state->index, shard, rep))
+ return -1;
+ size_t length = (size_t)bytes;
+ if (dest == 0)
+ return v4_read_direct_window(state, shard, rep, slot->slab, offset,
+ length, 0);
+ return v4_read_direct_copy(state, shard, rep, slot->slab + dest, offset,
+ length);
+}
+
+static int v4_read_per_matrix_segment(V4ExpertStoreState *state, int shard,
+ int rep, V4ExpertSlot *slot,
+ uint64_t dest, uint64_t offset,
+ uint64_t bytes, int *used_direct,
+ int *used_fallback) {
+ if (!v4_try_direct_segment(state, shard, rep, slot, dest, offset, bytes)) {
+ *used_direct = 1;
+ return 0;
+ }
+ if (slot->aligned_slab &&
+ coli_st_streaming_direct_available_rep(state->index, shard, rep))
+ *used_fallback = 1;
+ return coli_st_read_at_rep(state->index, shard, rep, offset, (size_t)bytes,
+ slot->slab + dest);
+}
+
static int v4_read_expert_record(V4ExpertStoreState *state,
const V4ExpertRecord *record,
V4ExpertSlot *slot, int rep) {
if (record->per_matrix) {
- int direct_available = slot->aligned_slab &&
- coli_st_streaming_direct_available_rep(state->index, record->m_scale_shard[0], rep);
- if (direct_available)
- __atomic_fetch_add(&v4_direct_fallbacks, UINT64_C(1),
- __ATOMIC_RELAXED);
+ int used_direct = 0;
+ int used_fallback = 0;
uint64_t scale_cursor = 0;
for (int matrix = 0; matrix < V4_MATRIX_COUNT; matrix++) {
- if (coli_st_read_at_rep(state->index, record->m_scale_shard[matrix], rep,
- record->m_scale_offset[matrix],
- (size_t)record->m_scale_bytes[matrix],
- slot->slab + scale_cursor) != 0)
+ if (v4_read_per_matrix_segment(
+ state, record->m_scale_shard[matrix], rep, slot,
+ scale_cursor, record->m_scale_offset[matrix],
+ record->m_scale_bytes[matrix], &used_direct,
+ &used_fallback) != 0)
return -1;
scale_cursor += record->m_scale_bytes[matrix];
}
uint64_t weight_cursor = scale_cursor;
for (int matrix = 0; matrix < V4_MATRIX_COUNT; matrix++) {
- if (coli_st_read_at_rep(state->index, record->m_weight_shard[matrix], rep,
- record->m_weight_offset[matrix],
- (size_t)record->m_weight_bytes[matrix],
- slot->slab + weight_cursor) != 0)
+ if (v4_read_per_matrix_segment(
+ state, record->m_weight_shard[matrix], rep, slot,
+ weight_cursor, record->m_weight_offset[matrix],
+ record->m_weight_bytes[matrix], &used_direct,
+ &used_fallback) != 0)
return -1;
weight_cursor += record->m_weight_bytes[matrix];
}
+ if (used_direct && !used_fallback) {
+ __atomic_fetch_add(&v4_direct_reads, UINT64_C(1), __ATOMIC_RELAXED);
+ __atomic_fetch_add(&v4_direct_payload_bytes, record->record_bytes,
+ __ATOMIC_RELAXED);
+ } else if (used_fallback) {
+ __atomic_fetch_add(&v4_direct_fallbacks, UINT64_C(1),
+ __ATOMIC_RELAXED);
+ }
return 0;
}
int direct_available = slot->aligned_slab &&
@@ -8782,6 +8837,37 @@ int coli_v4_test_expert_slot_index(ColiExpertStore *store, ColiExpertKey key) {
pthread_mutex_unlock(&state->mutex);
return result;
}
+
+void coli_v4_test_reset_direct_io_stats(void) {
+ __atomic_store_n(&v4_direct_reads, 0, __ATOMIC_RELAXED);
+ __atomic_store_n(&v4_direct_flock_reads, 0, __ATOMIC_RELAXED);
+ __atomic_store_n(&v4_direct_payload_bytes, 0, __ATOMIC_RELAXED);
+ __atomic_store_n(&v4_direct_fallbacks, 0, __ATOMIC_RELAXED);
+}
+
+uint64_t coli_v4_test_direct_reads(void) {
+ return __atomic_load_n(&v4_direct_reads, __ATOMIC_RELAXED);
+}
+
+uint64_t coli_v4_test_direct_fallbacks(void) {
+ return __atomic_load_n(&v4_direct_fallbacks, __ATOMIC_RELAXED);
+}
+
+int coli_v4_test_force_streaming_direct(ColiExpertStore *store) {
+ if (!store || !store->state) return -1;
+ V4ExpertStoreState *state = store->state;
+ if (!state->index) return -1;
+ int enabled = 0;
+ for (int i = 0; i < state->index->nfd; i++) {
+ if (state->index->dfds[i] < 0 && state->index->fds[i] >= 0) {
+ int twin = dup(state->index->fds[i]);
+ if (twin < 0) return -1;
+ state->index->dfds[i] = twin;
+ }
+ if (state->index->dfds[i] >= 0) enabled = 1;
+ }
+ return enabled ? 0 : -1;
+}
#endif
/* Let the layer currently sweeping a batched CPU prefill borrow the complete
@@ -11635,6 +11721,7 @@ int coli_v4_prompt_build(char **output, size_t *output_length,
#include "json.h"
#include "native_quant.h"
#include "serve_codec.h"
+#include "decode_batch.h" /* coli_logprob_tail: the numeric channel's tail, same bytes as the other engines */
#include "tok.h"
static int load_embedding(float *state, const ColiSafetensorsIndex *index,
@@ -11772,55 +11859,39 @@ static int head_argmax(ColiV4Engine *engine, const float *hidden,
g_v4_prof_head_s += spec_now() - t0;
return result;
}
-static int head_argmax_impl(ColiV4Engine *engine, const float *hidden,
+/* Every head score of one hidden row, in vocabulary order. head_argmax used
+ * to run this matmul and keep only the maximum; the numeric channel (SUBMIT
+ * logprobs=k, docs/brio.md) needs the whole row, so the row is computed here
+ * once and the argmax is a scan over it. Same head_bf16_dot per row, same scan
+ * order: the token picked and its logit do not change. */
+static int head_scores_impl(ColiV4Engine *engine, const float *hidden,
const ColiSafetensorsIndex *index,
- const ColiDeepSeekV4Config *config,
- int *best_token, float *best_logit) {
+ const ColiDeepSeekV4Config *config, float *scores) {
const ColiSafetensorsTensor *head = coli_st_find(index, "head.weight");
int d = config->hidden_size, vocab = config->vocab_size;
- if (!head || head->dtype != COLI_ST_BF16 || d < 1 || vocab < 1)
+ if (!head || head->dtype != COLI_ST_BF16 || d < 1 || vocab < 1 || !scores)
return -1;
int shard = coli_st_tensor_shard(index, head);
size_t resident_bytes = (size_t)vocab * (size_t)d * sizeof(uint16_t);
const uint16_t *resident = coli_v4_head_cache_data(
engine, shard, (uint64_t)head->off, resident_bytes);
-
/* The normal V4 memory plan keeps the BF16 head resident. Compute all
* rows in one OpenMP team directly from that allocation: the old tiled
* path copied the complete ~1 GiB head and created ~2,000 teams per token.
* Each row retains the same scalar accumulation order and the final scan
* retains vocabulary order, so logits/tie-breaking do not change. */
if (resident) {
- float *scores = malloc((size_t)vocab * sizeof(*scores));
- if (!scores) return -1;
#pragma omp parallel for schedule(static)
for (int row = 0; row < vocab; row++) {
const uint16_t *weight = resident + (size_t)row * d;
scores[row] = head_bf16_dot(weight, hidden, d);
}
- int winner = -1;
- float maximum = -FLT_MAX;
- for (int row = 0; row < vocab; row++)
- if (scores[row] > maximum) {
- maximum = scores[row];
- winner = row;
- }
- free(scores);
- *best_token = winner;
- *best_logit = maximum;
- return winner < 0 ? -1 : 0;
+ return 0;
}
-
/* Low-memory fallback: stream small row tiles exactly as before. */
enum { ROWS = 64 };
uint16_t *raw = malloc((size_t)ROWS * d * sizeof(*raw));
- float *scores = malloc((size_t)ROWS * sizeof(*scores));
- if (!raw || !scores) {
- free(scores); free(raw);
- return -1;
- }
- int winner = -1;
- float maximum = -FLT_MAX;
+ if (!raw) return -1;
for (int start = 0; start < vocab; start += ROWS) {
int rows = vocab - start < ROWS ? vocab - start : ROWS;
size_t bytes = (size_t)rows * d * sizeof(*raw);
@@ -11828,26 +11899,54 @@ static int head_argmax_impl(ColiV4Engine *engine, const float *hidden,
engine, index, shard,
(uint64_t)head->off + (uint64_t)start * d * sizeof(*raw),
bytes, raw)) {
- free(scores); free(raw);
+ free(raw);
return -1;
}
#pragma omp parallel for
for (int row = 0; row < rows; row++) {
const uint16_t *weight = raw + (size_t)row * d;
- scores[row] = head_bf16_dot(weight, hidden, d);
+ scores[start + row] = head_bf16_dot(weight, hidden, d);
}
- for (int row = 0; row < rows; row++)
- if (scores[row] > maximum) {
- maximum = scores[row];
- winner = start + row;
- }
}
- free(scores); free(raw);
+ free(raw);
+ return 0;
+}
+/* First maximum in vocabulary order: the tie-break head_argmax always had. */
+static int head_scores_argmax(const float *scores, int vocab,
+ int *best_token, float *best_logit) {
+ int winner = -1;
+ float maximum = -FLT_MAX;
+ for (int row = 0; row < vocab; row++)
+ if (scores[row] > maximum) {
+ maximum = scores[row];
+ winner = row;
+ }
*best_token = winner;
*best_logit = maximum;
return winner < 0 ? -1 : 0;
}
-
+static int head_argmax_impl(ColiV4Engine *engine, const float *hidden,
+ const ColiSafetensorsIndex *index,
+ const ColiDeepSeekV4Config *config,
+ int *best_token, float *best_logit) {
+ int vocab = config->vocab_size;
+ if (vocab < 1) return -1;
+ float *scores = malloc((size_t)vocab * sizeof(*scores));
+ if (!scores) return -1;
+ int result = head_scores_impl(engine, hidden, index, config, scores);
+ if (!result) result = head_scores_argmax(scores, vocab, best_token, best_logit);
+ free(scores);
+ return result;
+}
+/* The whole row, under the same head-time meter as head_argmax. */
+static int head_scores(ColiV4Engine *engine, const float *hidden,
+ const ColiSafetensorsIndex *index,
+ const ColiDeepSeekV4Config *config, float *scores) {
+ double t0 = spec_now();
+ int result = head_scores_impl(engine, hidden, index, config, scores);
+ g_v4_prof_head_s += spec_now() - t0;
+ return result;
+}
static int head_argmax_batch(ColiV4Engine *engine, const float *hidden,
const ColiSafetensorsIndex *index,
const ColiDeepSeekV4Config *config, int batch,
@@ -12651,6 +12750,8 @@ static void session_free_attention(ColiV4Session *session) {
void coli_v4_session_destroy(ColiV4Session *session) {
if (!session) return;
kv_prefix_free(&session->fed);
+ free(session->pin_ids); free(session->pin_scores);
+ free(session->echo_hidden); free(session->echo_scores);
session_free_attention(session);
session_free_buffers(session);
if (session->tokenizer_ready) {
@@ -13135,7 +13236,8 @@ int coli_v4_session_generate(ColiV4Session *session,
ColiV4SessionGenerateStats *stats_out,
char *error, size_t error_size) {
if (!session || !session->engine || !prompt || !options ||
- options->max_new_tokens < 1) {
+ (options->max_new_tokens < 1 &&
+ !(options->max_new_tokens == 0 && options->logprobs > 0))) {
if (error && error_size)
snprintf(error, error_size, "invalid V4 session generate arguments");
return -1;
@@ -13243,6 +13345,34 @@ int coli_v4_session_generate(ColiV4Session *session,
fprintf(stderr, "[PREFIX] hint boundary at %d tokens\n", ckpt_at);
}
session->prefix_reused = reuse;
+ /* The numeric channel (docs/brio.md). Scratch sized to the head, kept on
+ * the session so every early return below leaves nothing behind. */
+ const int vocab = config->vocab_size;
+ const int echo = options->logprobs > 0 && options->on_echo != NULL;
+ const int want_scores = echo || options->pin || options->on_scores != NULL;
+ if (want_scores && (!session->echo_hidden || !session->echo_scores)) {
+ free(session->echo_hidden);
+ free(session->echo_scores);
+ session->echo_hidden = malloc((size_t)config->hidden_size * sizeof(float));
+ session->echo_scores = malloc((size_t)vocab * sizeof(float));
+ if (!session->echo_hidden || !session->echo_scores) {
+ if (error && error_size)
+ snprintf(error, error_size, "out of memory for the logprob channel");
+ return -1;
+ }
+ }
+ /* Position `reuse` is the first fresh token, and its predictor lives in
+ * the state we continue from, which nothing below recomputes. When that
+ * state is the pinned prompt end, its scores were kept for exactly this: a
+ * closed-set caller pins the prompt, then asks about each option, and the
+ * option's first token is usually its only one. Any other reuse has no
+ * predictor to report; the caller sees the position missing, as with the
+ * other engines. */
+ if (echo && reuse > 0 && reuse < prompt_count && session->pin_scores &&
+ session->pin_len == reuse &&
+ !memcmp(session->pin_ids, session->prompt_ids, (size_t)reuse * sizeof(int)))
+ options->on_echo(options->scores_user_data, reuse,
+ session->prompt_ids[reuse], session->pin_scores, vocab);
if (reuse && getenv("V4_PREFIX_LOG"))
fprintf(stderr, "[PREFIX] reusing %d of %d prompt tokens\n",
reuse, prompt_count);
@@ -13313,6 +13443,25 @@ int coli_v4_session_generate(ColiV4Session *session,
}
kv_prefix_record(&session->fed, session->prompt_ids + done_upto,
done_upto, seg);
+ /* Read-out of the prefill: row `item` of this segment is position
+ * done_upto+item and predicts the token at the next one. One head pass
+ * per row, paid only by the requests that opened the channel. */
+ if (echo) {
+ for (int item = 0; item < seg; item++) {
+ int at = done_upto + item + 1;
+ if (at >= prompt_count) break;
+ if (final_hidden(session->echo_hidden, state + (size_t)item * hd,
+ index, config, error, error_size) ||
+ head_scores(engine, session->echo_hidden, index, config,
+ session->echo_scores)) {
+ kv_prefix_taint(&session->fed);
+ return -1;
+ }
+ options->on_echo(options->scores_user_data, at,
+ session->prompt_ids[at], session->echo_scores,
+ vocab);
+ }
+ }
done_upto += seg;
session->fed.len = done_upto;
tail_rows = seg;
@@ -13346,10 +13495,36 @@ int coli_v4_session_generate(ColiV4Session *session,
int current = 0;
float current_logit = 0.0f;
if (final_hidden(hidden, last, index, config, error, error_size) ||
- head_argmax(engine, hidden, index, config, ¤t, ¤t_logit)) {
+ (want_scores
+ ? (head_scores(engine, hidden, index, config, session->echo_scores) ||
+ head_scores_argmax(session->echo_scores, vocab, ¤t,
+ ¤t_logit))
+ : head_argmax(engine, hidden, index, config, ¤t,
+ ¤t_logit))) {
kv_prefix_taint(&session->fed);
return -1;
}
+ if (options->pin) {
+ /* Keep what the snapshot cannot: the scores at the prompt end. The
+ * attention state goes to a v4_ckpt slot regardless of the size gate
+ * above: a pinned prompt is short by nature (a document and a
+ * question) and is about to be extended by every option. */
+ int *ids = realloc(session->pin_ids, (size_t)prompt_count * sizeof(int));
+ float *keep = realloc(session->pin_scores, (size_t)vocab * sizeof(float));
+ if (ids) session->pin_ids = ids;
+ if (keep) session->pin_scores = keep;
+ if (ids && keep) {
+ memcpy(session->pin_ids, session->prompt_ids,
+ (size_t)prompt_count * sizeof(int));
+ memcpy(session->pin_scores, session->echo_scores,
+ (size_t)vocab * sizeof(float));
+ session->pin_len = prompt_count;
+ } else {
+ session->pin_len = 0; /* an optimisation, never an error */
+ }
+ if (v4_ckpt_min_tokens() && !v4_ckpt_have(session->prompt_ids, prompt_count))
+ v4_ckpt_store(session, prompt_count, 1);
+ }
/* The prompt is in the attention state from here on; record it before the
* decode loop so a failure mid-generation still leaves fed describing what
* was actually fed. */
@@ -13357,10 +13532,19 @@ int coli_v4_session_generate(ColiV4Session *session,
session->fed.len = prompt_count;
int generated_count = 0;
int last_processed = prompt_count - 1;
- generated[generated_count++] = current;
- int done = session_emit_token(session, on_token, user_data, current,
+ /* max_new == 0 is the read-only request of the numeric channel: the
+ * prompt is in the state, its read-out went through on_echo, nothing is
+ * generated and `done` skips the loop; the tail then reports zero. */
+ int done = 1;
+ if (max_new > 0) {
+ generated[generated_count++] = current;
+ if (options->on_scores)
+ options->on_scores(options->scores_user_data, last_processed, current,
+ session->echo_scores, vocab);
+ done = session_emit_token(session, on_token, user_data, current,
current_logit, last_processed,
generated_count, options->stop_at_sentence);
+ }
double first_at = spec_now();
int draft_limit = getenv("V4_DRAFT") ? atoi(getenv("V4_DRAFT")) : 0;
@@ -13372,7 +13556,11 @@ int coli_v4_session_generate(ColiV4Session *session,
while (!done && generated_count < max_new) {
int remaining = max_new - generated_count;
- if (!options->no_dspark && !session->spec_disabled && remaining >= 3) {
+ /* A draft block accepts several tokens from one target pass and has
+ * no per-token scores to report, so the numeric channel takes the
+ * plain path: same greedy tokens, one head row each. */
+ if (!options->no_dspark && options->logprobs <= 0 &&
+ !session->spec_disabled && remaining >= 3) {
int inputs[25] = {0}, drafts[24] = {0};
int predictions[25] = {0};
float logits[25] = {0};
@@ -13590,12 +13778,20 @@ int coli_v4_session_generate(ColiV4Session *session,
session->state = state;
session->next = next;
if (final_hidden(hidden, state, index, config, error, error_size) ||
- head_argmax(engine, hidden, index, config, ¤t, ¤t_logit)) {
+ (options->on_scores
+ ? (head_scores(engine, hidden, index, config, session->echo_scores) ||
+ head_scores_argmax(session->echo_scores, vocab, ¤t,
+ ¤t_logit))
+ : head_argmax(engine, hidden, index, config, ¤t,
+ ¤t_logit))) {
kv_prefix_taint(&session->fed);
return -1;
}
last_processed = position;
generated[generated_count++] = current;
+ if (options->on_scores)
+ options->on_scores(options->scores_user_data, last_processed, current,
+ session->echo_scores, vocab);
done = session_emit_token(session, on_token, user_data, current,
current_logit, last_processed,
generated_count,
@@ -13810,6 +14006,8 @@ typedef struct {
float top_p;
int extension_bytes;
int prefix_bytes;
+ int logprobs; /* SUBMIT logprobs=k: 0 = channel closed (opt-in) */
+ int pin; /* SUBMIT pin=1: keep the prompt end for the next prompts */
} V4ServeRequest;
typedef struct {
@@ -13817,6 +14015,8 @@ typedef struct {
const char *request_id;
int cancelled;
int fatal;
+ int logprobs;
+ char tail[1024]; /* the next DATA frame's logprob tail, from on_scores */
} V4ServeStream;
static const ColiServeWireProfile v4_wire = {
@@ -14049,6 +14249,8 @@ static int v4_serve_read_request(FILE *input, FILE *output,
request->top_p = command.top_p;
request->extension_bytes = (int)command.extension_bytes;
request->prefix_bytes = prefix_bytes;
+ request->logprobs = command.logprobs;
+ request->pin = command.pin;
coli_serve_command_dispose(&command);
return 2;
}
@@ -14092,8 +14294,15 @@ static int v4_serve_token(void *user_data, int token, float logit,
char piece[1024];
int bytes = tok_decode(&stream->session->tokenizer, &token, 1,
piece, (int)sizeof(piece) - 1);
- v4_serve_data(stdout, stream->request_id, piece, bytes);
+ /* With the channel open the frame carries the tail on_scores left
+ * here: "DATA [tid tlp]*k", one frame per token. */
+ if (stream->logprobs > 0 && bytes > 0)
+ coli_serve_write_data_lp(stdout, stream->request_id, piece,
+ (size_t)bytes, stream->tail);
+ else
+ v4_serve_data(stdout, stream->request_id, piece, bytes);
}
+ stream->tail[0] = 0;
if (v4_serve_drain_commands(stream)) {
stream->cancelled = 1;
return 1;
@@ -14129,6 +14338,34 @@ static void v4_serve_done(FILE *output, const char *id, int completion,
coli_serve_write_done_i32_suffix(output, id, &done, &prefix_reused, 1);
}
+/* ECHO frame of the numeric channel, the same bytes the other engines'
+ * serve_echo writes: "ECHO [tid tlp]*k" and the
+ * token's bytes DATA-framed after it. `scores` are raw head logits;
+ * coli_logprob_tail does the normalisation and the top-k. */
+static void v4_serve_echo(void *user_data, int position, int token,
+ const float *scores, int vocab) {
+ V4ServeStream *stream = user_data;
+ char tail[1024], piece[1024];
+ coli_logprob_tail(tail, sizeof tail, scores, vocab, token, stream->logprobs);
+ int bytes = tok_decode(&stream->session->tokenizer, &token, 1, piece,
+ (int)sizeof(piece) - 1);
+ if (bytes < 0) bytes = 0;
+ printf("ECHO %s %d %d%s\n", stream->request_id, bytes, position, tail);
+ if (bytes > 0) fwrite(piece, 1, (size_t)bytes, stdout);
+ fputc('\n', stdout);
+ fflush(stdout);
+}
+
+/* The tail of the next DATA frame, computed while the scores exist and
+ * written by v4_serve_token right after. */
+static void v4_serve_scores(void *user_data, int position, int token,
+ const float *scores, int vocab) {
+ (void)position;
+ V4ServeStream *stream = user_data;
+ coli_logprob_tail(stream->tail, sizeof stream->tail, scores, vocab, token,
+ stream->logprobs);
+}
+
static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session,
V4ServeRequest *request) {
if (request->extension_bytes) {
@@ -14190,7 +14427,7 @@ static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session,
engine->experts ? coli_v4_expert_store_matmul_sec(engine->experts) : 0.0;
double block_before = g_v4_prof_block_s, head_before = g_v4_prof_head_s;
long long forwards_before = g_v4_prof_forwards;
- V4ServeStream stream = {session, request->id, 0, 0};
+ V4ServeStream stream = {session, request->id, 0, 0, request->logprobs, {0}};
ColiV4SessionGenerateStats stats = {0};
char error[512] = {0};
double started = spec_now();
@@ -14203,6 +14440,11 @@ static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session,
.should_abort = v4_serve_abort,
.abort_user_data = &stream,
.prefix_bytes = (size_t)request->prefix_bytes,
+ .logprobs = request->logprobs,
+ .pin = request->pin,
+ .on_echo = request->logprobs > 0 ? v4_serve_echo : NULL,
+ .on_scores = request->logprobs > 0 ? v4_serve_scores : NULL,
+ .scores_user_data = &stream,
},
v4_serve_token, &stream, &stats, error, sizeof(error));
double elapsed = spec_now() - started;
@@ -14219,6 +14461,7 @@ static int v4_serve_one(ColiV4Engine *engine, ColiV4Session *session,
int completion = stats.generated_tokens - (stats.eos_stopped ? 1 : 0);
if (completion < 0) completion = 0;
int length_limited = !stream.cancelled && !stats.eos_stopped &&
+ request->max_tokens > 0 && /* a read-only request is not cut short */
stats.generated_tokens >= request->max_tokens;
double decode = stats.decode_sec > 0.0 ? stats.decode_sec : elapsed;
/* Trailing field: prompt tokens served from the previous turn's attention
@@ -16059,6 +16302,9 @@ int coli_v4_config_load(ColiDeepSeekV4Config *config, const char *model_dir,
#ifdef __AVX2__
#include
#endif
+#ifdef __ARM_NEON
+#include
+#endif
float coli_e8m0_decode(uint8_t value) {
if (value == 0xff) return NAN;
@@ -16424,6 +16670,87 @@ void coli_fp4_matmul_batch_rows16_order(float *y, const uint8_t *q4,
_mm256_storeu_ps(y + (int64_t)s * O + tile * 16 + 8, sum1[s]);
}
}
+#elif defined(__ARM_NEON)
+ /* NEON port of the AVX2 arm (issue #1696): same algorithm — vqtbl1q_u8
+ * nibble LUT decode of doubled e2m1 ints with an exact x0.5f un-double,
+ * 4x4 float transposes (vtrnq_f32 + vcombine_f32) making rows column-
+ * major, then strict (x*w)*scale rounding with separate mul/mul/add (no
+ * FMA fusion) so results stay bit-exact with the scalar and AVX2 arms.
+ * Tiles of 16 rows ride four float32x4_t lanes; one x broadcast serves
+ * all 4 columns of a group, keeping 4 independent add chains busy. */
+ #pragma omp parallel for schedule(static)
+ for (int64_t tile = 0; tile < O / 16; tile++) {
+ /* doubled e2m1 codes as uint8: 0,2,4,6,8,12,16,24 / 0,0xFE..0xE8 */
+ static const uint8_t lut2u[16] = {0,1,2,3,4,6,8,12,
+ 0,0xFF,0xFE,0xFD,0xFC,0xFA,0xF8,0xF4};
+ const uint8x16_t lut2 = vld1q_u8(lut2u);
+ const uint8x16_t m4 = vdupq_n_u8(0x0F);
+ const float32x4_t half = vdupq_n_f32(0.5f);
+ float32x4_t acc[4][128];
+ for (int g = 0; g < 4; g++)
+ for (int s = 0; s < S; s++) acc[g][s] = vdupq_n_f32(0.0f);
+ for (int base = 0; base < I; base += 32) {
+ float sc[16];
+ float32x4_t rowv[16][8];
+ for (int r = 0; r < 16; r++) {
+ int64_t row = tile * 16 + r;
+ sc[r] = e8lut[e8s[row * ng + base / 32]];
+ uint8x16_t by = vld1q_u8(q4 + row * rb + base / 2);
+ uint8x16_t lo = vandq_u8(by, m4);
+ uint8x16_t hi = vandq_u8(vshrq_n_u8(by, 4), m4);
+ uint8x16_t z0 = vzip1q_u8(lo, hi);
+ uint8x16_t z1 = vzip2q_u8(lo, hi);
+ int8x16_t n0 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z0));
+ int8x16_t n1 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z1));
+ int16x8_t sa = vmovl_s8(vget_low_s8(n0));
+ int16x8_t sb = vmovl_s8(vget_high_s8(n0));
+ int16x8_t sc16 = vmovl_s8(vget_low_s8(n1));
+ int16x8_t sd = vmovl_s8(vget_high_s8(n1));
+ rowv[r][0] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sa))), half);
+ rowv[r][1] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sa))), half);
+ rowv[r][2] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sb))), half);
+ rowv[r][3] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sb))), half);
+ rowv[r][4] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sc16))), half);
+ rowv[r][5] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sc16))), half);
+ rowv[r][6] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sd))), half);
+ rowv[r][7] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sd))), half);
+ }
+ for (int g = 0; g < 4; g++) {
+ const float32x4_t sc4 = vld1q_f32(sc + g * 4);
+ for (int k = 0; k < 8; k++) {
+ const float32x4_t a0 = rowv[g*4+0][k], a1 = rowv[g*4+1][k];
+ const float32x4_t a2 = rowv[g*4+2][k], a3 = rowv[g*4+3][k];
+ float32x4x2_t t0 = vtrnq_f32(a0, a1);
+ float32x4x2_t t1 = vtrnq_f32(a2, a3);
+ /* vcombine, not vzip: vzip yields lane order (r0,r2,r1,r3)
+ * which would swap columns 1<->2 of each group. */
+ const float32x4_t colv[4] = {
+ vcombine_f32(vget_low_f32(t0.val[0]), vget_low_f32(t1.val[0])),
+ vcombine_f32(vget_low_f32(t0.val[1]), vget_low_f32(t1.val[1])),
+ vcombine_f32(vget_high_f32(t0.val[0]), vget_high_f32(t1.val[0])),
+ vcombine_f32(vget_high_f32(t0.val[1]), vget_high_f32(t1.val[1])),
+ };
+ for (int s = 0; s < S; s++) {
+ const float xs0 = x[(int64_t)s * I + base + k * 4 + 0];
+ const float xs1 = x[(int64_t)s * I + base + k * 4 + 1];
+ const float xs2 = x[(int64_t)s * I + base + k * 4 + 2];
+ const float xs3 = x[(int64_t)s * I + base + k * 4 + 3];
+ float32x4_t xw0 = vmulq_f32(vdupq_n_f32(xs0), colv[0]);
+ float32x4_t xw1 = vmulq_f32(vdupq_n_f32(xs1), colv[1]);
+ float32x4_t xw2 = vmulq_f32(vdupq_n_f32(xs2), colv[2]);
+ float32x4_t xw3 = vmulq_f32(vdupq_n_f32(xs3), colv[3]);
+ acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw0, sc4));
+ acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw1, sc4));
+ acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw2, sc4));
+ acc[g][s] = vaddq_f32(acc[g][s], vmulq_f32(xw3, sc4));
+ }
+ }
+ }
+ }
+ for (int s = 0; s < S; s++)
+ for (int g = 0; g < 4; g++)
+ vst1q_f32(y + (int64_t)s * O + tile * 16 + g * 4, acc[g][s]);
+ }
#else
#pragma omp parallel for schedule(static)
for (int o = 0; o < O; o++) {
@@ -16546,6 +16873,70 @@ void coli_fp4_matvec_rows16_order(float *y, const uint8_t *q4,
_mm256_storeu_ps(y + tile * 16, sum0);
_mm256_storeu_ps(y + tile * 16 + 8, sum1);
}
+#elif defined(__ARM_NEON)
+ /* NEON port of the batch-one arm above: 4 row-groups of 4 rows so the
+ * accumulators stay float32x4_t registers for the whole row tile.
+ * Same doubled-int LUT decode and column-ascending (x*w)*scale-then-add
+ * order as the rows16 kernels — bit-exact vs the scalar arm below. */
+ #pragma omp parallel for schedule(static)
+ for (int64_t tile = 0; tile < O / 16; tile++) {
+ static const uint8_t lut2u[16] = {0,1,2,3,4,6,8,12,
+ 0,0xFF,0xFE,0xFD,0xFC,0xFA,0xF8,0xF4};
+ const uint8x16_t lut2 = vld1q_u8(lut2u);
+ const uint8x16_t m4 = vdupq_n_u8(0x0F);
+ const float32x4_t half = vdupq_n_f32(0.5f);
+ float32x4_t acc[4];
+ for (int g = 0; g < 4; g++) acc[g] = vdupq_n_f32(0.0f);
+ for (int base = 0; base < I; base += 32) {
+ float sc[16];
+ float32x4_t rowv[16][8];
+ for (int r = 0; r < 16; r++) {
+ int64_t row = tile * 16 + r;
+ sc[r] = e8lut[e8s[row * ng + base / 32]];
+ uint8x16_t by = vld1q_u8(q4 + row * rb + base / 2);
+ uint8x16_t lo = vandq_u8(by, m4);
+ uint8x16_t hi = vandq_u8(vshrq_n_u8(by, 4), m4);
+ uint8x16_t z0 = vzip1q_u8(lo, hi);
+ uint8x16_t z1 = vzip2q_u8(lo, hi);
+ int8x16_t n0 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z0));
+ int8x16_t n1 = vreinterpretq_s8_u8(vqtbl1q_u8(lut2, z1));
+ int16x8_t sa = vmovl_s8(vget_low_s8(n0));
+ int16x8_t sb = vmovl_s8(vget_high_s8(n0));
+ int16x8_t sc16 = vmovl_s8(vget_low_s8(n1));
+ int16x8_t sd = vmovl_s8(vget_high_s8(n1));
+ rowv[r][0] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sa))), half);
+ rowv[r][1] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sa))), half);
+ rowv[r][2] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sb))), half);
+ rowv[r][3] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sb))), half);
+ rowv[r][4] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sc16))), half);
+ rowv[r][5] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sc16))), half);
+ rowv[r][6] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_low_s16(sd))), half);
+ rowv[r][7] = vmulq_f32(vcvtq_f32_s32(vmovl_s16(vget_high_s16(sd))), half);
+ }
+ for (int g = 0; g < 4; g++) {
+ const float32x4_t sc4 = vld1q_f32(sc + g * 4);
+ for (int k = 0; k < 8; k++) {
+ const float32x4_t a0 = rowv[g*4+0][k], a1 = rowv[g*4+1][k];
+ const float32x4_t a2 = rowv[g*4+2][k], a3 = rowv[g*4+3][k];
+ float32x4x2_t t0 = vtrnq_f32(a0, a1);
+ float32x4x2_t t1 = vtrnq_f32(a2, a3);
+ const float32x4_t colv[4] = {
+ vcombine_f32(vget_low_f32(t0.val[0]), vget_low_f32(t1.val[0])),
+ vcombine_f32(vget_low_f32(t0.val[1]), vget_low_f32(t1.val[1])),
+ vcombine_f32(vget_high_f32(t0.val[0]), vget_high_f32(t1.val[0])),
+ vcombine_f32(vget_high_f32(t0.val[1]), vget_high_f32(t1.val[1])),
+ };
+ for (int ci = 0; ci < 4; ci++) {
+ const float xc = x[base + k * 4 + ci];
+ acc[g] = vaddq_f32(acc[g], vmulq_f32(
+ vmulq_f32(vdupq_n_f32(xc), colv[ci]), sc4));
+ }
+ }
+ }
+ }
+ for (int g = 0; g < 4; g++)
+ vst1q_f32(y + tile * 16 + g * 4, acc[g]);
+ }
#else
#pragma omp parallel for schedule(static)
for (int o = 0; o < O; o += 4) {
diff --git a/c/deepseek_v4.h b/c/deepseek_v4.h
index d57d1ab62..5b1ad3756 100644
--- a/c/deepseek_v4.h
+++ b/c/deepseek_v4.h
@@ -124,12 +124,31 @@ typedef struct {
* uninterruptible as before. */
typedef int (*ColiV4SessionAbortFn)(void *user_data);
+/* The numeric channel (SUBMIT logprobs=k, docs/brio.md): raw head scores,
+ * vocab_size floats, valid only for the duration of the callback. on_echo
+ * fires once per prompt position whose predictor this call computed, with the
+ * token that actually stands there; on_scores fires right before the on_token
+ * of every generated token, with the scores it was picked from. */
+typedef void (*ColiV4SessionScoresFn)(void *user_data, int position, int token,
+ const float *scores, int vocab);
+
typedef struct {
- int max_new_tokens; /* required; clamped by session cap */
+ int max_new_tokens; /* required; clamped by session cap; 0 only with logprobs > 0 */
int stop_at_sentence;
int no_dspark; /* disable speculative draft/verification */
ColiV4SessionAbortFn should_abort; /* optional prefill abort poll */
void *abort_user_data;
+ /* SUBMIT logprobs=k / pin=1. logprobs > 0 opens the channel above, turns
+ * speculative decoding off for the request (a draft accepted in a block
+ * has no scores of its own) and allows max_new_tokens == 0, "read the
+ * prompt and stop". pin keeps the prompt-end scores and a snapshot of the
+ * attention state, so the prompts that extend this one start from here
+ * with their first fresh token's predictor intact. */
+ int logprobs;
+ int pin;
+ ColiV4SessionScoresFn on_echo;
+ ColiV4SessionScoresFn on_scores;
+ void *scores_user_data;
/* Optional: byte length of the prompt's stable leading prefix (the
* rendered system turn). The session snapshots the attention state at
* that token boundary during this prefill so later conversations that
diff --git a/c/deepseek_v41.c b/c/deepseek_v41.c
index e4f31b28f..93860212a 100644
--- a/c/deepseek_v41.c
+++ b/c/deepseek_v41.c
@@ -3522,6 +3522,16 @@ static int serve_eos(Model *m, const char *snap, int *ids, int cap) {
return n;
}
+/* max_tokens is a ceiling, as on V4 and Qwen (#1641). A score-only
+ * request may fill the context; generation needs at least one free position. */
+static int serve_budget(int prompt, int requested, int context, int logprobs) {
+ if (prompt < 1 || prompt > context) return -1;
+ int budget = requested > 0 ? requested : (logprobs > 0 ? 0 : 256);
+ int room = context - prompt;
+ if (budget > 0 && room == 0) return -1;
+ return budget < room ? budget : room;
+}
+
static void serve_loop(Model *m, Tok *tokenizer, const char *snap) {
Cfg *c = &m->c;
coli_serve_stdio_init();
@@ -3530,7 +3540,9 @@ static void serve_loop(Model *m, Tok *tokenizer, const char *snap) {
coli_serve_write_ready(stdout, rss_gb());
serve_emap(m);
float *logits = xmalloc((size_t)c->vocab * sizeof(float), "logits");
- int *ids = xmalloc((size_t)c->max_positions * sizeof(int), "prompt ids");
+ /* tok_encode stops at its output capacity: one extra id distinguishes
+ * a full, valid read-only prompt from a silently truncated one. */
+ int *ids = xmalloc(((size_t)c->max_positions + 1) * sizeof(int), "prompt ids");
float *pending_image = NULL;
int pending_h = 0, pending_w = 0;
@@ -3586,7 +3598,22 @@ static void serve_loop(Model *m, Tok *tokenizer, const char *snap) {
mir_reads0[r] = g_mir_nread[r];
}
int n_prompt = tok_encode(tokenizer, (const char *)command.payload,
- (int)command.payload_bytes, ids, c->max_positions);
+ (int)command.payload_bytes, ids, c->max_positions + 1);
+ int budget = serve_budget(n_prompt, command.max_tokens, c->max_positions,
+ command.logprobs);
+ if (budget < 0) {
+ char message[128];
+ snprintf(message, sizeof(message),
+ "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d",
+ n_prompt, command.max_tokens, c->max_positions);
+ coli_serve_write_error(stdout, command.id,
+ n_prompt < 1 ? "EMPTY_PROMPT" : message);
+ coli_serve_command_dispose(&command); continue;
+ }
+ if (command.max_tokens > budget)
+ fprintf(stderr, "[serve] max_tokens %d clamped to %d (context %d - prompt %d); "
+ "raise CTX for longer answers\n",
+ command.max_tokens, budget, c->max_positions, n_prompt);
/* Decided BEFORE the reset, because the reset is what it decides about.
* A chat client resends the whole transcript every turn; if this prompt
* begins with the ids the state was built from, that state already IS
@@ -3634,23 +3661,6 @@ static void serve_loop(Model *m, Tok *tokenizer, const char *snap) {
fflush(stderr);
}
if (!reuse) model_reset(m);
- if (n_prompt < 1) {
- coli_serve_write_error(stdout, command.id, "EMPTY_PROMPT");
- coli_serve_command_dispose(&command); continue;
- }
- /* max_tokens=0 con logprobs>0 vuol dire "leggi e fermati": non e un
- * valore mancante da rimpiazzare con un default, ed e' proprio il caso
- * in cui un menu chiuso non vuole pagare un passo di decodifica per
- * opzione. Senza logprobs 0 resta "non specificato" -> 256, come prima. */
- int budget = command.max_tokens > 0 ? command.max_tokens
- : (command.logprobs > 0 ? 0 : 256);
- if (n_prompt + budget > c->max_positions) {
- char message[128];
- snprintf(message, sizeof(message), "CONTEXT_EXCEEDED %d %d",
- n_prompt + budget, c->max_positions);
- coli_serve_write_error(stdout, command.id, message);
- coli_serve_command_dispose(&command); continue;
- }
coli_serve_write_accept(stdout, command.id, n_prompt);
float *aligned = NULL;
uint8_t *image_mask = NULL;
diff --git a/c/deepseek_v4_internal.h b/c/deepseek_v4_internal.h
index 2bcdf9cde..9686a56ac 100644
--- a/c/deepseek_v4_internal.h
+++ b/c/deepseek_v4_internal.h
@@ -966,6 +966,20 @@ struct ColiV4Session {
uint64_t spec_drafted;
uint64_t spec_accepted;
int spec_disabled;
+ /* Prompt-end capture for SUBMIT pin=1: the ids fed and the head scores
+ * that predict the token after them. A later prompt that starts with
+ * exactly these ids gets its first fresh token's predictor from here; that
+ * token is the one a closed-set caller asks about (docs/brio.md). The
+ * attention state itself goes to a v4_ckpt slot; this is the part the
+ * snapshot does not hold. */
+ int *pin_ids;
+ int pin_len;
+ float *pin_scores;
+ /* Scratch for the numeric channel, one hidden row and one row of head
+ * scores, allocated on first use and freed with the session so the many
+ * early returns of generate() leave nothing behind. */
+ float *echo_hidden;
+ float *echo_scores;
};
/* RAM-tiered expert open used by coli_v4_engine_open (replaces ld --wrap).
@@ -1002,6 +1016,12 @@ extern void (*coli_v4_test_expert_wait_hook)(ColiExpertKey key);
extern uint64_t coli_v4_test_fp4_batch_calls;
extern uint64_t coli_v4_test_expert_victim_probes;
int coli_v4_test_expert_slot_index(ColiExpertStore *store, ColiExpertKey key);
+void coli_v4_test_reset_direct_io_stats(void);
+uint64_t coli_v4_test_direct_reads(void);
+uint64_t coli_v4_test_direct_fallbacks(void);
+/* Point missing O_DIRECT twins at a dup of the buffered fd so tests can
+ * exercise the direct-window path on filesystems that refuse O_DIRECT. */
+int coli_v4_test_force_streaming_direct(ColiExpertStore *store);
ColiV4Session *coli_v4_test_session_bare_create(ColiV4Engine *engine);
void coli_v4_test_session_bare_destroy(ColiV4Session *session);
diff --git a/c/doctor.py b/c/doctor.py
index b8fc8783e..9b7045bb6 100644
--- a/c/doctor.py
+++ b/c/doctor.py
@@ -457,6 +457,26 @@ def deep_container_report(model, mirror_dir=None):
}
+def windows_backend_dll(image):
+ """Which GPU backend DLL a Windows host compiled in, or None if CPU-only.
+
+ backend_loader.c bakes exactly one basename: coli_hip.dll under COLI_HIP_DLL
+ and coli_cuda.dll otherwise. That string is the build marker. The GLM/Qwen
+ banner "[CUDA] mode: routed experts" is only printed by those two engines;
+ a Kimi K3 CUDA_DLL host links the same loader and prints [K3-CUDA] instead.
+ DeepSeek V4 has its own pair and is not this function's job.
+ """
+ if not image or b"[DSV4 CUDA]" in image:
+ return None
+ if b"coli_hip.dll" in image:
+ return "coli_hip.dll"
+ if b"coli_cuda.dll" in image:
+ return "coli_cuda.dll"
+ if b"[CUDA] mode: routed experts" in image or b"[K3-CUDA]" in image:
+ return "coli_cuda.dll"
+ return None
+
+
def cuda_linkage(engine_path):
"""Return CUDA linkage state without loading the executable or CUDA runtime."""
engine = Path(engine_path)
@@ -484,17 +504,11 @@ def cuda_linkage(engine_path):
if sys.platform == "win32":
# Windows DLL-split builds never link the GPU runtime directly: the host
# LoadLibrary's its backend at runtime (backend_loader.c), so there's no
- # import-table entry for ldd/dumpbin to see. Detect the GPU build via a
- # marker string baked into the engine's #ifdef COLI_CUDA block, then
- # require the backend artifact to sit next to the executable.
- #
- # WHICH artifact is not a guess. backend_loader.c compiles exactly one
- # basename into the host -- COLI_BACKEND_DLL is "coli_hip.dll" under
- # COLI_HIP_DLL and "coli_cuda.dll" otherwise -- so the binary states
- # what it will load and we check for that. Asking for coli_cuda.dll
- # unconditionally failed a working HIP host (a hard error, not a
- # warning), and accepting either name would have passed a HIP host that
- # only had a stray CUDA backend beside it.
+ # import-table entry for ldd/dumpbin to see. Detect the GPU build from
+ # the backend basename compiled into the host, then require that file
+ # next to the executable. Asking for coli_cuda.dll unconditionally
+ # failed a working HIP host (a hard error, not a warning), and requiring
+ # the GLM routed-experts banner missed every Kimi K3 CUDA_DLL build.
try:
image = engine.read_bytes()
except OSError:
@@ -506,10 +520,7 @@ def cuda_linkage(engine_path):
present = any((engine.parent / name).is_file()
for name in ("coli_cuda_dsv4_dg.dll", "coli_cuda_dsv4.dll"))
return {"linked": present, "missing": not present}
- if b"[CUDA] mode: routed experts" not in image:
- return {"linked": False, "missing": False}
- expected = next((name for name in ("coli_hip.dll", "coli_cuda.dll")
- if name.encode() in image), None)
+ expected = windows_backend_dll(image)
if expected is None:
return {"linked": False, "missing": False}
dll_present = (engine.parent / expected).is_file()
diff --git a/c/exact_dot.h b/c/exact_dot.h
new file mode 100644
index 000000000..bc40f1692
--- /dev/null
+++ b/c/exact_dot.h
@@ -0,0 +1,155 @@
+/* exact_dot.h — order-independent dot products for the opt-in exact verify mode.
+ *
+ * Every product is formed exactly (integer mantissas, never rounded) and accumulated by
+ * exponent bin; the bins are folded into one wide two's-complement integer and rounded to
+ * double ONCE, correctly (round-to-nearest-even), then to float. No summation order, no
+ * contraction flag and no SIMD width can change the result, so CPU and GPU agree to the bit
+ * by construction and near-ties are decided identically everywhere. The cost is throughput:
+ * an integer path, no FMA, no tensor cores. That is why it is opt-in and verify-only.
+ *
+ * Covers the two shapes the engine's verify rows need:
+ * exd_add_ff(acc, a, b) a*b for f32 a, b (attention q.k, p.v over f32 rows)
+ * exd_add_wsx(acc, w, s, x) (w*s)*x for int w, f32 s, x (quantised weight rows: int4/int8
+ * * per-row/per-group scale * f32 act)
+ * Products are exact int64 (24+24 or 8+24+24 mantissa bits <= 56 bits); bins hold __int128
+ * partial sums (72 bits of headroom), so any n up to 2^72 products per bin is safe.
+ *
+ * NaN / inf: a float dot with a NaN input is NaN, with an inf input is +-inf or NaN; the exact
+ * path reproduces the same *classification* (flags), so callers see the same special values.
+ *
+ * Copyright (c) 2026 Anomly, Inc. Licensed under the same terms as colibri (see LICENSE).
+ * Author: Ry Bruscoe. */
+#ifndef COLI_EXACT_DOT_H
+#define COLI_EXACT_DOT_H
+#include
+#include
+#include
+
+#if defined(__SIZEOF_INT128__)
+typedef __int128 exd_i128;
+#else
+#error "exact_dot.h needs a 128-bit integer type (GCC/Clang); the exact verify mode is unavailable on this compiler"
+#endif
+
+/* Product exponents: f32 mantissa m (24 bits, value m*2^e with e = exp-150), normals e in
+ * [-149, 104]; products e in [-298, 208]; with an 8-bit integer weight (value w) the exponent is
+ * the same range. Bin index = e + EXD_BIAS. */
+#define EXD_BIAS 320
+#define EXD_NBIN 640
+/* wide accumulator: EXD_NBIN + 64 bits of headroom, in 64-bit limbs (two's complement) */
+#define EXD_LIMBS 12 /* 768 bits */
+
+typedef struct {
+ exd_i128 bin[EXD_NBIN];
+ int lo, hi; /* used bin range (inclusive); lo > hi means empty */
+ int nan, pinf, ninf; /* special-value flags, order-independent by construction */
+} exd_acc;
+
+static inline void exd_init(exd_acc *a){ a->lo = EXD_NBIN; a->hi = -1; a->nan = a->pinf = a->ninf = 0; }
+
+/* bins are zeroed lazily as the used range [lo, hi] grows (a full memset would cost more than a
+ * typical 512-element dot); returns the bin to add into */
+static inline exd_i128 *exd_touch(exd_acc *a, int idx){
+ if(a->lo > a->hi){ a->bin[idx] = 0; a->lo = a->hi = idx; return &a->bin[idx]; }
+ if(idx < a->lo){ for(int i = idx; i < a->lo; i++) a->bin[i] = 0; a->lo = idx; }
+ else if(idx > a->hi){ for(int i = a->hi + 1; i <= idx; i++) a->bin[i] = 0; a->hi = idx; }
+ return &a->bin[idx];
+}
+
+/* decode f32 into (signed integer mantissa, exponent) with value = m * 2^e; returns 0 for
+ * zero (m=0), 1 for finite non-zero, 2 for inf, 3 for nan. */
+static inline int exd_decode(float f, int64_t *m, int *e){
+ uint32_t u; memcpy(&u, &f, 4);
+ int s = (int)(u >> 31), ex = (int)((u >> 23) & 0xFF); uint32_t fr = u & 0x7FFFFF;
+ if(ex == 0xFF){ *m = 0; *e = 0; return fr ? 3 : 2; }
+ if(ex == 0){ if(!fr){ *m = 0; *e = 0; return 0; } *m = s ? -(int64_t)fr : (int64_t)fr; *e = -149; return 1; }
+ int64_t mm = (int64_t)(fr | 0x800000); *m = s ? -mm : mm; *e = ex - 150; return 1;
+}
+
+static inline void exd_special(exd_acc *a, int ca, int cb, int64_t ma, int64_t mb, float fa, float fb){
+ if(ca == 3 || cb == 3){ a->nan = 1; return; }
+ if(ca == 2 || cb == 2){
+ /* inf * 0 = nan; inf * x has the sign of the product */
+ if((ca == 2 && (cb == 0)) || (cb == 2 && (ca == 0))){ a->nan = 1; return; }
+ int sa = signbit(fa) ? 1 : 0, sb = signbit(fb) ? 1 : 0; (void)ma; (void)mb;
+ if(sa ^ sb) a->ninf = 1; else a->pinf = 1;
+ }
+}
+
+static inline void exd_add_ff(exd_acc *a, float fa, float fb){
+ int64_t ma, mb; int ea, eb;
+ int ca = exd_decode(fa, &ma, &ea), cb = exd_decode(fb, &mb, &eb);
+ if(ca > 1 || cb > 1){ exd_special(a, ca, cb, ma, mb, fa, fb); return; }
+ if(ca == 0 || cb == 0) return;
+ *exd_touch(a, ea + eb + EXD_BIAS) += (exd_i128)ma * mb;
+}
+
+/* (w * s) * x with an exact integer w (|w| < 2^8), f32 scale s, f32 activation x */
+static inline void exd_add_wsx(exd_acc *a, int w, float s, float x){
+ if(w == 0) return;
+ int64_t ms, mx; int es, ex;
+ int cs = exd_decode(s, &ms, &es), cx = exd_decode(x, &mx, &ex);
+ if(cs > 1 || cx > 1){ exd_special(a, cs, cx, ms, mx, s, x); return; }
+ if(cs == 0 || cx == 0) return;
+ *exd_touch(a, es + ex + EXD_BIAS) += (exd_i128)w * ms * mx; /* <= 8 + 24 + 24 bits: exact in int128 */
+}
+
+/* ---- fold bins into a 768-bit two's-complement integer scaled by 2^(lo - EXD_BIAS) ---- */
+static inline void exd_wide_add_shifted(uint64_t *w, exd_i128 v, int shift){
+ /* add v * 2^shift into w (12 limbs, little-endian, two's complement), 0 <= shift < 704 */
+ uint64_t t[EXD_LIMBS]; const uint64_t sx = (v < 0) ? ~(uint64_t)0 : 0;
+ for(int i = 0; i < EXD_LIMBS; i++) t[i] = sx;
+ t[0] = (uint64_t)v; t[1] = (uint64_t)(v >> 64);
+ int limbs = shift >> 6, bits = shift & 63;
+ if(bits){ uint64_t prev = 0; for(int i = 0; i < EXD_LIMBS; i++){ uint64_t cur = t[i]; t[i] = (cur << bits) | prev; prev = cur >> (64 - bits); } }
+ if(limbs){ for(int i = EXD_LIMBS - 1; i >= 0; i--) t[i] = (i - limbs >= 0) ? t[i - limbs] : 0; }
+ unsigned __int128 carry = 0;
+ for(int i = 0; i < EXD_LIMBS; i++){ unsigned __int128 sum = (unsigned __int128)w[i] + t[i] + carry; w[i] = (uint64_t)sum; carry = sum >> 64; }
+}
+
+/* correctly rounded (nearest-even) conversion of a two's-complement 768-bit integer * 2^e2 */
+static inline double exd_wide_to_double(const uint64_t *w, int e2){
+ uint64_t m[EXD_LIMBS]; memcpy(m, w, sizeof m);
+ int neg = (m[EXD_LIMBS - 1] >> 63) & 1;
+ if(neg){ unsigned __int128 c = 1; for(int i = 0; i < EXD_LIMBS; i++){ unsigned __int128 v = (unsigned __int128)(~m[i]) + c; m[i] = (uint64_t)v; c = v >> 64; } }
+ int top = -1;
+ for(int i = EXD_LIMBS - 1; i >= 0 && top < 0; i--) if(m[i]){ int b = 63; while(!((m[i] >> b) & 1)) b--; top = i * 64 + b; }
+ if(top < 0) return 0.0;
+ /* take the top 54 bits (53 + round bit) and a sticky bit for the rest */
+ int hi = top, lo = top - 53; /* bits [lo, hi] = 54 bits */
+ uint64_t bits54 = 0; int sticky = 0;
+ for(int b = hi; b >= lo; b--){
+ int v = (b >= 0) ? (int)((m[b >> 6] >> (b & 63)) & 1) : 0;
+ bits54 = (bits54 << 1) | (uint64_t)v;
+ }
+ for(int i = 0; i < EXD_LIMBS && !sticky; i++){
+ int base = i * 64;
+ if(base + 63 < lo){ if(m[i]) sticky = 1; continue; }
+ for(int b = base; b < base + 64 && b < lo; b++) if((m[i] >> (b - base)) & 1){ sticky = 1; break; }
+ }
+ uint64_t mant = bits54 >> 1; int round = (int)(bits54 & 1);
+ if(round && (sticky || (mant & 1))) mant += 1;
+ /* mant may have become 2^53: ldexp handles it exactly */
+ double d = ldexp((double)mant, lo + 1 + e2); /* mant = bits [lo+1, hi] */
+ return neg ? -d : d;
+}
+
+static inline double exd_finish_double(const exd_acc *a){
+ if(a->nan || (a->pinf && a->ninf)) return NAN;
+ if(a->pinf) return INFINITY;
+ if(a->ninf) return -INFINITY;
+ if(a->lo > a->hi) return 0.0;
+ uint64_t w[EXD_LIMBS]; memset(w, 0, sizeof w);
+ for(int i = a->lo; i <= a->hi; i++) if(a->bin[i]) exd_wide_add_shifted(w, a->bin[i], i - a->lo);
+ return exd_wide_to_double(w, a->lo - EXD_BIAS);
+}
+
+static inline float exd_finish(const exd_acc *a){ return (float)exd_finish_double(a); }
+
+/* convenience: exact f32 dot of length n */
+static inline float exd_dot_ff(const float *x, const float *y, int n){
+ exd_acc a; exd_init(&a);
+ for(int i = 0; i < n; i++) exd_add_ff(&a, x[i], y[i]);
+ return exd_finish(&a);
+}
+#endif /* COLI_EXACT_DOT_H */
diff --git a/c/family_registry.py b/c/family_registry.py
index 7b3c2e364..d1fb3877d 100644
--- a/c/family_registry.py
+++ b/c/family_registry.py
@@ -702,7 +702,7 @@ def _dsv4_geometry(config, context, _model_dir):
"""
layers = _required_int(config, "num_hidden_layers", "deepseek_v4")
experts = _required_int(config, "n_routed_experts", "deepseek_v4")
- hidden = _required_int(config, "hidden_size", "deepseek_v4")
+ _required_int(config, "hidden_size", "deepseek_v4")
heads = _required_int(config, "num_attention_heads", "deepseek_v4")
head_dim = _required_int(config, "head_dim", "deepseek_v4")
q_rank = _required_int(config, "q_lora_rank", "deepseek_v4")
@@ -1268,6 +1268,10 @@ def _qwen38_resident_inventory(name, size, _config, dtype=None):
# links NOCUDA_LDFLAGS. Left at the default this advertised a VRAM tier.
supports_accelerator=False,
expert_inventory=_individual_expert_inventory(_GLM_EXPERT),
+ # coli convert routes to convert_olmoe_merged.py (d4d11ef dispatch);
+ # the converter takes no precision flags (--ebits / --group-size etc.).
+ converter="convert_olmoe_merged.py",
+ converter_accepts=(),
config_section="root",
# implicit_cap 0, not 8: the engine sizes its expert cache from the RAM
# budget once the dense weights are resident (#1443), so "nobody chose a
diff --git a/c/fp8_format.h b/c/fp8_format.h
new file mode 100644
index 000000000..d10ea461e
--- /dev/null
+++ b/c/fp8_format.h
@@ -0,0 +1,35 @@
+/* fmt=8 (fp8-e4m3-b128) block geometry -- the single definition site for the
+ * 128x128 scale-block edge, shared by the CPU side (quant.h: matmul_fp8,
+ * e4m3 dequant plumbing; colibri.c: qt_addrow/qt_matvec_rows, qt_from_disk)
+ * and the CUDA backend's three converted sites (backend_cuda.cu:
+ * absorb_scale, quant_matmul's fmt=8 branch, the upload-time ng/scale_count
+ * computation); the f8-warp and f8-group kernels there keep their own
+ * 128 / >>7 literals, pinned against drift by backend_cuda.cu's
+ * static_assert(FP8_BLOCK == 128). Before this header the two backends
+ * agreed by IDENTICAL LITERALS restated in each file, so an edit to one side
+ * could not break the other side's build -- only a full-scale CPU-vs-CUDA
+ * parity run would have noticed. Kept deliberately tiny (no LUTs, no
+ * functions with OpenMP pragmas, no intrinsics) so the CUDA translation unit
+ * can include it without dragging in quant.h.
+ *
+ * The in-repo FP8 tests are NOT independent of this constant: the reference
+ * decoders in test_fp8_passthrough.c, test_fp8_load.c, test_fp8_e2e_loader.c,
+ * test_qwen38_native_weights.c, test_qt_addrow.c and test_shard_kvb_refuse.c
+ * all index with FP8_BLOCK/fp8_nblk via quant.h, so they move in lockstep
+ * with an edit here (test_backend_metal.mm's ref_fp8_nblk and
+ * test_backend_cuda.cu's fmt=8 reference keep their own literals, on
+ * purpose). An edit to FP8_BLOCK is a format change, not a tunable -- those
+ * tests cannot catch a wrong edit on their own. */
+#ifndef COLI_FP8_FORMAT_H
+#define COLI_FP8_FORMAT_H
+
+#include
+
+#define FP8_BLOCK 128
+
+/* Blocks covering n elements: ceil(n/FP8_BLOCK). Host-side helper (device
+ * code uses the FP8_BLOCK macro arithmetic directly -- this is not decorated
+ * for device compilation on purpose, to keep the header plain C). */
+static inline int64_t fp8_nblk(int n){ return ((int64_t)n + FP8_BLOCK - 1) / FP8_BLOCK; }
+
+#endif /* COLI_FP8_FORMAT_H */
diff --git a/c/glm53.c b/c/glm53.c
index 8f33d4cd9..ac34e953b 100644
--- a/c/glm53.c
+++ b/c/glm53.c
@@ -971,7 +971,7 @@ static void mv(float *out, const Mat *w, const float *x) {
}
#endif
#ifdef COLI_VULKAN
- if (g_vk_ready && (w->fmt == 1 || w->fmt == 4)) {
+ if (g_vk_ready && w->resident && (w->fmt == 1 || w->fmt == 4)) {
Mat *mutable_w = (Mat *)w;
if (coli_vk_matmul((ColiVkTensor **)&mutable_w->vk, out, x,
w->fmt == 4 ? (const void *)w->q4 : (const void *)w->q8,
@@ -1277,15 +1277,6 @@ static void expert_table_init(GModel *m) {
/* Quanti slot per layer: il budget diviso i layer sparsi. Il pavimento e' 1 e
* non topk, perche' un pavimento a topk impegnerebbe topk*layer slot comunque,
* cioe' molti GB, a dispetto del budget chiesto. */
-/* Quanta RAM il sistema dice di poter dare adesso. MemAvailable e non MemFree:
- * la seconda ignora la page cache riutilizzabile e farebbe stimare molto meno
- * di quello che c'e'. */
-static double memory_available_gb(void) {
- /* #1375: era una lettura di /proc/meminfo, che su Windows e macOS non
- * esiste: 0 -> budget 1 GB -> uno slot per layer, in silenzio. */
- return compat_mem_available_gb();
-}
-
static void expert_cache_init(GModel *m) {
const Cfg *c = &m->c;
const char *setting = getenv("GLM53_EXPERT_GB");
@@ -1295,19 +1286,39 @@ static void expert_cache_init(GModel *m) {
* versi -- su una macchina piccola va in OOM, su una grande lascia RAM
* inutilizzata mentre il disco fa tutto il lavoro, che e' esattamente
* quello che e' successo alla prima esecuzione vera. */
+ const int from = c->first_dense > m->layer_begin ? c->first_dense : m->layer_begin;
+ int sparse = m->layer_end - from;
+ if (sparse < 0) sparse = 0;
double budget;
if (setting) budget = atof(setting);
else {
- const double free_now = memory_available_gb();
- budget = free_now - 3.0;
+ /* MemAvailable counts reclaimable page cache as free. Sizing this LRU
+ * from it means allocating, as anonymous memory, the very pages the
+ * next slot miss would have been served from: both caches hold the
+ * same bytes, the model is paid for twice, and the cheap copy is the
+ * one that loses. So leave the model room to stay in page cache and
+ * take only what is left over. */
+ double total = 0.0, free_now = 0.0;
+ compat_meminfo_gb(&total, &free_now); /* one pass, both fields */
+ /* the routed experts dominate; the rest of the weights are ~7% */
+ const double model_gb = (double)sparse * c->n_experts * (double)m->e_slot / 1e9 * 1.07;
+ double margin = total * 0.08;
+ if (margin < 4.0) margin = 4.0;
+ if (total <= 0.0 || model_gb >= total - margin) {
+ /* The model does not fit in RAM anyway, so page cache cannot help:
+ * keep as many slots as possible, exactly as before. `total <= 0`
+ * is "could not measure" and takes the same safe path. */
+ budget = free_now - 3.0;
+ } else {
+ budget = total - model_gb - margin;
+ if (budget > free_now - 3.0) budget = free_now - 3.0;
+ }
if (budget < 1.0) budget = 1.0;
if (getenv("GLM53_VERBOSE"))
- fprintf(stderr, "expert budget: %.1f GB (%.1f available, 3 GB reserved)\n",
- budget, free_now);
+ fprintf(stderr, "expert budget: %.1f GB (%.1f total, %.1f model, "
+ "%.1f margin, %.1f available)\n",
+ budget, total, model_gb, margin, free_now);
}
- const int from = c->first_dense > m->layer_begin ? c->first_dense : m->layer_begin;
- int sparse = m->layer_end - from;
- if (sparse < 0) sparse = 0;
int cap = (int)((budget * 1e9) / ((double)m->e_slot * (sparse > 0 ? sparse : 1)));
if (g_cap_override > 0) cap = g_cap_override; /* scelta esplicita: vince */
if (cap < 1) cap = 1;
@@ -2153,7 +2164,7 @@ static void model_load_range(GModel *m, const char *dir, int layer_begin,
else snprintf(spv, sizeof(spv), "%s/qmatmul.spv", given ? given : "shaders");
g_vk_ready = coli_vk_init(spv) && coli_vk_available();
if (g_vk_ready) coli_vk_set_swiglu_limit(m->c.swiglu_limit);
- fprintf(stderr, g_vk_ready)
+ fprintf(stderr, g_vk_ready
? "Vulkan: active for resident matrices\n"
: "Vulkan: no usable device (%s), falling back to CPU\n", spv);
}
@@ -2376,6 +2387,9 @@ static float *run_layers(GModel *m, GSession *s, float *streams, float *next,
static void mat_release(Mat *mat) {
#ifdef COLI_METAL
if (mat->metal) coli_metal_tensor_free((ColiMetalTensor *)mat->metal);
+#endif
+#ifdef COLI_VULKAN
+ if (mat->vk) coli_vk_tensor_free((ColiVkTensor *)mat->vk);
#endif
free((void *)mat->f); free((void *)mat->q8);
free((void *)mat->q4); free((void *)mat->s);
diff --git a/c/gsgemv.h b/c/gsgemv.h
new file mode 100644
index 000000000..fede49259
--- /dev/null
+++ b/c/gsgemv.h
@@ -0,0 +1,212 @@
+#ifndef COLIBRI_GSGEMV_H
+#define COLIBRI_GSGEMV_H
+/* Group-scaled int8 GEMV: one f32 scale per `gs` input elements per row, the
+ * layout the gs64 expert containers use. Row layout of `scale`: [O][I/gs]
+ * row-major. Its own header so tests/test_gsgemv.c can link the very kernel
+ * the engine runs; qwen36.c carries a main and cannot be linked into a test.
+ *
+ * This is qwen36's hottest kernel: every expert matmul goes through it, three
+ * per expert, eight experts, forty layers. Each row's float operations must
+ * stay in exactly this order -- float addition is not associative and the
+ * engine's token stream is required to be byte-identical to the reference.
+ * tests/test_gsgemv.c holds the pre-restructure kernel verbatim and compares
+ * raw float bits, so any reassociation fails there rather than surfacing as
+ * drifted text much later.
+ *
+ * The SSE4.1 tier below is the exception to that byte-identical rule, by
+ * design: it is new code with no pre-existing output to match, its tree
+ * reduction is a genuinely different shape than the scalar reference, and it
+ * is checked by tests/test_gsgemv.c against a tolerance, not memcmp. It
+ * routes its FMA and float loads through sse41_kernels.h -- the same shared
+ * primitives header olmoe.c uses -- rather than inlining its own copy. */
+#include
+#include
+#if (defined(__AVX2__) && defined(__FMA__)) || defined(__SSE4_1__)
+#include
+#endif
+#if defined(__SSE4_1__)
+#include "sse41_kernels.h"
+#endif
+
+#if defined(__AVX2__) && defined(__FMA__)
+/* The group reduction, lifted verbatim so the four-row body and the tail row
+ * cannot drift apart. Order of the adds is load-bearing, not stylistic. */
+static inline float gs_group_sum(__m256 a0, __m256 a1) {
+ a0 = _mm256_add_ps(a0, a1);
+ __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1));
+ s = _mm_add_ps(s, _mm_movehl_ps(s,s));
+ s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1));
+ return _mm_cvtss_f32(s);
+}
+#elif defined(__SSE4_1__)
+/* Same reduction as gs_group_sum above, one level shallower: a0/a1 are
+ * already the two four-lane halves, so there is no 256->128 fold first. */
+static inline float gs_group_sum_sse41(__m128 a0, __m128 a1) {
+ a0 = _mm_add_ps(a0, a1);
+ a0 = _mm_add_ps(a0, _mm_movehl_ps(a0,a0));
+ a0 = _mm_add_ss(a0, _mm_shuffle_ps(a0,a0,1));
+ return _mm_cvtss_f32(a0);
+}
+#endif
+
+static void matmul_q_gs(float *y, const float *x, const int8_t *q, const float *scale,
+ int I, int O, int gs) {
+ int ng = (I + gs - 1) / gs;
+#if defined(__AVX2__) && defined(__FMA__)
+ if ((gs & 31) == 0) {
+ /* Four output rows in flight at once. Each row keeps its own pair of
+ * accumulators, its own reduction and its own running acc, so its float
+ * operations happen in exactly the order the single-row loop used them
+ * -- the interleave is a schedule change, not an algebraic one.
+ *
+ * Why it pays: one group is 8 FMAs but a dependency chain of roughly 36
+ * cycles (accumulate, then the reduction tree, then the loop-carried
+ * acc +=), against a throughput floor near 4. The single-row loop stalls
+ * on that chain about nine times out of ten. Four independent rows fill
+ * the gaps, and the x block gets loaded once instead of four times. */
+ int o4 = O & ~3;
+ #pragma omp parallel for schedule(static) if(O >= 256)
+ for (int ob = 0; ob < o4; ob += 4) {
+ const int8_t *w0 = q + (int64_t)(ob+0) * I, *w1 = q + (int64_t)(ob+1) * I;
+ const int8_t *w2 = q + (int64_t)(ob+2) * I, *w3 = q + (int64_t)(ob+3) * I;
+ const float *sc0 = scale + (int64_t)(ob+0) * ng, *sc1 = scale + (int64_t)(ob+1) * ng;
+ const float *sc2 = scale + (int64_t)(ob+2) * ng, *sc3 = scale + (int64_t)(ob+3) * ng;
+ float acc0 = 0.f, acc1 = 0.f, acc2 = 0.f, acc3 = 0.f;
+ for (int gi = 0; gi < ng; gi++) {
+ __m256 a00 = _mm256_setzero_ps(), a01 = _mm256_setzero_ps();
+ __m256 a10 = _mm256_setzero_ps(), a11 = _mm256_setzero_ps();
+ __m256 a20 = _mm256_setzero_ps(), a21 = _mm256_setzero_ps();
+ __m256 a30 = _mm256_setzero_ps(), a31 = _mm256_setzero_ps();
+ int base = gi * gs, end = base + gs; if (end > I) end = I;
+ for (int i = base; i + 16 <= end; i += 16) {
+ __m256 xl = _mm256_loadu_ps(x+i), xh = _mm256_loadu_ps(x+i+8);
+ __m128i b0 = _mm_loadu_si128((const __m128i*)(w0 + i));
+ a00 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a00);
+ a01 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a01);
+ __m128i b1 = _mm_loadu_si128((const __m128i*)(w1 + i));
+ a10 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a10);
+ a11 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a11);
+ __m128i b2 = _mm_loadu_si128((const __m128i*)(w2 + i));
+ a20 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b2)), a20);
+ a21 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b2,8))), a21);
+ __m128i b3 = _mm_loadu_si128((const __m128i*)(w3 + i));
+ a30 = _mm256_fmadd_ps(xl, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b3)), a30);
+ a31 = _mm256_fmadd_ps(xh, _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b3,8))), a31);
+ }
+ /* fmaf, not `acc += sum * sc`: the shipped single-row loop was
+ * written as a separate multiply and add, but -ffp-contract=fast
+ * fused it, and that fused form is what produced the reference
+ * output. Whether the optimizer also fuses it in this differently
+ * shaped loop is not something to leave to chance -- spelling it
+ * out pins the single rounding the reference depends on. */
+ acc0 = fmaf(gs_group_sum(a00, a01), sc0[gi], acc0);
+ acc1 = fmaf(gs_group_sum(a10, a11), sc1[gi], acc1);
+ acc2 = fmaf(gs_group_sum(a20, a21), sc2[gi], acc2);
+ acc3 = fmaf(gs_group_sum(a30, a31), sc3[gi], acc3);
+ }
+ y[ob+0] = acc0; y[ob+1] = acc1; y[ob+2] = acc2; y[ob+3] = acc3;
+ }
+ for (int o = o4; o < O; o++) { /* at most three rows */
+ const int8_t *w = q + (int64_t)o * I;
+ const float *sc = scale + (int64_t)o * ng;
+ float acc = 0.f;
+ for (int gi = 0; gi < ng; gi++) {
+ __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps();
+ int base = gi * gs, end = base + gs; if (end > I) end = I;
+ for (int i = base; i + 16 <= end; i += 16) {
+ __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i));
+ a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0);
+ a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1);
+ }
+ acc = fmaf(gs_group_sum(a0, a1), sc[gi], acc);
+ }
+ y[o] = acc;
+ }
+ return;
+ }
+#elif defined(__SSE4_1__)
+ if ((gs & 15) == 0) {
+ /* Same four-row interleave as the AVX2 tier above, sized down to this
+ * ISA: each __m128 lane holds 4 lanes instead of 8, so a group step is
+ * 8 int8s, not 16 -- the gate halves to 16 and each row keeps two
+ * __m128 accumulators instead of two __m256. The multiply-add goes
+ * through COLIBRI_FMA (sse41_kernels.h): on real SSE4.1-only hardware
+ * (no __FMA__) that macro expands to the same explicit mul-then-add
+ * this tier always used, so nothing changes there; it only takes the
+ * single-rounded _mm_fmadd_ps path if __FMA__ is somehow defined
+ * without __AVX2__, which does not happen on real hardware but keeps
+ * the tier correct rather than silently wrong if it ever did. Float
+ * loads of `x` go through colibri_sse41_loadu_ps for the same reason
+ * olmoe.c does: one definition shared with future consumers instead
+ * of a fourth inline copy of _mm_loadu_ps. */
+ int o4 = O & ~3;
+ #pragma omp parallel for schedule(static) if(O >= 256)
+ for (int ob = 0; ob < o4; ob += 4) {
+ const int8_t *w0 = q + (int64_t)(ob+0) * I, *w1 = q + (int64_t)(ob+1) * I;
+ const int8_t *w2 = q + (int64_t)(ob+2) * I, *w3 = q + (int64_t)(ob+3) * I;
+ const float *sc0 = scale + (int64_t)(ob+0) * ng, *sc1 = scale + (int64_t)(ob+1) * ng;
+ const float *sc2 = scale + (int64_t)(ob+2) * ng, *sc3 = scale + (int64_t)(ob+3) * ng;
+ float acc0 = 0.f, acc1 = 0.f, acc2 = 0.f, acc3 = 0.f;
+ for (int gi = 0; gi < ng; gi++) {
+ __m128 a00 = _mm_setzero_ps(), a01 = _mm_setzero_ps();
+ __m128 a10 = _mm_setzero_ps(), a11 = _mm_setzero_ps();
+ __m128 a20 = _mm_setzero_ps(), a21 = _mm_setzero_ps();
+ __m128 a30 = _mm_setzero_ps(), a31 = _mm_setzero_ps();
+ int base = gi * gs, end = base + gs; if (end > I) end = I;
+ for (int i = base; i + 8 <= end; i += 8) {
+ __m128 xl = colibri_sse41_loadu_ps(x+i), xh = colibri_sse41_loadu_ps(x+i+4);
+ __m128i b0 = _mm_loadl_epi64((const __m128i*)(w0 + i));
+ a00 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b0)), a00);
+ a01 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,4))), a01);
+ __m128i b1 = _mm_loadl_epi64((const __m128i*)(w1 + i));
+ a10 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b1)), a10);
+ a11 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b1,4))), a11);
+ __m128i b2 = _mm_loadl_epi64((const __m128i*)(w2 + i));
+ a20 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b2)), a20);
+ a21 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b2,4))), a21);
+ __m128i b3 = _mm_loadl_epi64((const __m128i*)(w3 + i));
+ a30 = COLIBRI_FMA(xl, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b3)), a30);
+ a31 = COLIBRI_FMA(xh, _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b3,4))), a31);
+ }
+ acc0 = fmaf(gs_group_sum_sse41(a00, a01), sc0[gi], acc0);
+ acc1 = fmaf(gs_group_sum_sse41(a10, a11), sc1[gi], acc1);
+ acc2 = fmaf(gs_group_sum_sse41(a20, a21), sc2[gi], acc2);
+ acc3 = fmaf(gs_group_sum_sse41(a30, a31), sc3[gi], acc3);
+ }
+ y[ob+0] = acc0; y[ob+1] = acc1; y[ob+2] = acc2; y[ob+3] = acc3;
+ }
+ for (int o = o4; o < O; o++) { /* at most three rows */
+ const int8_t *w = q + (int64_t)o * I;
+ const float *sc = scale + (int64_t)o * ng;
+ float acc = 0.f;
+ for (int gi = 0; gi < ng; gi++) {
+ __m128 a0 = _mm_setzero_ps(), a1 = _mm_setzero_ps();
+ int base = gi * gs, end = base + gs; if (end > I) end = I;
+ for (int i = base; i + 8 <= end; i += 8) {
+ __m128i b0 = _mm_loadl_epi64((const __m128i*)(w + i));
+ a0 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b0)), a0);
+ a1 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+4), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,4))), a1);
+ }
+ acc = fmaf(gs_group_sum_sse41(a0, a1), sc[gi], acc);
+ }
+ y[o] = acc;
+ }
+ return;
+ }
+#endif
+ #pragma omp parallel for schedule(static) if(O >= 256)
+ for (int o = 0; o < O; o++) {
+ const int8_t *w = q + (int64_t)o * I;
+ const float *sc = scale + (int64_t)o * ng;
+ float acc = 0.f;
+ for (int gi = 0; gi < ng; gi++) {
+ int base = gi * gs, end = base + gs; if (end > I) end = I;
+ float part = 0.f;
+ for (int i = base; i < end; i++) part += x[i] * (float)w[i];
+ acc += part * sc[gi];
+ }
+ y[o] = acc;
+ }
+}
+
+#endif /* COLIBRI_GSGEMV_H */
diff --git a/c/idot.h b/c/idot.h
new file mode 100644
index 000000000..34eaf3eb4
--- /dev/null
+++ b/c/idot.h
@@ -0,0 +1,1028 @@
+/* idot.h -- integer dot-product kernels shared by every engine.
+ *
+ * The activation is quantized to int8 once (one scale per row, amax/127,
+ * qrow_i8) and the weights, int8 rows or int4 planar blocks with per-row or
+ * per-group scales, are multiplied with integer instructions: maddubs on
+ * AVX2, vpdpbusd on AVX-VNNI and AVX-512 VNNI, sdot/smmla on NEON, and the
+ * AMX tile kernel where the CPU has it. Exact int32 sums, scaled once per
+ * output, so every ISA path is bit-identical to the scalar reference.
+ *
+ * Lived inside quant.h; moved here so an engine that has its own dense
+ * kernels (qwen36.c uses qgemv.h/gsgemv.h, whose matmul_q would collide
+ * with quant.h's) can still take the integer path for its dense trunk.
+ * quant.h includes this file at the top, so its other consumers see the
+ * same declarations in the same place as before. */
+#ifndef COLIBRI_IDOT_H
+#define COLIBRI_IDOT_H
+#include
+#include
+#include
+#include
+#include
+/* ---- SIMD includes -------------------------------------------------------- */
+#ifdef __AVX2__
+#include
+static inline float hsum256(__m256 v){
+ __m128 lo=_mm256_castps256_ps128(v), hi=_mm256_extractf128_ps(v,1);
+ lo=_mm_add_ps(lo,hi); __m128 sh=_mm_movehl_ps(lo,lo); lo=_mm_add_ps(lo,sh);
+ sh=_mm_shuffle_ps(lo,lo,1); lo=_mm_add_ss(lo,sh); return _mm_cvtss_f32(lo);
+}
+static inline int hsum256_i32(__m256i v){
+ __m128i lo=_mm256_castsi256_si128(v), hi=_mm256_extracti128_si256(v,1);
+ lo=_mm_add_epi32(lo,hi); lo=_mm_hadd_epi32(lo,lo); lo=_mm_hadd_epi32(lo,lo);
+ return _mm_cvtsi128_si32(lo);
+}
+#endif
+#if defined(__AVXVNNI__) && defined(__AVX2__)
+static inline int hsum128_i32(__m128i v){
+ v=_mm_hadd_epi32(v,v); v=_mm_hadd_epi32(v,v); return _mm_cvtsi128_si32(v);
+}
+#endif
+#ifdef __ARM_NEON
+#include
+#endif
+#ifdef __VSX__
+#include
+#undef vector
+#undef pixel
+#undef bool
+#endif
+
+
+/* ---- IDOT: integer dot kernels (int8-quantized activations) --------------- */
+#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
+#define IDOT_KERNEL "avx512-vnni"
+#elif defined(__AVXVNNI__) && defined(__AVX2__)
+#define IDOT_KERNEL "avx-vnni"
+#elif defined(__AVX2__)
+#define IDOT_KERNEL "avx2"
+#elif defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
+#define IDOT_KERNEL "neon-i8mm"
+#elif defined(__ARM_NEON)
+#define IDOT_KERNEL "neon"
+#elif defined(__VSX__)
+#define IDOT_KERNEL "vsx"
+#else
+#define IDOT_KERNEL "scalar"
+#endif
+
+static inline float qrow_i8(const float *x, int8_t *q, int I){
+ float amax=0; for(int i=0;iamax)amax=a; }
+ float s=amax/127.f; if(s<1e-12f) s=1e-12f; float inv=1.f/s;
+ for(int i=0;i int8 with one scale (shared by every engine whose trunk
+ * meets the integer kernels below), plus the int32 sum of every block of 64
+ * (the K1b kernel subtracts 8*sum per group because its nibbles are unsigned).
+ * Vectorized: the scalar lrintf loop was measured at ~7 us for 4096 values,
+ * which times ~150 GEMVs per token is a millisecond thrown away. Rounding is
+ * to nearest even in both paths, so the vector path equals qrow_i8 bit for bit. */
+static float dense_act_i8(const float *x, int I, int8_t *xq, int32_t *xsg){
+ float amax = 0.f;
+ int i = 0;
+#ifdef __AVX2__
+ {
+ __m256 am = _mm256_setzero_ps();
+ const __m256 sign = _mm256_set1_ps(-0.0f);
+ for (; i + 8 <= I; i += 8) am = _mm256_max_ps(am, _mm256_andnot_ps(sign, _mm256_loadu_ps(x + i)));
+ float tmp[8]; _mm256_storeu_ps(tmp, am);
+ for (int k = 0; k < 8; k++) if (tmp[k] > amax) amax = tmp[k];
+ }
+#endif
+ for (; i < I; i++) { float a = fabsf(x[i]); if (a > amax) amax = a; }
+ float s = amax / 127.f; if (s < 1e-12f) s = 1e-12f;
+ float inv = 1.f / s;
+ i = 0;
+#ifdef __AVX2__
+ {
+ const __m256 vinv = _mm256_set1_ps(inv);
+ for (; i + 32 <= I; i += 32) {
+ __m256i a = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i), vinv));
+ __m256i b = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i + 8), vinv));
+ __m256i c = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i + 16), vinv));
+ __m256i d = _mm256_cvtps_epi32(_mm256_mul_ps(_mm256_loadu_ps(x + i + 24), vinv));
+ /* packs interleave 128-bit lanes: fix the order with one permute */
+ __m256i ab = _mm256_packs_epi32(a, b), cd = _mm256_packs_epi32(c, d);
+ __m256i abcd = _mm256_packs_epi16(ab, cd);
+ abcd = _mm256_permutevar8x32_epi32(abcd, _mm256_setr_epi32(0, 4, 1, 5, 2, 6, 3, 7));
+ _mm256_storeu_si256((__m256i *)(xq + i), abcd);
+ }
+ }
+#endif
+ for (; i < I; i++) xq[i] = (int8_t)lrintf(x[i] * inv);
+ if (xsg) {
+ int ng = I / 64;
+ for (int g = 0; g < ng; g++) {
+ int32_t sum = 0;
+ for (int k = 0; k < 64; k++) sum += xq[g * 64 + k];
+ xsg[g] = sum;
+ }
+ }
+ return s;
+}
+
+/* dot int8*int8 */
+static inline int32_t dot_i8i8(const int8_t *w, const int8_t *x, int I){
+ int32_t sum=0; int i=0;
+#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
+ __m512i acc=_mm512_setzero_si512();
+ for(;i+64<=I;i+=64){
+ __m512i wv=_mm512_loadu_si512((const void*)(w+i));
+ __m512i xv=_mm512_loadu_si512((const void*)(x+i));
+ __mmask64 neg=_mm512_movepi8_mask(wv);
+ __m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
+ acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
+ }
+ sum=_mm512_reduce_add_epi32(acc);
+#elif defined(__AVXVNNI__) && defined(__AVX2__)
+ /* 4 accumulatori indipendenti (64 byte/iter): un solo acc incatena i vpdpbusd
+ * (latenza-bound ~5c). Somme intere associative -> bit-identico. Stessa struttura
+ * dei 4 accumulatori del ramo NEON piu' sotto.
+ * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer
+ * adds are associative, so the result is bit-identical (mirrors the NEON path). */
+ __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128();
+ for(;i+64<=I;i+=64){
+ __m128i w0=_mm_loadu_si128((const __m128i*)(w+i)), x0=_mm_loadu_si128((const __m128i*)(x+i));
+ __m128i w1=_mm_loadu_si128((const __m128i*)(w+i+16)), x1=_mm_loadu_si128((const __m128i*)(x+i+16));
+ __m128i w2=_mm_loadu_si128((const __m128i*)(w+i+32)), x2=_mm_loadu_si128((const __m128i*)(x+i+32));
+ __m128i w3=_mm_loadu_si128((const __m128i*)(w+i+48)), x3=_mm_loadu_si128((const __m128i*)(x+i+48));
+ a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
+ a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
+ a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2));
+ a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3));
+ }
+ __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3));
+ for(;i+16<=I;i+=16){
+ __m128i wv=_mm_loadu_si128((const __m128i*)(w+i));
+ __m128i xv=_mm_loadu_si128((const __m128i*)(x+i));
+ acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),_mm_sign_epi8(xv,wv));
+ }
+ sum=hsum128_i32(acc);
+#elif defined(__AVX2__)
+ __m256i acc=_mm256_setzero_si256(); const __m256i ones=_mm256_set1_epi16(1);
+ for(;i+32<=I;i+=32){
+ __m256i wv=_mm256_loadu_si256((const __m256i*)(w+i));
+ __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
+ __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
+ acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
+ }
+ sum=hsum256_i32(acc);
+#elif defined(__ARM_NEON)
+#if defined(__ARM_FEATURE_DOTPROD)
+ int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
+ for(;i+64<=I;i+=64){
+ a0=vdotq_s32(a0,vld1q_s8(w+i), vld1q_s8(x+i));
+ a1=vdotq_s32(a1,vld1q_s8(w+i+16),vld1q_s8(x+i+16));
+ a2=vdotq_s32(a2,vld1q_s8(w+i+32),vld1q_s8(x+i+32));
+ a3=vdotq_s32(a3,vld1q_s8(w+i+48),vld1q_s8(x+i+48));
+ }
+ int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
+ for(;i+16<=I;i+=16) acc=vdotq_s32(acc,vld1q_s8(w+i),vld1q_s8(x+i));
+ sum=vaddvq_s32(acc);
+#else
+ int32x4_t acc=vdupq_n_s32(0);
+ for(;i+16<=I;i+=16){
+ int8x16_t wv=vld1q_s8(w+i), xv=vld1q_s8(x+i);
+ int16x8_t p=vmull_s8(vget_low_s8(wv),vget_low_s8(xv));
+ p=vmlal_s8(p,vget_high_s8(wv),vget_high_s8(xv));
+ acc=vpadalq_s16(acc,p);
+ }
+ sum=vaddvq_s32(acc);
+#endif
+#elif defined(__VSX__)
+ __vector signed int acc=vec_splats(0);
+ const __vector signed char vz=vec_splats((signed char)0);
+ for(;i+16<=I;i+=16){
+ __vector signed char wv=vec_xl(0,(const signed char*)(w+i));
+ __vector signed char xv=vec_xl(0,(const signed char*)(x+i));
+ __vector __bool char neg=vec_cmplt(wv,vz);
+ __vector signed char xs=vec_sel(xv,vec_sub(vz,xv),neg);
+ __vector unsigned char wa=(__vector unsigned char)vec_sel(wv,vec_sub(vz,wv),neg);
+ acc=vec_msum(xs,wa,acc);
+ }
+ sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
+#endif
+ for(;i>1)));
+ __m256i lo=_mm256_and_si256(by,m4v), hi=_mm256_and_si256(_mm256_srli_epi16(by,4),m4v);
+ __m256i z0=_mm256_unpacklo_epi8(lo,hi), z1=_mm256_unpackhi_epi8(lo,hi);
+ __m512i wv=_mm512_sub_epi8(_mm512_inserti64x4(_mm512_castsi256_si512(z0),z1,1),b8v);
+ __m512i xv=_mm512_permutexvar_epi64(xidx,_mm512_loadu_si512((const void*)(x+i)));
+ __mmask64 neg=_mm512_movepi8_mask(wv);
+ __m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
+ acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
+ }
+ sum=_mm512_reduce_add_epi32(acc);
+#elif defined(__AVXVNNI__) && defined(__AVX2__)
+ /* 4 accumulatori indipendenti (64 elementi = 32 byte packed/iter): un solo acc
+ * incatena i vpdpbusd (latenza-bound ~5c). Somme intere associative -> bit-identico.
+ * Stessa struttura dei 4 accumulatori del ramo NEON piu' sotto.
+ * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer
+ * adds are associative, so the result is bit-identical (mirrors the NEON path). */
+ const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8);
+ __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128();
+ for(;i+64<=I;i+=64){
+ __m128i by0=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); /* elem i..i+31 */
+ __m128i by1=_mm_loadu_si128((const __m128i*)(w4+(i>>1)+16)); /* elem i+32..i+63 */
+ __m128i lo0=_mm_and_si128(by0,m4), hi0=_mm_and_si128(_mm_srli_epi16(by0,4),m4);
+ __m128i lo1=_mm_and_si128(by1,m4), hi1=_mm_and_si128(_mm_srli_epi16(by1,4),m4);
+ __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo0,hi0),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo0,hi0),b8);
+ __m128i w2=_mm_sub_epi8(_mm_unpacklo_epi8(lo1,hi1),b8), w3=_mm_sub_epi8(_mm_unpackhi_epi8(lo1,hi1),b8);
+ __m128i x0=_mm_loadu_si128((const __m128i*)(x+i)), x1=_mm_loadu_si128((const __m128i*)(x+i+16));
+ __m128i x2=_mm_loadu_si128((const __m128i*)(x+i+32)), x3=_mm_loadu_si128((const __m128i*)(x+i+48));
+ a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
+ a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
+ a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2));
+ a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3));
+ }
+ __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3));
+ for(;i+32<=I;i+=32){ /* 32-nibble remainder: 2 dpbusd, same unpack */
+ __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
+ __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
+ __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo,hi),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo,hi),b8);
+ __m128i x0=_mm_loadu_si128((const __m128i*)(x+i));
+ __m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16));
+ acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
+ acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
+ }
+ sum=hsum128_i32(acc);
+#elif defined(__AVX2__)
+ const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi8(8);
+ const __m256i ones=_mm256_set1_epi16(1);
+ __m256i acc=_mm256_setzero_si256();
+ for(;i+32<=I;i+=32){
+ __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
+ __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
+ __m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi);
+ __m256i wv=_mm256_sub_epi8(_mm256_set_m128i(n1,n0),b8);
+ __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
+ __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
+ acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
+ }
+ sum=hsum256_i32(acc);
+#elif defined(__ARM_NEON)
+ const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8);
+#if defined(__ARM_FEATURE_DOTPROD)
+ int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
+ for(;i+64<=I;i+=64){
+ uint8x16_t byA=vld1q_u8(w4+(i>>1)), byB=vld1q_u8(w4+(i>>1)+16);
+ uint8x16x2_t zA=vzipq_u8(vandq_u8(byA,m4q), vshrq_n_u8(byA,4));
+ uint8x16x2_t zB=vzipq_u8(vandq_u8(byB,m4q), vshrq_n_u8(byB,4));
+ a0=vdotq_s32(a0,vsubq_s8(vreinterpretq_s8_u8(zA.val[0]),b8q),vld1q_s8(x+i));
+ a1=vdotq_s32(a1,vsubq_s8(vreinterpretq_s8_u8(zA.val[1]),b8q),vld1q_s8(x+i+16));
+ a2=vdotq_s32(a2,vsubq_s8(vreinterpretq_s8_u8(zB.val[0]),b8q),vld1q_s8(x+i+32));
+ a3=vdotq_s32(a3,vsubq_s8(vreinterpretq_s8_u8(zB.val[1]),b8q),vld1q_s8(x+i+48));
+ }
+ int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
+ for(;i+32<=I;i+=32){
+ uint8x16_t by=vld1q_u8(w4+(i>>1));
+ uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
+ acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q),vld1q_s8(x+i));
+ acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q),vld1q_s8(x+i+16));
+ }
+ sum=vaddvq_s32(acc);
+#else
+ int32x4_t acc=vdupq_n_s32(0);
+ for(;i+32<=I;i+=32){
+ uint8x16_t by=vld1q_u8(w4+(i>>1));
+ uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
+ int8x16_t w0=vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q);
+ int8x16_t w1=vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q);
+ int8x16_t x0=vld1q_s8(x+i), x1=vld1q_s8(x+i+16);
+ int16x8_t p=vmull_s8(vget_low_s8(w0),vget_low_s8(x0));
+ p=vmlal_s8(p,vget_high_s8(w0),vget_high_s8(x0));
+ acc=vpadalq_s16(acc,p);
+ p=vmull_s8(vget_low_s8(w1),vget_low_s8(x1));
+ p=vmlal_s8(p,vget_high_s8(w1),vget_high_s8(x1));
+ acc=vpadalq_s16(acc,p);
+ }
+ sum=vaddvq_s32(acc);
+#endif
+#elif defined(__VSX__)
+ const __vector unsigned char m4v=vec_splats((unsigned char)0x0F);
+ const __vector unsigned char sh4=vec_splats((unsigned char)4);
+ const __vector signed char b8v=vec_splats((signed char)8);
+ const __vector signed char vz=vec_splats((signed char)0);
+ __vector signed int acc=vec_splats(0);
+ for(;i+32<=I;i+=32){
+ __vector unsigned char by=vec_xl(0,w4+(i>>1));
+ __vector unsigned char lo=vec_and(by,m4v), hi=vec_sr(by,sh4);
+ __vector signed char w0=vec_sub((__vector signed char)vec_mergeh(lo,hi),b8v);
+ __vector signed char w1=vec_sub((__vector signed char)vec_mergel(lo,hi),b8v);
+ __vector signed char x0=vec_xl(0,(const signed char*)(x+i));
+ __vector signed char x1=vec_xl(0,(const signed char*)(x+i+16));
+ __vector __bool char n0=vec_cmplt(w0,vz), n1=vec_cmplt(w1,vz);
+ acc=vec_msum(vec_sel(x0,vec_sub(vz,x0),n0),
+ (__vector unsigned char)vec_sel(w0,vec_sub(vz,w0),n0),acc);
+ acc=vec_msum(vec_sel(x1,vec_sub(vz,x1),n1),
+ (__vector unsigned char)vec_sel(w1,vec_sub(vz,w1),n1),acc);
+ }
+ sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
+#endif
+ for(;i+1>1]; sum+=((int)(b&0xF)-8)*x[i]+((int)(b>>4)-8)*x[i+1]; }
+ if(i>1]; sum+=((int)(b&0xF)-8)*x[i]; }
+ return sum;
+}
+
+/* ---- ARM i8mm SMMLA tiled kernels ---------------------------------------- */
+#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
+static inline int32x4_t mm_tile16(int32x4_t acc, int8x16_t wo, int8x16_t wo1,
+ int8x16_t xs, int8x16_t xs1){
+ acc=vmmlaq_s32(acc, vcombine_s8(vget_low_s8(wo), vget_low_s8(wo1)),
+ vcombine_s8(vget_low_s8(xs), vget_low_s8(xs1)));
+ return vmmlaq_s32(acc, vcombine_s8(vget_high_s8(wo), vget_high_s8(wo1)),
+ vcombine_s8(vget_high_s8(xs), vget_high_s8(xs1)));
+}
+static void matmul_q_idot_mm(float *y, const int8_t *xq, const float *sx, const int8_t *q,
+ const float *scale, int S, int I, int O){
+ #pragma omp parallel for schedule(static)
+ for(int o=0;o<(O&~1);o+=2){
+ const int8_t *wo=q+(int64_t)o*I, *wo1=q+(int64_t)(o+1)*I;
+ float sc0=scale[o], sc1=scale[o+1];
+ for(int s=0;s<(S&~1);s+=2){
+ const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I;
+ int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0;
+ for(;i+64<=I;i+=64){
+ a0=mm_tile16(a0,vld1q_s8(wo+i), vld1q_s8(wo1+i), vld1q_s8(xs+i), vld1q_s8(xs1+i));
+ a1=mm_tile16(a1,vld1q_s8(wo+i+16),vld1q_s8(wo1+i+16),vld1q_s8(xs+i+16),vld1q_s8(xs1+i+16));
+ a2=mm_tile16(a2,vld1q_s8(wo+i+32),vld1q_s8(wo1+i+32),vld1q_s8(xs+i+32),vld1q_s8(xs1+i+32));
+ a3=mm_tile16(a3,vld1q_s8(wo+i+48),vld1q_s8(wo1+i+48),vld1q_s8(xs+i+48),vld1q_s8(xs1+i+48));
+ }
+ for(;i+16<=I;i+=16)
+ a0=mm_tile16(a0,vld1q_s8(wo+i),vld1q_s8(wo1+i),vld1q_s8(xs+i),vld1q_s8(xs1+i));
+ int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
+ int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
+ int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
+ for(;i>1)), byo1=vld1q_u8(wo1+(i>>1));
+ uint8x16_t cyo=vld1q_u8(wo+(i>>1)+16), cyo1=vld1q_u8(wo1+(i>>1)+16);
+ uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
+ uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
+ uint8x16x2_t ko =vzipq_u8(vandq_u8(cyo, m4q), vshrq_n_u8(cyo, 4));
+ uint8x16x2_t ko1=vzipq_u8(vandq_u8(cyo1,m4q), vshrq_n_u8(cyo1,4));
+ a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
+ vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
+ vld1q_s8(xs+i), vld1q_s8(xs1+i));
+ a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
+ vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
+ vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
+ a2=mm_tile16(a2, vsubq_s8(vreinterpretq_s8_u8(ko.val[0]),b8q),
+ vsubq_s8(vreinterpretq_s8_u8(ko1.val[0]),b8q),
+ vld1q_s8(xs+i+32), vld1q_s8(xs1+i+32));
+ a3=mm_tile16(a3, vsubq_s8(vreinterpretq_s8_u8(ko.val[1]),b8q),
+ vsubq_s8(vreinterpretq_s8_u8(ko1.val[1]),b8q),
+ vld1q_s8(xs+i+48), vld1q_s8(xs1+i+48));
+ }
+ for(;i+32<=I;i+=32){
+ uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1));
+ uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
+ uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
+ a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
+ vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
+ vld1q_s8(xs+i), vld1q_s8(xs1+i));
+ a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
+ vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
+ vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
+ }
+ int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
+ int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
+ int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
+ for(;i+1>1], bo1=wo1[i>>1];
+ int a0=(int)(bo&0xF)-8, a1=(int)(bo>>4)-8, b0=(int)(bo1&0xF)-8, b1=(int)(bo1>>4)-8;
+ int u0=xs[i],u1=xs[i+1],v0=xs1[i],v1=xs1[i+1];
+ d00+=a0*u0+a1*u1; d01+=a0*v0+a1*v1; d10+=b0*u0+b1*u1; d11+=b0*v0+b1*v1; }
+ if(i>1], bo1=wo1[i>>1];
+ int a0=(int)(bo&0xF)-8, b0=(int)(bo1&0xF)-8;
+ d00+=a0*xs[i]; d01+=a0*xs1[i]; d10+=b0*xs[i]; d11+=b0*xs1[i]; }
+ y[(int64_t)s*O+o] =(float)d00*sc0*sx[s];
+ y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s];
+ y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1];
+ y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1];
+ }
+ if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I;
+ y[(int64_t)s*O+o] =(float)dot_i4i8(wo, xs,I)*sc0*sx[s];
+ y[(int64_t)s*O+(o+1)]=(float)dot_i4i8(wo1,xs,I)*sc1*sx[s]; }
+ }
+ if(O&1){ int o=O-1; const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
+ #pragma omp parallel for schedule(static)
+ for(int s=0;s=2){ matmul_q_idot_mm(y,xq,sx,q,scale,S,I,O); return; }
+#endif
+ #pragma omp parallel for schedule(static)
+ for(int o=0;o=2){ matmul_i4_idot_mm(y,xq,sx,q4,scale,S,I,O); return; }
+#endif
+ #pragma omp parallel for schedule(static)
+ for(int o=0;oplanar repack. */
+static void planarize_i4_row(uint8_t *row, int I){
+ uint8_t tmp[32];
+ int nb=I/64;
+ for(int b=0;b>1]>>((src_lo&1)*4))&0xF;
+ uint8_t nib_hi=(blk[src_hi>>1]>>((src_hi&1)*4))&0xF;
+ tmp[k]=(uint8_t)(nib_lo|(nib_hi<<4));
+ }
+ memcpy(blk,tmp,32);
+ }
+}
+static void planarize_i4(uint8_t *q4, int O, int I){
+ int rb=(I+1)/2;
+ #pragma omp parallel for schedule(static)
+ for(int o=0;o>1)));
+ __m256i b1=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)+32));
+ a0=coli_dpbusd256(a0,_mm256_and_si256(b0,m4),
+ _mm256_loadu_si256((const __m256i*)(x+i)));
+ a1=coli_dpbusd256(a1,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4),
+ _mm256_loadu_si256((const __m256i*)(x+i+32)));
+ a2=coli_dpbusd256(a2,_mm256_and_si256(b1,m4),
+ _mm256_loadu_si256((const __m256i*)(x+i+64)));
+ a3=coli_dpbusd256(a3,_mm256_and_si256(_mm256_srli_epi16(b1,4),m4),
+ _mm256_loadu_si256((const __m256i*)(x+i+96)));
+ }
+ __m256i acc=_mm256_add_epi32(_mm256_add_epi32(a0,a1),_mm256_add_epi32(a2,a3));
+ for(;i+64<=I;i+=64){
+ __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)));
+ acc=coli_dpbusd256(acc,_mm256_and_si256(b0,m4),
+ _mm256_loadu_si256((const __m256i*)(x+i)));
+ acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4),
+ _mm256_loadu_si256((const __m256i*)(x+i+32)));
+ }
+ sum=hsum256_i32(acc);
+#elif defined(__AVX2__)
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ const __m256i ones=_mm256_set1_epi16(1);
+ __m256i acc=_mm256_setzero_si256();
+ for(;i+64<=I;i+=64){
+ __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)));
+ /* maddubs(u8, s8): u<=15, |x|<=127 -> coppia <= 3810, int16 sicuro */
+ __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(b0,m4),
+ _mm256_loadu_si256((const __m256i*)(x+i)));
+ __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(b0,4),m4),
+ _mm256_loadu_si256((const __m256i*)(x+i+32)));
+ acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p0,ones));
+ acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p1,ones));
+ }
+ sum=hsum256_i32(acc);
+#elif defined(__ARM_NEON)
+ int32x4_t acc=vdupq_n_s32(0);
+ for(;i+64<=I;i+=64){
+ uint8x16_t b0=vld1q_u8(w4+(i>>1)), b1=vld1q_u8(w4+(i>>1)+16);
+ int8x16_t lo0=vreinterpretq_s8_u8(vandq_u8(b0,vdupq_n_u8(0x0F)));
+ int8x16_t lo1=vreinterpretq_s8_u8(vandq_u8(b1,vdupq_n_u8(0x0F)));
+ int8x16_t hi0=vreinterpretq_s8_u8(vshrq_n_u8(b0,4));
+ int8x16_t hi1=vreinterpretq_s8_u8(vshrq_n_u8(b1,4));
+#if defined(__ARM_FEATURE_DOTPROD)
+ acc=vdotq_s32(acc,lo0,vld1q_s8(x+i));
+ acc=vdotq_s32(acc,lo1,vld1q_s8(x+i+16));
+ acc=vdotq_s32(acc,hi0,vld1q_s8(x+i+32));
+ acc=vdotq_s32(acc,hi1,vld1q_s8(x+i+48));
+#else
+ int8x16_t xs0=vld1q_s8(x+i), xs1=vld1q_s8(x+i+16);
+ int8x16_t xs2=vld1q_s8(x+i+32), xs3=vld1q_s8(x+i+48);
+ int16x8_t m;
+ m=vmull_s8(vget_low_s8(lo0),vget_low_s8(xs0)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_high_s8(lo0),vget_high_s8(xs0)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_low_s8(lo1),vget_low_s8(xs1)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_high_s8(lo1),vget_high_s8(xs1)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_low_s8(hi0),vget_low_s8(xs2)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_high_s8(hi0),vget_high_s8(xs2)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_low_s8(hi1),vget_low_s8(xs3)); acc=vpadalq_s16(acc,m);
+ m=vmull_s8(vget_high_s8(hi1),vget_high_s8(xs3)); acc=vpadalq_s16(acc,m);
+#endif
+ }
+ sum=vaddvq_s32(acc);
+#endif
+ for(;i+64<=I;i+=64){ /* fallback scalare sui blocchi planari */
+ const uint8_t *blk=w4+(i>>1);
+ for(int k=0;k<32;k++){
+ sum+=(int32_t)(blk[k]&0xF)*x[i+k];
+ sum+=(int32_t)(blk[k]>>4)*x[i+k+32];
+ }
+ }
+ for(;i>1];
+ sum+=(int32_t)(byte&0xF)*x[i];
+ if(i+1>4)*x[i+1];
+ }
+ return sum;
+}
+
+/* ---- K1b (OPT-IN, IDOT_GS=1): IDOT planare A GRUPPI (fmt=4, gs%64==0) -----
+ * Con gs=64 il gruppo di scala COINCIDE col blocco-piano da 64 elementi: il
+ * dot unsigned del blocco (2 dpbusd) -> int32 di gruppo, meno 8*somma(x) del
+ * gruppo, per la scala f32 del gruppo. Attivazioni int8 (stessa famiglia
+ * qrow_i8 del resto dell'IDOT): NON bit-identico al kernel f32 a gruppi --
+ * per questo e' dietro flag, in attesa dell'ablazione. xsg = somme int32
+ * per (riga, gruppo), calcolate dal chiamante in una passata esatta.
+ * EN: grouped planar IDOT, opt-in. With gs=64 the scale group IS the plane
+ * block; per-group unsigned dot minus 8*group-sum, times the group scale.
+ * int8 activations: not bit-identical to the f32 grouped kernel, hence the
+ * flag until the ablation blesses a default.
+ *
+ * Three execution shapes below, one contract: per (row, output) the float
+ * accumulation is the SAME sequence of fmaf((float)group_int, scale[g], a)
+ * in ascending g, and every group_int is an exact int32 — so the per-row
+ * path, the 1x4 row tile, and the AMX tile are bit-identical to each other
+ * and to the pure-C reference on every ISA. */
+
+/* one planar 64-element block, unpacked once and shared across a row tile:
+ * lo nibbles = elements base..base+31 in order, hi = base+32..base+63. */
+#if defined(coli_dpbusd256)
+static inline void i4p_blk256(const uint8_t *blk, __m256i *lo, __m256i *hi){
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ __m256i b=_mm256_loadu_si256((const __m256i*)blk);
+ *lo=_mm256_and_si256(b,m4);
+ *hi=_mm256_and_si256(_mm256_srli_epi16(b,4),m4);
+}
+#endif
+#if defined(__AVX512VNNI__) && defined(__AVX512BW__)
+/* whole block in one zmm: lanes 0..31 = lo elements, 32..63 = hi — matches a
+ * straight 64-byte load of the activation block, so ONE dpbusd per block. */
+static inline __m512i i4p_blk512(const uint8_t *blk){
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ __m256i b=_mm256_loadu_si256((const __m256i*)blk);
+ return _mm512_inserti64x4(_mm512_castsi256_si512(_mm256_and_si256(b,m4)),
+ _mm256_and_si256(_mm256_srli_epi16(b,4),m4),1);
+}
+#endif
+
+/* ---- K1c: AMX int8 tile kernel (Sapphire Rapids+, opt-in via the same
+ * IDOT_GS=1 family gate; AMX=0 kills it, AMX_S_MIN sets the row threshold).
+ * With gs a multiple of 64, K=64 tile-multiplies cover a scale group exactly:
+ * B tile = 16 output rows' int4 block unpacked once to SIGNED int8 (v-8, so
+ * tdpbssd returns d - 8*sum(x_group) directly — the same int32 the vector
+ * path computes as d_unsigned - 8*xsg), A tile = up to 16 activation rows.
+ * The unpack cost is paid once per (output tile, group) and amortized over
+ * every activation row — the multi-row reuse the pair-layout kernels lack.
+ * Linux-only arming: tile data needs an ARCH_REQ_XCOMP_PERM handshake. */
+#if defined(__AMX_INT8__) && defined(__AMX_TILE__) && defined(__AVX512F__)
+#define COLI_HAVE_AMX_I4P 1
+#if defined(__linux__)
+#include
+#include
+#elif defined(_WIN32)
+#ifndef WIN32_LEAN_AND_MEAN
+#define WIN32_LEAN_AND_MEAN
+#endif
+#include
+#endif
+static int coli_amx_state=-1;
+static int coli_amx_s_min=8;
+static int coli_amx_ok(void){
+ if(coli_amx_state<0){
+ const char *e=getenv("AMX");
+ const char *sm=getenv("AMX_S_MIN"); if(sm&&atoi(sm)>0) coli_amx_s_min=atoi(sm);
+#if defined(__linux__)
+ /* ARCH_REQ_XCOMP_PERM(XFEATURE_XTILEDATA): kernel >= 5.16 grants tile
+ * state per-process; without it the first tile op SIGILLs. */
+ coli_amx_state=(e&&*e=='0')?0:(syscall(SYS_arch_prctl,0x1023,18)==0);
+#elif defined(_WIN32)
+ /* Windows 11: tile data is an opt-in per-process XState feature.
+ * Dynamic lookup keeps older kernels building/running (arming just
+ * fails closed there). XSTATE_AMX_TILE_DATA = 18. */
+ if(e&&*e=='0') coli_amx_state=0;
+ else {
+ HMODULE k32=GetModuleHandleA("kernel32.dll");
+ typedef BOOL (WINAPI *coli_xstate_fn)(ULONG64);
+ coli_xstate_fn f=k32?(coli_xstate_fn)(void*)GetProcAddress(k32,"EnableProcessOptionalXStateFeatures"):NULL;
+ coli_amx_state=(f && f(1ull<<18))?1:0;
+ }
+#else
+ (void)e; coli_amx_state=0;
+#endif
+ if(coli_amx_state)
+ fprintf(stderr,"[K1c] AMX int8 tile kernel armed for gs64 tensors "
+ "(AMX=0 disables, engages at S>=%d)\n",coli_amx_s_min);
+ }
+ return coli_amx_state;
+}
+/* ldtilecfg layout (palette 1): tmm0=C [rows x 16 i32], tmm1=A [rows x 64 i8],
+ * tmm2=B [16 x 64 i8 VNNI]. rows<16 reconfigures for the last partial s-tile. */
+struct coli_tilecfg { uint8_t palette,start_row,rsvd[14]; uint16_t colsb[16]; uint8_t rows[16]; };
+static void coli_amx_cfg(int arows){
+ struct coli_tilecfg c; memset(&c,0,sizeof c); c.palette=1;
+ c.rows[0]=(uint8_t)arows; c.colsb[0]=64;
+ c.rows[1]=(uint8_t)arows; c.colsb[1]=64;
+ c.rows[2]=16; c.colsb[2]=64;
+ _tile_loadconfig(&c);
+}
+static void matmul_i4p_gidot_amx(float *y, const int8_t *xq, const float *sx,
+ const uint8_t *q4, const float *scale,
+ int S, int I, int O16, int O, int gs){
+ int rb=(I+1)/2, ng=(I+gs-1)/gs, bpg=gs/64;
+ #pragma omp parallel
+ {
+ float *acc=malloc((size_t)S*16*sizeof(float));
+ if(!acc){ fprintf(stderr,"OOM: amx acc\n"); exit(1); }
+ int8_t bstage[4*1024] __attribute__((aligned(64))); /* bpg<=4 gated below */
+ int32_t cbuf[16*16] __attribute__((aligned(64)));
+ float sclT[16];
+ int cur=16; coli_amx_cfg(16);
+ #pragma omp for schedule(static)
+ for(int ot=0; ot>1;
+ for(int n=0;n<16;n++){
+ const uint8_t *blk=q4+(int64_t)(ot+n)*rb+boff;
+ for(int k=0;k<32;k++){
+ dst[(k>>2)*64+n*4+(k&3)] =(int8_t)((blk[k]&0xF)-8);
+ dst[((k+32)>>2)*64+n*4+((k+32)&3)]=(int8_t)((blk[k]>>4)-8);
+ }
+ }
+ }
+ for(int st=0; st>1));
+ A0=_mm512_dpbusd_epi32(A0,wz,_mm512_loadu_si512((const void*)(x0+base)));
+ A1=_mm512_dpbusd_epi32(A1,wz,_mm512_loadu_si512((const void*)(x1+base)));
+ A2=_mm512_dpbusd_epi32(A2,wz,_mm512_loadu_si512((const void*)(x2+base)));
+ A3=_mm512_dpbusd_epi32(A3,wz,_mm512_loadu_si512((const void*)(x3+base)));
+ }
+ d0=_mm512_reduce_add_epi32(A0); d1=_mm512_reduce_add_epi32(A1);
+ d2=_mm512_reduce_add_epi32(A2); d3=_mm512_reduce_add_epi32(A3);
+#elif defined(coli_dpbusd256)
+ __m256i A0=_mm256_setzero_si256(),A1=_mm256_setzero_si256();
+ __m256i A2=_mm256_setzero_si256(),A3=_mm256_setzero_si256();
+ for(int b=0;b>1),&lo,&hi);
+ A0=coli_dpbusd256(A0,lo,_mm256_loadu_si256((const __m256i*)(x0+base)));
+ A0=coli_dpbusd256(A0,hi,_mm256_loadu_si256((const __m256i*)(x0+base+32)));
+ A1=coli_dpbusd256(A1,lo,_mm256_loadu_si256((const __m256i*)(x1+base)));
+ A1=coli_dpbusd256(A1,hi,_mm256_loadu_si256((const __m256i*)(x1+base+32)));
+ A2=coli_dpbusd256(A2,lo,_mm256_loadu_si256((const __m256i*)(x2+base)));
+ A2=coli_dpbusd256(A2,hi,_mm256_loadu_si256((const __m256i*)(x2+base+32)));
+ A3=coli_dpbusd256(A3,lo,_mm256_loadu_si256((const __m256i*)(x3+base)));
+ A3=coli_dpbusd256(A3,hi,_mm256_loadu_si256((const __m256i*)(x3+base+32)));
+ }
+ d0=hsum256_i32(A0); d1=hsum256_i32(A1);
+ d2=hsum256_i32(A2); d3=hsum256_i32(A3);
+#elif defined(__AVX2__)
+ const __m256i ones=_mm256_set1_epi16(1);
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ __m256i A0=_mm256_setzero_si256(),A1=_mm256_setzero_si256();
+ __m256i A2=_mm256_setzero_si256(),A3=_mm256_setzero_si256();
+ for(int b=0;b>1)));
+ __m256i lo=_mm256_and_si256(bb,m4);
+ __m256i hi=_mm256_and_si256(_mm256_srli_epi16(bb,4),m4);
+ /* maddubs(u8,s8): u<=15, |x|<=127 -> pair <= 3810, int16-safe */
+ A0=_mm256_add_epi32(A0,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x0+base))),ones));
+ A0=_mm256_add_epi32(A0,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x0+base+32))),ones));
+ A1=_mm256_add_epi32(A1,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x1+base))),ones));
+ A1=_mm256_add_epi32(A1,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x1+base+32))),ones));
+ A2=_mm256_add_epi32(A2,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x2+base))),ones));
+ A2=_mm256_add_epi32(A2,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x2+base+32))),ones));
+ A3=_mm256_add_epi32(A3,_mm256_madd_epi16(_mm256_maddubs_epi16(lo,_mm256_loadu_si256((const __m256i*)(x3+base))),ones));
+ A3=_mm256_add_epi32(A3,_mm256_madd_epi16(_mm256_maddubs_epi16(hi,_mm256_loadu_si256((const __m256i*)(x3+base+32))),ones));
+ }
+ d0=hsum256_i32(A0); d1=hsum256_i32(A1);
+ d2=hsum256_i32(A2); d3=hsum256_i32(A3);
+#else
+ d0=d1=d2=d3=0;
+ for(int b=0;b>1);
+ for(int k=0;k<32;k++){
+ int32_t ul=(int32_t)(blk[k]&0xF), uh=(int32_t)(blk[k]>>4);
+ d0+=ul*x0[base+k]+uh*x0[base+k+32];
+ d1+=ul*x1[base+k]+uh*x1[base+k+32];
+ d2+=ul*x2[base+k]+uh*x2[base+k+32];
+ d3+=ul*x3[base+k]+uh*x3[base+k+32];
+ }
+ }
+#endif
+ a0=fmaf((float)(d0-8*g0[g]),scl[g],a0);
+ a1=fmaf((float)(d1-8*g1[g]),scl[g],a1);
+ a2=fmaf((float)(d2-8*g2[g]),scl[g],a2);
+ a3=fmaf((float)(d3-8*g3[g]),scl[g],a3);
+ }
+ if(g*gs>1];
+ int32_t u=(int32_t)((i&1)?(byte>>4):(byte&0xF));
+ d0+=u*x0[i]; d1+=u*x1[i]; d2+=u*x2[i]; d3+=u*x3[i];
+ }
+ a0=fmaf((float)(d0-8*g0[g]),scl[g],a0);
+ a1=fmaf((float)(d1-8*g1[g]),scl[g],a1);
+ a2=fmaf((float)(d2-8*g2[g]),scl[g],a2);
+ a3=fmaf((float)(d3-8*g3[g]),scl[g],a3);
+ }
+ y[(int64_t)s*O+o] =a0*sx[s];
+ y[(int64_t)(s+1)*O+o]=a1*sx[s+1];
+ y[(int64_t)(s+2)*O+o]=a2*sx[s+2];
+ y[(int64_t)(s+3)*O+o]=a3*sx[s+3];
+ }
+ for(; s>1)),
+ _mm512_loadu_si512((const void*)(xr+base)));
+ }
+ d=_mm512_reduce_add_epi32(acc);
+#else
+ for(int b=0;b>1);
+ const int8_t *xb=xr+base;
+#if defined(coli_dpbusd256)
+ __m256i lo,hi; i4p_blk256(blk,&lo,&hi);
+ __m256i acc=_mm256_setzero_si256();
+ acc=coli_dpbusd256(acc,lo,_mm256_loadu_si256((const __m256i*)xb));
+ acc=coli_dpbusd256(acc,hi,_mm256_loadu_si256((const __m256i*)(xb+32)));
+ d+=hsum256_i32(acc);
+#elif defined(__AVX2__)
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ const __m256i ones=_mm256_set1_epi16(1);
+ __m256i bb=_mm256_loadu_si256((const __m256i*)blk);
+ __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(bb,m4),
+ _mm256_loadu_si256((const __m256i*)xb));
+ __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(bb,4),m4),
+ _mm256_loadu_si256((const __m256i*)(xb+32)));
+ __m256i acc=_mm256_add_epi32(_mm256_madd_epi16(p0,ones),
+ _mm256_madd_epi16(p1,ones));
+ d+=hsum256_i32(acc);
+#else
+ for(int k=0;k<32;k++){
+ d+=(int32_t)(blk[k]&0xF)*xb[k];
+ d+=(int32_t)(blk[k]>>4)*xb[k+32];
+ }
+#endif
+ }
+#endif
+ a=fmaf((float)(d-8*xg[g]),scl[g],a);
+ }
+ if(g*gs>1];
+ d+=(int32_t)((i&1)?(byte>>4):(byte&0xF))*xr[i];
+ }
+ a=fmaf((float)(d-8*xg[g]),scl[g],a);
+ }
+ y[(int64_t)s*O+o]=a*sx[s];
+ }
+ }
+}
+
+static void matmul_i4p_grouped_idot(float *y, const int8_t *xq, const float *sx,
+ const int32_t *xsg, const uint8_t *q4,
+ const float *scale, int S, int I, int O, int gs){
+#if defined(COLI_HAVE_AMX_I4P)
+ /* AMX takes the aligned bulk (full 16-output tiles, full groups only —
+ * the real gs64 checkpoints have I%gs==0); the vector path finishes any
+ * output remainder. Below the row threshold the B-tile unpack does not
+ * amortize and the vector tile is the better kernel. */
+ if(coli_amx_ok() && S>=coli_amx_s_min && gs%64==0 && gs<=256 && I%gs==0 && O>=16){
+ int O16=O&~15;
+ matmul_i4p_gidot_amx(y,xq,sx,q4,scale,S,I,O16,O,gs);
+ if(O16
+ * bit-identico al path per-riga per associativita'.
+ * EN: 1x4 register tile — the weight block's load+mask cost is paid
+ * once per 4 activation rows. Integer sums: bit-identical to the
+ * per-row path by associativity. */
+ const __m256i m4t=_mm256_set1_epi8(0x0F);
+ for(;s+4<=S;s+=4){
+ const int8_t *x0=xq+(int64_t)s*I, *x1=x0+I, *x2=x1+I, *x3=x2+I;
+ __m256i a0=_mm256_setzero_si256(), a1=_mm256_setzero_si256();
+ __m256i a2=_mm256_setzero_si256(), a3=_mm256_setzero_si256();
+ int i=0;
+ for(;i+64<=I;i+=64){
+ __m256i b =_mm256_loadu_si256((const __m256i*)(w+(i>>1)));
+ __m256i lo=_mm256_and_si256(b,m4t);
+ __m256i hi=_mm256_and_si256(_mm256_srli_epi16(b,4),m4t);
+ a0=coli_dpbusd256(a0,lo,_mm256_loadu_si256((const __m256i*)(x0+i)));
+ a0=coli_dpbusd256(a0,hi,_mm256_loadu_si256((const __m256i*)(x0+i+32)));
+ a1=coli_dpbusd256(a1,lo,_mm256_loadu_si256((const __m256i*)(x1+i)));
+ a1=coli_dpbusd256(a1,hi,_mm256_loadu_si256((const __m256i*)(x1+i+32)));
+ a2=coli_dpbusd256(a2,lo,_mm256_loadu_si256((const __m256i*)(x2+i)));
+ a2=coli_dpbusd256(a2,hi,_mm256_loadu_si256((const __m256i*)(x2+i+32)));
+ a3=coli_dpbusd256(a3,lo,_mm256_loadu_si256((const __m256i*)(x3+i)));
+ a3=coli_dpbusd256(a3,hi,_mm256_loadu_si256((const __m256i*)(x3+i+32)));
+ }
+ int32_t d0=hsum256_i32(a0), d1=hsum256_i32(a1);
+ int32_t d2=hsum256_i32(a2), d3=hsum256_i32(a3);
+ /* coda a coppie, unsigned: stessa identita' -8*xsum del per-riga */
+ for(;i>1];
+ d0+=(int32_t)(byte&0xF)*x0[i]; d1+=(int32_t)(byte&0xF)*x1[i];
+ d2+=(int32_t)(byte&0xF)*x2[i]; d3+=(int32_t)(byte&0xF)*x3[i];
+ if(i+1>4)*x0[i+1]; d1+=(int32_t)(byte>>4)*x1[i+1];
+ d2+=(int32_t)(byte>>4)*x2[i+1]; d3+=(int32_t)(byte>>4)*x3[i+1];
+ }
+ }
+ y[(int64_t)(s+0)*O+o]=(float)(d0-8*xsum[s+0])*sc*sx[s+0];
+ y[(int64_t)(s+1)*O+o]=(float)(d1-8*xsum[s+1])*sc*sx[s+1];
+ y[(int64_t)(s+2)*O+o]=(float)(d2-8*xsum[s+2])*sc*sx[s+2];
+ y[(int64_t)(s+3)*O+o]=(float)(d3-8*xsum[s+3])*sc*sx[s+3];
+ }
+#endif
+ for(;s 0.0) return gb * 1e9;
+ static int noted = 0;
+ if (!noted) {
+ noted = 1;
+ fprintf(stderr, "[inkling] could not measure available RAM on this platform; "
+ "auto cache falls back to 16 experts/layer. Pass --cap to set it.\n");
+ }
return 0;
-#endif
}
/* ---------- routed-expert slots: serial bookkeeping, parallel fills ---------- */
@@ -2198,15 +2193,20 @@ static void apply_rep_penalty(float *logit, int n, const int *hist, int nhist, f
}
}
-/* reject a prompt that would overrun the served KV bound (CTX_MAX, default 8192).
- * The refusal is the frame the gateway turns into a 400 context_length_exceeded
- * (#506, #1381); free text here reached the client as a 500. One request is
- * served at a time, so the returned buffer is only read before the next call. */
+/* Refuse only a prompt that does not fit the served KV bound (CTX_MAX,
+ * default 8192). max_tokens is a ceiling: coli chat's interactive default
+ * (16384) used to 400 every turn because 2 + 16384 > 8192. The refusal is
+ * the frame the gateway turns into a 400 context_length_exceeded (#506,
+ * #1381); free text here reached the client as a 500. One request is served
+ * at a time, so the returned buffer is only read before the next call. */
+static int ink_ctx_max(void) {
+ const char *cm = getenv("CTX_MAX");
+ return cm ? atoi(cm) : 8192;
+}
static const char *prompt_reject(int np, int want) {
static char message[96];
- const char *cm = getenv("CTX_MAX");
- int ctx_max = cm ? atoi(cm) : 8192;
- if (np + want <= ctx_max) return NULL;
+ int ctx_max = ink_ctx_max();
+ if (coli_serve_budget(np, want, ctx_max, 0) >= 0) return NULL;
snprintf(message, sizeof(message),
"CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d",
np, want, ctx_max);
@@ -2323,8 +2323,19 @@ static int serve_one(Model *m, Tok *T, SReq *q) {
int *ids = malloc((size_t)cap * sizeof(int));
int np = tok_encode(T, q->payload, q->plen, ids, cap);
if (np <= 0) { coli_serve_write_error(stdout,q->id,"empty prompt"); free(ids); return 0; }
- const char *bad = prompt_reject(np, q->max_tok);
- if (bad) { coli_serve_write_error(stdout,q->id,bad); free(ids); return 0; }
+ int ctx_max = ink_ctx_max();
+ int budget = coli_serve_budget(np, q->max_tok, ctx_max, q->logprobs > 0);
+ if (budget < 0) {
+ const char *bad = prompt_reject(np, q->max_tok);
+ coli_serve_write_error(stdout,q->id,bad ? bad : "CONTEXT_EXCEEDED");
+ free(ids); return 0;
+ }
+ if (budget < q->max_tok) {
+ fprintf(stderr, "[serve] max_tokens %d clamped to %d (context %d - prompt %d); "
+ "raise CTX_MAX for longer answers\n",
+ q->max_tok, budget, ctx_max, np);
+ q->max_tok = budget;
+ }
/* audio: every <|audio|> placeholder must have exactly one DMel frame */
int naud = q->alen / m->c.mel_bins;
if (q->alen % m->c.mel_bins != 0 || audio_tok_count(m, ids, np) != naud) {
@@ -2482,6 +2493,25 @@ static void serve_hwinfo(Model *m) {
if (sscanf(ln, "MemTotal: %lf", &v) == 1) rt = v/1e6;
if (sscanf(ln, "MemAvailable: %lf", &v) == 1) ra = v/1e6;
} fclose(mi); }
+ if (rt <= 0.0 || ra <= 0.0) {
+ double t2 = 0, a2 = 0;
+ compat_meminfo_gb(&t2, &a2);
+ if (rt <= 0.0) rt = t2;
+ if (ra <= 0.0) ra = a2;
+ }
+#ifdef _WIN32
+ if (cores <= 0) {
+ SYSTEM_INFO si;
+ GetSystemInfo(&si);
+ cores = (int)si.dwNumberOfProcessors;
+ }
+#endif
+#ifdef __APPLE__
+ if (!cpu[0]) {
+ size_t sl = sizeof(cpu);
+ if (sysctlbyname("machdep.cpu.brand_string", cpu, &sl, NULL, 0)) cpu[0] = 0;
+ }
+#endif
int ngpu = 0; double vram = 0;
const char *gpu = "";
#ifdef COLI_CUDA
diff --git a/c/json.h b/c/json.h
index 649b92cae..3a938b67a 100644
--- a/c/json.h
+++ b/c/json.h
@@ -28,6 +28,7 @@ typedef struct {
size_t acap, aoff;
int depth; /* annidamento corrente: bound contro lo stack-overflow
* da JSON malevolo tipo [[[[...]]]] (discesa ricorsiva) */
+ int error; /* used by json_parse_checked; legacy parsing stays permissive */
} jparser;
/* tetto di annidamento: gli header safetensors / config sono piatti (profondita'
@@ -43,7 +44,13 @@ static char *j_dup(jparser *p, const char *b, int n) {
return d;
}
-static void j_ws(jparser *p) { while (*p->s && isspace((unsigned char)*p->s)) p->s++; }
+static void j_ws(jparser *p) {
+ while (*p->s && isspace((unsigned char)*p->s)) {
+ char c=*p->s++;
+ /* Preserve legacy consumption, but only JSON whitespace is valid. */
+ if (c!=' ' && c!='\t' && c!='\r' && c!='\n') p->error=1;
+ }
+}
static jval *j_new(jtype t) {
jval *v = (jval *)calloc(1, sizeof(jval));
@@ -52,12 +59,12 @@ static jval *j_new(jtype t) {
static jval *j_parse_val(jparser *p);
-static char *j_parse_str_raw(jparser *p) {
+static char *j_parse_str_raw(jparser *p, int is_key) {
/* SEC (GHSA-2qrj): fail closed if not actually at a quote. The old comment
* "assume *p->s == '\"'" was violated on the object-key path, and the
* unconditional p->s++ would step past the buffer's NUL terminator and scan
* adjacent heap (OOB read leaking into tensor names). */
- if (*p->s != '"') return j_dup(p, "", 0);
+ if (*p->s != '"') { p->error = 1; return j_dup(p, "", 0); }
p->s++;
/* buffer su heap che CRESCE: niente troncamento silenzioso a 64KB (le stringhe
* lunghe di tokenizer.json/config venivano tagliate) e niente 64KB di stack. */
@@ -67,6 +74,7 @@ static char *j_parse_str_raw(jparser *p) {
if (!tmp) { fprintf(stderr, "OOM parsing JSON string\n"); exit(1); } } tmp[n++] = (char)(ch); }while(0)
while (*p->s && *p->s != '"') {
char c = *p->s++;
+ if ((unsigned char)c < 0x20) p->error = 1;
if (c == '\\' && *p->s) {
char e = *p->s++;
switch (e) {
@@ -75,9 +83,12 @@ static char *j_parse_str_raw(jparser *p) {
case 'f': c = '\f'; break; case '/': c = '/'; break;
case '\\': c = '\\'; break; case '"': c = '"'; break;
case 'u': { /* \uXXXX -> codepoint UTF-8 (con coppie surrogate) */
- if (!p->s[0]||!p->s[1]||!p->s[2]||!p->s[3]) { c='?'; break; } /* \u troncato: non leggere oltre il NUL */
+ if (!p->s[0]||!p->s[1]||!p->s[2]||!p->s[3]) { p->error=1; c='?'; break; } /* \u troncato: non leggere oltre il NUL */
+ for (int i=0; i<4; i++) if (!isxdigit((unsigned char)p->s[i])) p->error=1;
unsigned cp = (unsigned)strtoul((char[]){p->s[0],p->s[1],p->s[2],p->s[3],0}, NULL, 16);
p->s += 4;
+ /* Checked keys must retain their identity in C-string lookups. */
+ if (is_key && cp==0) p->error=1;
if (cp >= 0xD800 && cp <= 0xDBFF && p->s[0]=='\\' && p->s[1]=='u'
&& p->s[2] && p->s[3] && p->s[4] && p->s[5]) {
unsigned lo = (unsigned)strtoul((char[]){p->s[2],p->s[3],p->s[4],p->s[5],0}, NULL, 16);
@@ -89,13 +100,13 @@ static char *j_parse_str_raw(jparser *p) {
else { J_PUT(0xF0|(cp>>18)); J_PUT(0x80|((cp>>12)&0x3F)); J_PUT(0x80|((cp>>6)&0x3F)); J_PUT(0x80|(cp&0x3F)); }
continue;
}
- default: c = e; break;
+ default: p->error = 1; c = e; break;
}
}
J_PUT(c);
}
#undef J_PUT
- if (*p->s == '"') p->s++;
+ if (*p->s == '"') p->s++; else p->error = 1;
char *out = j_dup(p, tmp, (int)n); free(tmp);
return out;
}
@@ -103,9 +114,9 @@ static char *j_parse_str_raw(jparser *p) {
static jval *j_parse_val(jparser *p) {
j_ws(p);
char c = *p->s;
- if (c == '"') { jval *v = j_new(J_STR); v->str = j_parse_str_raw(p); return v; }
+ if (c == '"') { jval *v = j_new(J_STR); v->str = j_parse_str_raw(p,0); return v; }
if (c == '{') {
- if (++p->depth > J_MAX_DEPTH) { p->depth--; return j_new(J_NULL); }
+ if (++p->depth > J_MAX_DEPTH) { p->error=1; p->depth--; return j_new(J_NULL); }
p->s++; jval *v = j_new(J_OBJ);
int cap = 8;
v->keys = malloc(cap * sizeof(char*));
@@ -116,9 +127,9 @@ static jval *j_parse_val(jparser *p) {
if (*p->s == '}') { p->s++; p->depth--; return v; }
for (;;) {
j_ws(p);
- if (*p->s != '"') break; /* SEC (GHSA-2qrj): object key must be a quoted string; stop on malformed input */
- char *key = j_parse_str_raw(p);
- j_ws(p); if (*p->s == ':') p->s++;
+ if (*p->s != '"') { p->error=1; break; } /* SEC (GHSA-2qrj): object key must be a quoted string; stop on malformed input */
+ char *key = j_parse_str_raw(p,1);
+ j_ws(p); if (*p->s == ':') p->s++; else p->error=1;
jval *val = j_parse_val(p);
if (v->len == cap) { cap *= 2;
char **nk = (char**)realloc(v->keys, cap*sizeof(char*));
@@ -131,13 +142,14 @@ static jval *j_parse_val(jparser *p) {
j_ws(p);
if (*p->s == ',') { p->s++; continue; }
if (*p->s == '}') { p->s++; break; }
+ p->error = 1;
break;
}
p->depth--;
return v;
}
if (c == '[') {
- if (++p->depth > J_MAX_DEPTH) { p->depth--; return j_new(J_NULL); }
+ if (++p->depth > J_MAX_DEPTH) { p->error=1; p->depth--; return j_new(J_NULL); }
p->s++; jval *v = j_new(J_ARR);
int cap = 8;
v->kids = malloc(cap * sizeof(jval*));
@@ -154,6 +166,7 @@ static jval *j_parse_val(jparser *p) {
j_ws(p);
if (*p->s == ',') { p->s++; continue; }
if (*p->s == ']') { p->s++; break; }
+ p->error = 1;
break;
}
p->depth--;
@@ -163,12 +176,28 @@ static jval *j_parse_val(jparser *p) {
if (c == 'f' && !strncmp(p->s, "false", 5)) { p->s += 5; jval *v = j_new(J_BOOL); v->boolean = 0; return v; }
if (c == 'n' && !strncmp(p->s, "null", 4)) { p->s += 4; return j_new(J_NULL); }
/* numero */
- { char *end; double d = strtod(p->s, &end); p->s = end; jval *v = j_new(J_NUM); v->num = d; return v; }
+ { const char *start=p->s, *q=start;
+ if (*q=='-') q++;
+ if (*q=='0') q++;
+ else if (*q>='1' && *q<='9') { do { q++; } while (*q>='0' && *q<='9'); }
+ else p->error=1;
+ if (*q=='.') {
+ q++; if (*q<'0' || *q>'9') p->error=1;
+ while (*q>='0' && *q<='9') q++;
+ }
+ if (*q=='e' || *q=='E') {
+ q++; if (*q=='+' || *q=='-') q++;
+ if (*q<'0' || *q>'9') p->error=1;
+ while (*q>='0' && *q<='9') q++;
+ }
+ char *end; double d = strtod(start, &end);
+ if (end==start || end!=q) p->error=1;
+ p->s = end; jval *v = j_new(J_NUM); v->num = d; return v; }
}
/* API */
static jval *json_parse(const char *text, char **arena_out) {
- jparser p = { text, NULL, 0, 0, 0 };
+ jparser p = { text, NULL, 0, 0, 0, 0 };
jval *v = j_parse_val(&p);
if (arena_out) *arena_out = p.arena; else free(p.arena);
return v;
@@ -195,4 +224,16 @@ static void json_free(jval *v) {
free(v);
}
+/* An oracle must not accept a partial tree from a truncated/malformed file.
+ * Keep the existing API's permissive behavior for other engine consumers. */
+static jval *json_parse_checked(const char *text) {
+ if (!text) return NULL;
+ jparser p = { text, NULL, 0, 0, 0, 0 };
+ jval *v = j_parse_val(&p);
+ j_ws(&p);
+ if (p.error || *p.s) { json_free(v); v=NULL; }
+ free(p.arena);
+ return v;
+}
+
#endif
diff --git a/c/kimi_k3.c b/c/kimi_k3.c
index 5543ad615..054b0e0d4 100644
--- a/c/kimi_k3.c
+++ b/c/kimi_k3.c
@@ -117,6 +117,7 @@
#include "pin_pool.h" /* coli_pin_slots_wanted: quanti scatti tenere */
#include "hybrid_split.h" /* KV prefix reuse (shared) */
#include "serve_codec.h"
+#include "serve_budget.h"
#ifdef COLI_SEGMENT_ADAPTER
#include "segment_runtime.h"
#include "segment_adapters.h"
@@ -189,7 +190,9 @@ typedef struct {
/* ---------- routed-expert streaming (native MXFP4 from the HF shards) ---- */
typedef struct { int fd[6]; int64_t off[6]; int contig; } ERef; /* w1p w1s w2p w2s w3p w3s */
+#ifndef KIMI_K3_NO_MAIN
static char g_k3_usage[2100]; /* /.coli_usage, or COLI_USAGE */
+#endif
typedef struct { int eid; uint8_t *buf, *base; uint64_t used; int pinned; } Slot;
/* pinned: seeded from .coli_usage at startup and never evicted. The LRU adapts to
* THIS session; the pin knows the history of every session before it. Capped at
@@ -1612,28 +1615,22 @@ static inline float situf_(float g, float u, float b1, float b2){
}
#ifdef COLI_CUDA
-/* CUDA apply for one expert, decode only (S==1).
- *
- * Same shape as the Vulkan path below and the CPU expert_apply above -- w1/w3,
- * SiTU-GLU on the host, then w2 down -- but stateless: the routed tier streams,
- * so there is nothing resident to keep a device handle for. Weights ride up
- * with the call.
- *
- * Returns 0 with u untouched on ANY failure, so the caller falls through to the
- * disk+CPU path exactly as it does when Vulkan declines. That is the contract
- * vLLM's MXFP4 backends use too -- FlashInfer/AITER when they can, an emulation
- * path when they cannot -- and it is what makes the fast path safe to attempt
- * unconditionally. */
+/* Keep SiTU-GLU and intermediate activations on the GPU when supported.
+ * Older DLLs lack the optional fused entry point, so retain the three-matmul
+ * path as fallback. Accumulate into u only after a complete expert succeeds. */
static int cuda_expert_apply(Model *m, const uint8_t *w1p, const uint8_t *w1s,
const uint8_t *w2p, const uint8_t *w2s,
const uint8_t *w3p, const uint8_t *w3s,
const float *z, float wk,
float *u, float *gate, float *up, float *hz){
Cfg *c=&m->c;
- if(!coli_cuda_matmul_mxfp4(gate,z,w1p,w1s,1,c->latent,c->moe_inter)) return 0;
- if(!coli_cuda_matmul_mxfp4(up, z,w3p,w3s,1,c->latent,c->moe_inter)) return 0;
- for(int i=0;imoe_inter;i++) gate[i]=situf_(gate[i],up[i],c->situ_b1,c->situ_b2);
- if(!coli_cuda_matmul_mxfp4(hz,gate,w2p,w2s,1,c->moe_inter,c->latent)) return 0;
+ if(!coli_cuda_expert_mxfp4(hz,z,w1p,w1s,w3p,w3s,w2p,w2s,
+ 1,c->latent,c->moe_inter,c->situ_b1,c->situ_b2)) {
+ if(!coli_cuda_matmul_mxfp4(gate,z,w1p,w1s,1,c->latent,c->moe_inter)) return 0;
+ if(!coli_cuda_matmul_mxfp4(up, z,w3p,w3s,1,c->latent,c->moe_inter)) return 0;
+ for(int i=0;imoe_inter;i++) gate[i]=situf_(gate[i],up[i],c->situ_b1,c->situ_b2);
+ if(!coli_cuda_matmul_mxfp4(hz,gate,w2p,w2s,1,c->moe_inter,c->latent)) return 0;
+ }
for(int i=0;ilatent;i++) u[i]+=wk*hz[i];
return 1;
}
@@ -2370,7 +2367,8 @@ static int sample_tok(const float *lo, int V, float temp, float top_p){
* (K3_THINK=0 opens directly = non-thinking mode). The model then
* closes think, opens response, and finishes with <|end_of_msg|> (the eos). */
typedef struct { Tok *T; int *ids; int n, cap;
- int sp_open, sp_close, sp_sep, sp_eom; } ChatB;
+ int sp_open, sp_close, sp_sep, sp_eom;
+ int cont; } ChatB; /* cont: the final turn was left open (continuation) */
static void cb_special(ChatB *b, int id){
if(b->n>=b->cap){ fprintf(stderr,"chat prompt too long\n"); exit(1); }
b->ids[b->n++]=id;
@@ -2398,7 +2396,7 @@ static int chat_build(Tok *T, const char *sys, const char *user, int thinking,
int *ids, int cap, int *sp){
ChatB b={T,ids,0,cap,
chat_special(T,"<|open|>"), chat_special(T,"<|close|>"),
- chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>")};
+ chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>"), 0};
if(b.sp_open<0||b.sp_close<0||b.sp_sep<0||b.sp_eom<0){
fprintf(stderr,"chat: XTML special tokens not in tokenizer.json\n"); exit(1); }
sp[0]=b.sp_open; sp[1]=b.sp_close; sp[2]=b.sp_sep; sp[3]=b.sp_eom;
@@ -2475,7 +2473,7 @@ static int chat_build_wire(Tok *T, const char *wire, int nwire, int *thinking,
int *ids, int cap, int *sp){
ChatB b={T,ids,0,cap,
chat_special(T,"<|open|>"), chat_special(T,"<|close|>"),
- chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>")};
+ chat_special(T,"<|sep|>"), chat_special(T,"<|end_of_msg|>"), 0};
/* -2, not -1: the caller must be able to tell a bad payload from a snapshot
* whose tokenizer has no XTML tokens. Serve reported both as "invalid K3
* chat payload", which sent at least one user hunting through a request
@@ -2489,6 +2487,7 @@ static int chat_build_wire(Tok *T, const char *wire, int nwire, int *thinking,
while(p \n -- the same
+ * body as A, but rendered as the trailing open turn (no
+ * <|close|>, no <|end_of_msg|>, and the caller appends no
+ * fresh generation cue). An old engine has no 'C' record:
+ * it falls to the 'M' branch below, fails to parse, and
+ * rejects the payload -- fail-closed, never miswired. */
+ int nr=-1, nt=-1;
+ if(sscanf(p,"C %d %d",&nr,&nt)!=2||nr<0||nt<0||nl+1+nr+nt>end) return -1;
+ char *reason=malloc((size_t)nr+1), *text=malloc((size_t)nt+1);
+ if(!reason||!text){ fprintf(stderr,"OOM chat continuation\n"); exit(1); }
+ memcpy(reason,nl+1,(size_t)nr); reason[nr]=0;
+ memcpy(text,nl+1+nr,(size_t)nt); text[nt]=0;
+ cb_open(&b,"message","assistant");
+ if(nr){ cb_open(&b,"think",NULL); cb_text(&b,reason); cb_close(&b,"think"); }
+ cb_open(&b,"response",NULL); cb_text(&b,text); /* OPEN: no close, no eom */
+ b.cont=1;
+ free(reason); free(text); p=nl+1+nr+nt; continue;
+ }
if(*p=='Y'){ /* typed system message (#1143): tool-declare / tool-choice */
int ntp=-1, nb=-1;
if(sscanf(p,"Y %d %d",&ntp,&nb)!=2||ntp<1||ntp>64||nb<0||nl+1+ntp+nb>end) return -1;
@@ -2601,8 +2619,10 @@ static int chat_build_wire(Tok *T, const char *wire, int nwire, int *thinking,
chat_message(&b,r,text,!strcmp(r,"assistant"));
free(text); p=nl+1+nb;
}
- cb_open(&b,"message","assistant");
- cb_open(&b,*thinking?"think":"response",NULL);
+ if(!b.cont){ /* a continuation already emitted the open final turn */
+ cb_open(&b,"message","assistant");
+ cb_open(&b,*thinking?"think":"response",NULL);
+ }
return b.n;
}
@@ -3053,12 +3073,24 @@ static int serve_one(Model *m, Tok *T, ServeReq *q){
np+=tok_encode(T,q->payload,q->plen,ids+np,cap-np);
}
int max_ctx=getenv("K3_MAXT")?atoi(getenv("K3_MAXT")):8192;
- if(np<1||(int64_t)np+q->max_tok>max_ctx){ /* SEC (GHSA-gf38): int64 so np+max_tok can't wrap negative */
- char message[160];
- snprintf(message,sizeof(message),
- "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d",
- np,q->max_tok,max_ctx);
- coli_serve_write_error(stdout,q->id,message); free(ids); return 0;
+ int budget=coli_serve_budget(np,q->max_tok,max_ctx,q->logprobs>0);
+ if(budget<0){
+ if(np<1){
+ coli_serve_write_error(stdout,q->id,"EMPTY_PROMPT");
+ }else{
+ char message[160];
+ snprintf(message,sizeof(message),
+ "CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d",
+ np,q->max_tok,max_ctx);
+ coli_serve_write_error(stdout,q->id,message);
+ }
+ free(ids); return 0;
+ }
+ if(budgetmax_tok){
+ fprintf(stderr,"[serve] max_tokens %d clamped to %d (context %d - prompt %d); "
+ "raise K3_MAXT for longer answers\n",
+ q->max_tok,budget,max_ctx,np);
+ q->max_tok=budget;
}
coli_serve_write_accept(stdout,q->id,np);
/* Declare the structured sideband before any generated DATA. Even an
diff --git a/c/olmoe.c b/c/olmoe.c
index 01f929690..d56b606d1 100644
--- a/c/olmoe.c
+++ b/c/olmoe.c
@@ -18,6 +18,11 @@
* PILOT_EVICT_GUARD=0/1 : 1=enable LFRU prefetch eviction guard (default), 0=disable
* EXPERT_DROP=0/1: 1=fadvise(DONTNEED) after each expert read (old behaviour,
* for RAM-tight boxes); 0=keep pages cached (default)
+ * ROUTE_TRACE=: log every routing decision (one line per moe call,
+ * position and layer: " : ...")
+ * for offline analysis — tools/route_pairs.py,
+ * tools/route_coupling_report.py, tools/residency_sim.py.
+ * Measurement only: it cannot change which experts run.
* (expert queue is sorted by eid for SSD read locality)
*/
#define _GNU_SOURCE
@@ -41,6 +46,7 @@
#include "kv_prefix.h"
#include "pin_pool.h" /* piu scatti annidati */ /* riuso del prefisso tra turni (shared) */
#include "serve_codec.h"
+#include "serve_budget.h"
#ifdef COLI_SEGMENT_ADAPTER
#include "segment_runtime.h"
#include "segment_adapters.h"
@@ -127,6 +133,7 @@ typedef struct {
} Model;
static pthread_mutex_t g_pilot_mx = PTHREAD_MUTEX_INITIALIZER;
+static pthread_cond_t g_pilot_cv = PTHREAD_COND_INITIALIZER; /* broadcast on every publish */
static struct { int l, e; } pilot_q[4096];
static volatile unsigned pilot_r = 0, pilot_w = 0;
static Model *pilot_m = NULL;
@@ -190,6 +197,28 @@ static void cache_publish(Model *m, int layer, Slot *s, int eid) {
lc->slot_by_expert[eid] = (int)(s - lc->slots);
}
+/* A slot being read keeps the index entry of the expert it is loading, marked
+ * -(eid+2) as in colibri.c's ecache_reserve: lookups still miss and eviction
+ * still skips it (eid < 0), but a second loader of the same expert can see the
+ * read in flight instead of starting another one into another slot. */
+static void cache_reserve(Model *m, int layer, Slot *s, int eid) {
+ LCache *lc = &m->cache[layer];
+ cache_hide(m, layer, s);
+ s->eid = -(eid + 2);
+ if (lc->slot_by_expert && eid >= 0 && eid < m->c.n_experts)
+ lc->slot_by_expert[eid] = (int)(s - lc->slots);
+}
+
+/* Caller holds g_pilot_mx. */
+static int slot_in_flight(Model *m, int layer, int eid) {
+ if (layer < 0 || layer >= m->c.n_layers || eid < 0 ||
+ eid >= m->c.n_experts) return 0;
+ LCache *lc = &m->cache[layer];
+ if (!lc->slot_by_expert) return 0;
+ int i = lc->slot_by_expert[eid];
+ return i >= 0 && i < lc->n && lc->slots[i].eid == -(eid + 2);
+}
+
static void ensure_pilot_worker_started(Model *m) {
if (!pilot_m) {
pilot_m = m;
@@ -318,6 +347,29 @@ static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) {
return _mm_cvtsi128_si32(sum32);
}
#define HAVE_FAST_DOT_I8 1
+#elif defined(__SSE4_1__)
+#include
+#include "sse41_kernels.h"
+/* Sandy Bridge-EP path: AVX 1.0 only, no FMA, no AVX-2.
+ * 16 int8 dot via two 8-wide SSE2 sign-extend + SSE4.1 madd pairs.
+ * Bit-for-bit identical to the AVX2 version above (just 2x 128-bit ops
+ * instead of 1x 256-bit op). NO FMA here -- this branch targets Sandy Bridge
+ * which has no FMA -- so use explicit mul+add for the inner accumulation. */
+static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) {
+ __m128i va_lo = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)a)); /* lower 8 int8 -> 8 int16 */
+ __m128i vb_lo = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)b));
+ __m128i va_hi = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)(a + 8))); /* upper 8 int8 -> 8 int16 */
+ __m128i vb_hi = _mm_cvtepi8_epi16(_mm_loadu_si128((const __m128i*)(b + 8)));
+ __m128i p_lo = _mm_madd_epi16(va_lo, vb_lo); /* 4 x int32 from 8 int16 pairs */
+ __m128i p_hi = _mm_madd_epi16(va_hi, vb_hi); /* 4 x int32 from 8 int16 pairs */
+ __m128i sum = _mm_add_epi32(p_lo, p_hi);
+ /* horizontal reduce 4 x int32 -> 1 x int32 */
+ __m128i hi64 = _mm_unpackhi_epi64(sum, sum);
+ __m128i sum64 = _mm_add_epi32(sum, hi64);
+ __m128i hi32 = _mm_shuffle_epi32(sum64, _MM_SHUFFLE(2, 3, 0, 1));
+ return _mm_cvtsi128_si32(_mm_add_epi32(sum64, hi32));
+}
+#define HAVE_FAST_DOT_I8 1
#endif
/* Test-only hook, compiled out of the shipping binary.
*
@@ -634,7 +686,16 @@ static void slot_ensure_allocated(Model *m, Slot *s) {
s->pinned = 0;
}
+#ifdef COLI_CACHE_INDEX_TEST
+/* Model-free tests stand in for the disk read, so they can hold a load open
+ * and count how many times each expert is read. */
+static void (*g_test_expert_load)(Model *m, int layer, int eid, Slot *s);
+#endif
+
static void load_expert_merged(Model *m, int layer, int eid, Slot *s) {
+#ifdef COLI_CACHE_INDEX_TEST
+ if (g_test_expert_load) { g_test_expert_load(m, layer, eid, s); return; }
+#endif
char nm[256], qsnm[256];
snprintf(nm, sizeof(nm), "model.layers.%d.mlp.experts.%d.merged_weight", layer, eid);
snprintf(qsnm, sizeof(qsnm), "model.layers.%d.mlp.experts.%d.qs", layer, eid);
@@ -681,6 +742,13 @@ static void expert_get(Model *m, int layer, int eid, Slot **out) {
pthread_mutex_lock(&g_pilot_mx);
ehit_mark(m, layer, eid); /* under the lock: the routing loop is parallel */
Slot *hit = slot_indexed(m, layer, eid);
+ /* The prefetcher usually reads the next layer's experts while this one
+ * computes, so a routed expert is often already on its way: wait for that
+ * read to publish rather than read the same bytes again into another slot. */
+ while (!hit && slot_in_flight(m, layer, eid)) {
+ pthread_cond_wait(&g_pilot_cv, &g_pilot_mx);
+ hit = slot_indexed(m, layer, eid);
+ }
if (hit) {
m->hits++; hit->used = ++m->clock; *out = hit;
if (m->last_access) m->last_access[layer * m->c.n_experts + eid] = m->clock;
@@ -727,7 +795,7 @@ static void expert_get(Model *m, int layer, int eid, Slot **out) {
s = &lc->slots[lru];
s->pinned = 0;
}
- cache_hide(m, layer, s);
+ cache_reserve(m, layer, s, eid);
s->used = ++m->clock;
pthread_mutex_unlock(&g_pilot_mx);
@@ -739,6 +807,7 @@ static void expert_get(Model *m, int layer, int eid, Slot **out) {
s->used = ++m->clock;
if (m->last_access) m->last_access[layer * c->n_experts + eid] = m->clock;
*out = s;
+ pthread_cond_broadcast(&g_pilot_cv);
pthread_mutex_unlock(&g_pilot_mx);
}
@@ -917,11 +986,24 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
idx[kk] = best; val[kk] = pr[best];
}
if (c->norm_topk) { float sm=0; for(int kk=0;kkhot_pinned && m->freq) {
- uint32_t *freq_l = m->freq[layer];
- if (freq_l) for (int kk = 0; kk < K; kk++) if (idx[kk] >= 0) freq_l[idx[kk]]++;
- }
+ /* IMPROVEMENT 2 activation heatmap AND the ROUTE_TRACE stream, in one
+ * call. The counters were the only thing this engine recorded, and it
+ * recorded them HERE, before pinning activates — rt_count keeps that
+ * placement exactly. The trace is the half olmoe never had: it emits a
+ * line per (moe call, position, layer), so tools/route_pairs.py,
+ * route_coupling_report.py and residency_sim.py can read this engine's
+ * routing the same way they read GLM's. Until now olmoe announced
+ * ROUTE_TRACE at startup and then wrote a zero-byte file, because
+ * rt_init() opens the stream but nothing here ever called rt_trace():
+ * every consumer silently saw "no data" instead of an error.
+ *
+ * Only rt_route() is unconditional: it is a no-op for the counts when
+ * this engine has no counter row (the !hot_pinned guard below is
+ * unchanged) and a no-op for the trace when ROUTE_TRACE is unset, so a
+ * run without the variable behaves exactly as before. Measurement only,
+ * never the computation: idx[] and val[] are the ids and the
+ * post-normalisation gates the layer is about to apply. */
+ if (!m->hot_pinned) rt_route(layer, s, idx, val, K);
const float *xs = x + (int64_t)s*D;
for (int kk = 0; kk < K; kk++) {
Slot *e; expert_get(m, layer, idx[kk], &e);
@@ -950,6 +1032,15 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
}
}
free(logits); free(g); free(u); free(hh);
+ /* Advance the trace call counter: once per moe() invocation, after all of
+ * its rows are traced. rt_trace_end() is a no-op when no stream is open.
+ *
+ * Outside the row loop on purpose. A batch of S == 0 traces no rows and
+ * must still consume a call id, or the ids stop being consecutive and
+ * residency_sim.py rejects the trace outright ("trace lacks advancing GLM
+ * call ids") rather than merging two forwards into one position space. GLM
+ * and glm53 advance theirs the same way. */
+ rt_trace_end();
}
/* PROF phases (#1449): wall time in attention, in the MoE blocks (expert
@@ -1076,7 +1167,7 @@ static void pilot_realload(Model *m, int layer, int eid) {
pthread_mutex_unlock(&g_pilot_mx);
return;
}
- if (slot_indexed(m, layer, eid)) {
+ if (slot_indexed(m, layer, eid) || slot_in_flight(m, layer, eid)) {
m->is_queued[layer * c->n_experts + eid] = 0;
pthread_mutex_unlock(&g_pilot_mx);
return;
@@ -1113,7 +1204,7 @@ static void pilot_realload(Model *m, int layer, int eid) {
s = &lc->slots[lru]; s->pinned = 0;
}
- cache_hide(m, layer, s); s->used = ++m->clock;
+ cache_reserve(m, layer, s, eid); s->used = ++m->clock;
pthread_mutex_unlock(&g_pilot_mx);
load_expert_merged(m, layer, eid, s);
@@ -1124,6 +1215,7 @@ static void pilot_realload(Model *m, int layer, int eid) {
s->used = ++m->clock;
if (m->last_access) m->last_access[layer * c->n_experts + eid] = m->clock;
m->is_queued[layer * c->n_experts + eid] = 0;
+ pthread_cond_broadcast(&g_pilot_cv);
pthread_mutex_unlock(&g_pilot_mx);
}
@@ -1521,7 +1613,8 @@ static int serve_one(Model *m, Tok *T, SReq *q, int ctx_cap) {
int *ids = malloc((size_t)cap * sizeof(int));
int np = tok_encode(T, q->payload, q->plen, ids, cap);
if (np <= 0) { coli_serve_write_error(stdout, q->id, "empty prompt"); free(ids); return 0; }
- if (np + q->max_tok > ctx_cap) {
+ int budget = coli_serve_budget(np, q->max_tok, ctx_cap, q->logprobs > 0);
+ if (budget < 0) {
char message[128];
/* The frame the gateway turns into a 400 context_length_exceeded
* (#506, #1381). Free text here reached the client as a 500. */
@@ -1530,6 +1623,12 @@ static int serve_one(Model *m, Tok *T, SReq *q, int ctx_cap) {
np, q->max_tok, ctx_cap);
coli_serve_write_error(stdout, q->id, message); free(ids); return 0;
}
+ if (budget < q->max_tok) {
+ fprintf(stderr, "[serve] max_tokens %d clamped to %d (context %d - prompt %d); "
+ "raise CTX for longer answers\n",
+ q->max_tok, budget, ctx_cap, np);
+ q->max_tok = budget;
+ }
g_temp = q->temp; g_nuc = q->top_p;
/* A chat client resends the whole transcript every turn. If this prompt
* begins with the ids the cache was built from, those keys and values ARE
diff --git a/c/openai_server.py b/c/openai_server.py
index 42fe36354..488aefcec 100644
--- a/c/openai_server.py
+++ b/c/openai_server.py
@@ -110,6 +110,8 @@ def _engine_error(fields, message):
class GenerationScheduler:
"""Bounded FIFO admission for the engine's independent KV contexts."""
+ _buckets = (0.001, 0.01, 0.05, 0.1, 0.5, 1, 5, 10, 30, 60, 300, math.inf)
+
def __init__(self, max_queue=8, queue_timeout=300, capacity=1):
if max_queue < 0:
raise ValueError("max_queue cannot be negative")
@@ -127,9 +129,13 @@ def __init__(self, max_queue=8, queue_timeout=300, capacity=1):
self.closed = False
self.admitted = 0
self.completed = 0
+ self.failed = 0
self.rejected = 0
self.timed_out = 0
self.cancelled = 0
+ self.timings = {name: {"sum": 0.0, "buckets": [0] * len(self._buckets)}
+ for name in ("queue_wait_seconds", "slot_duration_seconds",
+ "first_output_seconds", "engine_call_seconds")}
@contextlib.contextmanager
def admit(self, cancelled=None, slot=None):
@@ -140,7 +146,7 @@ def admit(self, cancelled=None, slot=None):
if self.closed:
raise APIError(503, "The inference scheduler is shutting down.", None,
"scheduler_closed", "server_error")
- if (self.active >= self.capacity or self.queue) and len(self.queue) >= self.max_queue:
+ if self._available_slot(slot) is None and len(self.queue) >= self.max_queue:
self.rejected += 1
raise APIError(429, "The inference queue is full.", None, "queue_full",
"rate_limit_error", {"Retry-After": "1"})
@@ -152,24 +158,6 @@ def admit(self, cancelled=None, slot=None):
self.condition.notify_all()
raise APIError(503, "The inference scheduler is shutting down.", None,
"scheduler_closed", "server_error")
- available = min(self.free_slots) if slot is None and self.free_slots else slot
- # (#B2) Admit as soon as our target slot is free AND no strictly-earlier
- # waiter also wants it (an earlier waiter "wants" it if it is any-slot or
- # pinned to the same slot). This replaces the old strict FIFO-head rule,
- # which let a head pinned to a busy slot block every request behind it —
- # even ones targeting a currently-free slot (head-of-line blocking).
- # ponytail: O(queue) scan per wakeup — negligible at the default max_queue;
- # switch to per-slot wait sets if max_queue is ever raised to thousands.
- can_admit = available in self.free_slots
- if can_admit:
- for t2, s2 in self.queue:
- if t2 is ticket:
- break
- if s2 is None or s2 == available:
- can_admit = False
- break
- if can_admit:
- break
if cancelled and cancelled():
self.queue.remove(entry)
self.cancelled += 1
@@ -182,37 +170,100 @@ def admit(self, cancelled=None, slot=None):
self.condition.notify_all()
raise APIError(429, "Timed out waiting for the inference engine.", None,
"queue_timeout", "rate_limit_error", {"Retry-After": "1"})
+ available = self._available_slot(slot, ticket)
+ if available is not None:
+ break
self.condition.wait(min(remaining, 0.25))
self.queue.remove(entry)
self.free_slots.remove(available)
self.active += 1
self.admitted += 1
- wait_seconds = time.monotonic() - queued_at
- cancelled_after_admission = False
+ admitted_at = time.monotonic()
+ wait_seconds = admitted_at - queued_at
+ self._observe("queue_wait_seconds", wait_seconds)
+ outcome = "failed"
try:
yield wait_seconds, available
+ outcome = "completed"
except ClientCancelled:
- cancelled_after_admission = True
+ outcome = "cancelled"
raise
finally:
with self.condition:
self.active -= 1
self.free_slots.add(available)
- if cancelled_after_admission:
- self.cancelled += 1
- else:
- self.completed += 1
+ setattr(self, outcome, getattr(self, outcome) + 1)
+ self._observe("slot_duration_seconds", time.monotonic() - admitted_at)
self.condition.notify_all()
+ def _available_slot(self, slot, ticket=None):
+ # Caller holds the condition lock. Pinned waiters reserve only their
+ # target; an older any-slot waiter has priority over every free slot.
+ candidates = self.free_slots.copy() if slot is None else self.free_slots & {slot}
+ for earlier_ticket, earlier_slot in self.queue:
+ if earlier_ticket is ticket:
+ break
+ if earlier_slot is None:
+ return None
+ candidates.discard(earlier_slot)
+ return min(candidates, default=None)
+
def snapshot(self):
with self.condition:
return {"active": self.active, "queued": len(self.queue),
"capacity": self.capacity,
"max_queue": self.max_queue, "queue_timeout_seconds": self.queue_timeout,
- "admitted": self.admitted, "completed": self.completed,
+ "admitted": self.admitted, "completed": self.completed, "failed": self.failed,
"rejected": self.rejected, "timed_out": self.timed_out,
"cancelled": self.cancelled}
+ def _observe(self, name, seconds):
+ # Called with condition held. Cumulative buckets need no request history.
+ timing = self.timings[name]
+ timing["sum"] += seconds
+ for i, bound in enumerate(self._buckets):
+ if seconds <= bound:
+ timing["buckets"][i] += 1
+
+ def observe_timing(self, name, seconds):
+ with self.condition:
+ self._observe(name, seconds)
+
+ def prometheus(self):
+ """One consistent, bounded snapshot; no prompt or request-ID labels."""
+ gauges = {"active": "Currently admitted requests.",
+ "queued": "Requests waiting for a KV slot.",
+ "capacity": "Concurrent KV slots configured.",
+ "max_queue": "Maximum waiting requests configured."}
+ counters = {"admitted": "Requests admitted to a KV slot.",
+ "completed": "Admitted requests that returned normally.",
+ "failed": "Admitted requests that raised an error.",
+ "rejected": "Requests rejected because the queue was full.",
+ "timed_out": "Requests that timed out waiting for a slot.",
+ "cancelled": "Requests cancelled while queued or admitted."}
+ lines = []
+ with self.condition:
+ for kind, fields in (("gauge", gauges), ("counter", counters)):
+ for field, help_text in fields.items():
+ name = "colibri_scheduler_" + field + ("_total" if kind == "counter" else "")
+ value = len(self.queue) if field == "queued" else getattr(self, field)
+ lines.extend((f"# HELP {name} {help_text}", f"# TYPE {name} {kind}",
+ f"{name} {value}"))
+ for field, help_text in (
+ ("queue_wait_seconds", "Queue wait of admitted requests only."),
+ ("slot_duration_seconds", "Slot occupancy of finished admitted requests, including errors and cancellation."),
+ ("first_output_seconds", "Engine-call start to first nonempty text or tool callback, excluding queue wait."),
+ ("engine_call_seconds", "Duration of finished engine generation calls, including errors and cancellation.")):
+ name = "colibri_scheduler_" + field
+ timing = self.timings[field]
+ lines.extend((f"# HELP {name} {help_text}", f"# TYPE {name} histogram"))
+ for bound, count in zip(self._buckets, timing["buckets"]):
+ label = "+Inf" if math.isinf(bound) else str(bound)
+ lines.append(f'{name}_bucket{{le="{label}"}} {count}')
+ lines.extend((f'{name}_sum {timing["sum"]}',
+ f'{name}_count {timing["buckets"][-1]}'))
+ return "\n".join(lines) + "\n"
+
def close(self):
with self.condition:
self.closed = True
@@ -251,6 +302,63 @@ def content_text(content, param):
_ARG_RE = re.compile(r"([^<]*)(.*?)", re.DOTALL)
_NAME_RE = re.compile(r"\s*([A-Za-z0-9_.\-]+)")
_TAG_RE = re.compile(r"?arg_key>|?arg_value>")
+
+
+def _fallback_tool_preamble(tools):
+ """Tool declaration for a family with no native tool tokens.
+
+ Mirrors the GLM-5.2 block because ``parse_tool_calls`` -- the parser these
+ families fall back to in ``parse_arch_tool_calls`` -- reads exactly that
+ wire format. Asking for a format the parser does not accept would produce
+ tool calls nobody can read back.
+ """
+ out = ["You have access to the following functions. Call one only when it "
+ "is needed to answer the user.\n\n\n"]
+ for tool in tools:
+ fn = tool.get("function", tool) if isinstance(tool, dict) else {}
+ out.append(json.dumps(fn, ensure_ascii=False) + "\n")
+ out.append("\n\nTo call a function, reply with the call and nothing "
+ "else, in this exact format:\n" + BOX_START + "{function-name}"
+ "{arg-key}{arg-value}"
+ + BOX_END)
+ return "".join(out)
+
+
+def _fallback_tool_calls(tool_calls, index):
+ """Render assistant tool_calls in the format parse_tool_calls() reads."""
+ out = []
+ for position, call in enumerate(tool_calls or []):
+ if not isinstance(call, dict):
+ raise APIError(400, "Each tool call must be an object.",
+ f"messages.{index}.tool_calls.{position}")
+ fn = call.get("function", call)
+ if not isinstance(fn, dict):
+ raise APIError(400, "`function` must be an object.",
+ f"messages.{index}.tool_calls.{position}.function")
+ name = fn.get("name")
+ if not isinstance(name, str) or not name:
+ raise APIError(400, "`function.name` must be a non-empty string.",
+ f"messages.{index}.tool_calls.{position}.function.name")
+ args = fn.get("arguments", "{}")
+ if isinstance(args, str):
+ try:
+ args = json.loads(args) if args else {}
+ except (json.JSONDecodeError, TypeError, ValueError):
+ raise APIError(400, "`function.arguments` must be a JSON object.",
+ f"messages.{index}.tool_calls.{position}.function.arguments")
+ out.append(BOX_START + name)
+ for key, value in (args or {}).items():
+ rendered = value if isinstance(value, str) else json.dumps(
+ value, ensure_ascii=False)
+ out.append(f"{key}{rendered}")
+ out.append(BOX_END)
+ return "".join(out)
+
+
+def _fallback_tool_result(message, index):
+ """Render a role:"tool" message as prose these templates can carry."""
+ body = content_text(message.get("content"), f"messages.{index}.content")
+ return TR_OPEN + body + TR_CLOSE
# A closing tag the model started but never finished ("K structure. Default OFF (never rewrites well-formed output).
_SALVAGE = os.environ.get("COLI_TOOL_SALVAGE", "0") == "1"
+# Families whose chat template has no tool syntax at all (OLMoE, Qwen3.6) refuse
+# tools[] and role:"tool" rather than invent a format. COLI_TOOL_FALLBACK=1 opts
+# into a prompt-injected translation for them: the declaration block, the prior
+# assistant calls and the tool results are written as ordinary turns, in the
+# same wire format parse_tool_calls() already reads back (#1378). Default OFF --
+# these models were never trained on tool syntax, so this trades a clean 400 for
+# output the parser may or may not recognise.
+_TOOL_FALLBACK = os.environ.get("COLI_TOOL_FALLBACK", "0") == "1"
+
def _tool_choice_name(tool_choice):
"""The tool name a dict `tool_choice` forces, or None.
@@ -275,6 +392,21 @@ def _tool_choice_name(tool_choice):
or tool_choice.get("name"))
+def _tool_function(tool):
+ """The function object on a tools[] entry, or {} if it is missing or not an object.
+
+ OpenAI dual spelling: {"function": {"name": ...}} or a bare function object.
+ .items() is taken only from a dict. Writing the name where the object goes
+ ({"type": "function", "function": "search"}) raised AttributeError in the GLM
+ and DeepSeek declaration blocks, and do_POST answered HTTP 500 "The colibri
+ engine failed to process the request." for a payload generation_options()
+ already has a 400 for. Same shape as the tool_choice fix (#1598): read the
+ member, then check it.
+ """
+ fn = tool.get("function", tool) if isinstance(tool, dict) else {}
+ return fn if isinstance(fn, dict) else {}
+
+
def _tool_param_order(tools):
"""name -> ordered param names (required first) from the request schema, for de-mangling."""
out = {}
@@ -447,7 +579,7 @@ def _dsv4_tools_block(tools):
"""V4 tool-declaration block, rendered by the vendored reference template."""
schemas = []
for tool in (tools or []):
- fn = tool.get("function", tool) if isinstance(tool, dict) else {}
+ fn = _tool_function(tool)
# Gateway-side scrub: OpenAI clients attach routing hints the model
# schema must not carry.
schemas.append({k: v for k, v in fn.items() if k not in ("defer_loading", "strict")})
@@ -455,7 +587,7 @@ def _dsv4_tools_block(tools):
def _dsv4_tool_calls(tool_calls):
- """Render OpenAI-format tool_calls into a V4 DSML block (incl. the leading
+ """Render OpenAI-format tool_calls into a V4 DSML block (incl. the leading
)."""
return v4_dsml.render_tool_calls(tool_calls)
@@ -997,12 +1129,18 @@ def _k3_order_tool_results(messages):
def render_chat_kimi(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
+ tool_choice=None, add_generation_prompt=True):
"""Validated multi-turn K3 payload for the C engine.
K3's rank-BPE makes ordinary-text segment boundaries part of the tokenizer
contract. This private length-framed payload preserves roles, UTF-8 bytes,
and message boundaries; kimi_k3.c constructs the native XTML tokens.
+
+ add_generation_prompt=False continues a trailing assistant turn. Kimi frames turns
+ engine-side, so unlike the string renderers there is no terminator to drop here: the final
+ assistant turn is emitted as a `C` record (reasoning + text), which kimi_k3.c renders as
+ the open turn -- no <|close|>/<|end_of_msg|>, and no fresh generation cue. An engine that
+ predates the record rejects the payload rather than miswiring it.
"""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
@@ -1055,6 +1193,14 @@ def render_chat_kimi(messages, enable_thinking=False, reasoning_effort=None, too
if role == "assistant":
last_calls = calls or []
tool_index = 0
+ if not add_generation_prompt and index == len(messages) - 1:
+ # Continuation: the trailing assistant turn is left OPEN. resolve_generation_prompt
+ # has already refused tools/tool_calls and a non-assistant trailing turn, so this is
+ # a plain assistant turn; the C record carries its reasoning (if any) and text, and
+ # kimi_k3.c renders it as the open turn with no cue.
+ r = reasoning or ""
+ parts.append(f"C {len(r.encode('utf-8'))} {len(text.encode('utf-8'))}\n{r}{text}")
+ continue
if calls:
if len(calls) > 64:
raise APIError(400, "Too many tool calls in one message (max 64).",
@@ -1083,13 +1229,17 @@ def render_chat_kimi(messages, enable_thinking=False, reasoning_effort=None, too
def render_chat_v4(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
+ tool_choice=None, add_generation_prompt=True):
"""DeepSeek V4's native multi-turn chat template.
The target engine receives this as a raw prompt. Prior assistant turns end
with the checkpoint's EOS marker; the final assistant marker selects the
thinking or direct-answer prefix for the new turn.
+ add_generation_prompt=False continues a trailing assistant turn: the last assistant turn
+ is rendered open, i.e. without its closing EOS and with no cue, the position the model
+ occupies mid-turn. EOS is the terminator to drop here, as <|im_end|> is for ChatML.
+
Tool use follows the official DSML format (encoding/encoding_dsv4.py): tool
schemas are declared on the first system/developer message, assistant tool
calls are DSML blocks, and tool results are blocks merged into
@@ -1163,7 +1313,7 @@ def render_chat_v4(messages, enable_thinking=False, reasoning_effort=None, tools
effort = DSV4_REASONING_EFFORT.get(reasoning_effort, "low")
if effort != "low":
parts.append(DSV4_REASONING_EFFORT_PROMPTS[effort])
- for message in merged:
+ for m_index, message in enumerate(merged):
role = message["role"]
if role in ("system", "developer"):
if role == "developer":
@@ -1185,13 +1335,16 @@ def render_chat_v4(messages, enable_thinking=False, reasoning_effort=None, tools
parts.append(message["content"])
if message.get("tool_calls"):
parts.append(_dsv4_tool_calls(message["tool_calls"]))
- parts.append(eos)
- parts.extend((assistant, "" if enable_thinking else ""))
+ # A continued turn is the last message rendered open: no EOS, no cue below.
+ if add_generation_prompt or m_index != len(merged) - 1:
+ parts.append(eos)
+ if add_generation_prompt:
+ parts.extend((assistant, "" if enable_thinking else ""))
return "".join(parts)
def render_chat_olmoe(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
+ tool_choice=None, add_generation_prompt=True):
"""OLMoE-Instruct's native chat_template (tokenizer_config.json): one
bos_token, then per-message <|system|>/<|user|>/<|assistant|> turns each
closed by a newline, prior assistant turns also closed by eos_token
@@ -1199,20 +1352,32 @@ def render_chat_olmoe(messages, enable_thinking=False, reasoning_effort=None, to
repurposed as this tokenizer's BOS/EOS marker), and a trailing
"<|assistant|>\\n" generation prompt. No tool-call syntax and no thinking
mode exist in this template, so both parameters are accepted but unused.
- """
+
+ add_generation_prompt=False continues a trailing assistant turn. The template closes
+ even the last assistant turn with eos_token, so the open-turn shape is that turn without
+ the eos and with no cue -- the same drop-the-terminator move as the ChatML families, with
+ eos_token as the terminator here."""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
- if tools or tool_choice not in (None, "none"):
- raise APIError(400, "Tool use is not wired up for the OLMoE engine yet.",
- "tools", "unsupported_parameter")
+ if tool_choice == "none":
+ tools = None
+ if (tools or tool_choice not in (None, "none")) and not _TOOL_FALLBACK:
+ raise APIError(400, "Tool use is not wired up for the OLMoE engine yet. "
+ "Set COLI_TOOL_FALLBACK=1 to opt into prompt-injected "
+ "tool translation.", "tools", "unsupported_parameter")
boundary = "|||IP_ADDRESS|||" # bos_token == eos_token in this tokenizer
parts = [boundary]
+ if tools and _TOOL_FALLBACK:
+ parts.append(f"<|system|>\n{_fallback_tool_preamble(tools)}\n")
last = len(messages) - 1
for index, message in enumerate(messages):
if not isinstance(message, dict):
raise APIError(400, "Each message must be an object.", f"messages.{index}")
role = message.get("role")
- if role not in ("system", "developer", "user", "assistant"):
+ allowed = ("system", "developer", "user", "assistant")
+ if _TOOL_FALLBACK:
+ allowed += ("tool",)
+ if role not in allowed:
raise APIError(400, f"Unsupported role {role!r}.", f"messages.{index}.role")
raw = message.get("content")
text = content_text(raw, f"messages.{index}.content") if raw is not None else ""
@@ -1220,42 +1385,83 @@ def render_chat_olmoe(messages, enable_thinking=False, reasoning_effort=None, to
parts.append(f"<|system|>\n{text}\n")
elif role == "user":
parts.append(f"<|user|>\n{text}\n")
+ elif role == "tool":
+ # No tool role in this template: the result rides in as a user turn.
+ parts.append(f"<|user|>\n{_fallback_tool_result(message, index)}\n")
else:
- parts.append(f"<|assistant|>\n{text}{boundary}")
+ calls = (_fallback_tool_calls(message.get("tool_calls"), index)
+ if _TOOL_FALLBACK else "")
+ # A continued turn is the last message rendered open: no eos, no cue.
+ terminator = "" if (not add_generation_prompt and index == last) else boundary
+ parts.append(f"<|assistant|>\n{text}{calls}{terminator}")
if index != last:
parts.append("\n")
- parts.append("<|assistant|>\n")
+ if add_generation_prompt:
+ parts.append("<|assistant|>\n")
return "".join(parts)
def render_chat_qwen(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
+ tool_choice=None, add_generation_prompt=True):
"""Text-only subset of Qwen3.6's chat_template: <|im_start|>role\\n ...
<|im_end|>\\n frames, then the generation prompt. The official template
opens a mandatory block after `<|im_start|>assistant\\n` — the
model was never trained on the bare `assistant\\n` state, and greedy
argmax there lands on an EOS special (measured: gen=0). With thinking
disabled the template pre-closes the block instead; both branches are
- mirrored here byte for byte."""
+ mirrored here byte for byte.
+
+ add_generation_prompt=False continues a trailing assistant turn. The template renders an
+ assistant turn AFTER the last user query with its block (an earlier one,
+ from history, has it stripped) -- so the open-turn shape is that think-form minus the
+ <|im_end|> terminator and with no cue, not the bare history form the loop emits otherwise.
+ ChatML's per-turn terminator is why the marker has to be dropped explicitly, as on qwen38."""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
- if tools or tool_choice not in (None, "none"):
- raise APIError(400, "Tool use is not wired up for the qwen36 engine yet.",
- "tools", "unsupported_parameter")
+ if tool_choice == "none":
+ tools = None
+ if (tools or tool_choice not in (None, "none")) and not _TOOL_FALLBACK:
+ raise APIError(400, "Tool use is not wired up for the qwen36 engine yet. "
+ "Set COLI_TOOL_FALLBACK=1 to opt into prompt-injected "
+ "tool translation.", "tools", "unsupported_parameter")
parts = []
+ if tools and _TOOL_FALLBACK:
+ parts.append("<|im_start|>system\n"
+ + _fallback_tool_preamble(tools) + "<|im_end|>\n")
for index, message in enumerate(messages):
if not isinstance(message, dict):
raise APIError(400, "Each message must be an object.", f"messages.{index}")
role = message.get("role")
if role == "developer":
role = "system"
- if role not in ("system", "user", "assistant"):
+ allowed = ("system", "user", "assistant")
+ if _TOOL_FALLBACK:
+ allowed += ("tool",)
+ if role not in allowed:
raise APIError(400, f"Unsupported role {role!r}.", f"messages.{index}.role")
raw = message.get("content")
text = content_text(raw, f"messages.{index}.content") if raw is not None else ""
+ if not add_generation_prompt and role == "assistant" and index == len(messages) - 1:
+ # Continued turn: the template gives a post-query assistant turn a
+ # block, then the model resumes the content. Match it, minus the terminator/cue.
+ reasoning = message.get("reasoning_content", "")
+ if not isinstance(reasoning, str):
+ raise APIError(400, "`reasoning_content` must be a string.",
+ f"messages.{index}.reasoning_content")
+ parts.append(f"<|im_start|>assistant\n\n{reasoning.strip()}\n\n\n"
+ f"{text.strip()}")
+ continue
+ if role == "tool":
+ # No tool role in this template: the result rides in as a user turn.
+ parts.append("<|im_start|>user\n"
+ + _fallback_tool_result(message, index) + "<|im_end|>\n")
+ continue
+ if role == "assistant" and _TOOL_FALLBACK:
+ text += _fallback_tool_calls(message.get("tool_calls"), index)
parts.append(f"<|im_start|>{role}\n{text}<|im_end|>\n")
- parts.append("<|im_start|>assistant\n")
- parts.append("\n" if enable_thinking else "\n\n\n\n")
+ if add_generation_prompt:
+ parts.append("<|im_start|>assistant\n")
+ parts.append("\n" if enable_thinking else "\n\n\n\n")
return "".join(parts)
@@ -1372,8 +1578,14 @@ def parse_qwen38_tool_calls(reply, tools=None):
def render_chat_qwen38(messages, enable_thinking=True, reasoning_effort=None, tools=None,
- tool_choice=None):
- """Text-only Qwen3.8 chat-template subset with native reasoning hints."""
+ tool_choice=None, add_generation_prompt=True):
+ """Text-only Qwen3.8 chat-template subset with native reasoning hints.
+
+ add_generation_prompt=False continues a trailing assistant turn. ChatML closes every
+ turn with <|im_end|>, so the open-turn shape is the past-turn render of that last message
+ MINUS its terminator, and no generation cue after it -- the position the model occupies
+ while writing an assistant turn. (GLM has no per-turn terminator, so there suppressing the
+ cue is enough; here the terminator has to be dropped too.)"""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
if tool_choice in ("none",):
@@ -1465,22 +1677,30 @@ def render_chat_qwen38(messages, enable_thinking=True, reasoning_effort=None, to
rendered = f"\n{reasoning.strip()}\n\n\n{text}"
if calls:
rendered += _qwen38_tool_calls(calls, bool(text.strip()), index)
- parts.append(f"<|im_start|>assistant\n{rendered}<|im_end|>\n")
+ # A continued turn is the last message rendered open: no <|im_end|>, no cue.
+ terminator = "" if (not add_generation_prompt and index == len(messages) - 1) \
+ else "<|im_end|>\n"
+ parts.append(f"<|im_start|>assistant\n{rendered}{terminator}")
continue
parts.append(f"<|im_start|>{role}\n{text}<|im_end|>\n")
- parts.append("<|im_start|>assistant\n")
- parts.append("\n" if enable_thinking else "\n\n\n\n")
+ if add_generation_prompt:
+ parts.append("<|im_start|>assistant\n")
+ parts.append("\n" if enable_thinking else "\n\n\n\n")
return "".join(parts)
def render_chat_inkling(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None, audio_out=None):
+ tool_choice=None, audio_out=None, add_generation_prompt=True):
"""Text-only subset of Inkling's chat_template.jinja: role tokens with
<|content_text|> parts and <|end_message|> terminators, an assistant
<|content_model_end_sampling|> after each prior model turn, the
thinking-effort hint appended after the messages (the template's fallback
- branch), then <|message_model|> as the generation prompt."""
+ branch), then <|message_model|> as the generation prompt.
+
+ add_generation_prompt=False continues a trailing assistant turn: the last model turn is
+ rendered open -- without its <|end_message|> and the <|content_model_end_sampling|> that
+ close it, and with no cue. Those two markers are the terminator to drop here."""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
if tools or (tool_choice not in (None, "none")):
@@ -1516,6 +1736,8 @@ def render_chat_inkling(messages, enable_thinking=False, reasoning_effort=None,
if not effort_emitted and role not in ("system", "developer"):
prompt.append(effort_str)
effort_emitted = True
+ open_turn = (not add_generation_prompt and role == "assistant"
+ and index == len(messages) - 1)
raw = message.get("content")
if audio_out is not None and role == "user" and isinstance(raw, list):
# multipart user content: text runs and audio clips become separate
@@ -1530,26 +1752,34 @@ def render_chat_inkling(messages, enable_thinking=False, reasoning_effort=None,
+ "<|audio|>" * val + "<|audio_end|><|end_message|>")
else:
text = content_text(raw, f"messages.{index}.content") if raw is not None else ""
- prompt.append(f"{rtok}<|content_text|>{text}<|end_message|>")
- if role == "assistant":
+ # A continued turn is the last model message rendered open: no <|end_message|>.
+ terminator = "" if open_turn else "<|end_message|>"
+ prompt.append(f"{rtok}<|content_text|>{text}{terminator}")
+ if role == "assistant" and not open_turn:
prompt.append("<|content_model_end_sampling|>")
if not effort_emitted: # all-system edge case: fallback
prompt.append(effort_str)
- prompt.append("<|message_model|>") # add_generation_prompt
- # Thinking off: prefill the content channel. Without this the model can still
- # sample <|content_thinking|> as its first token (the effort hint is only a
- # soft signal), open a reasoning block, and burn the whole token budget before
- # reaching <|content_text|> — which the splitter then strips to an empty
- # answer. Ending the prompt at <|message_model|><|content_text|> forces content
- # mode; it is exactly the sequence every non-thinking turn is trained on.
- if eff == 0.0:
- prompt.append("<|content_text|>")
+ if add_generation_prompt:
+ prompt.append("<|message_model|>") # generation cue
+ # Thinking off: prefill the content channel. Without this the model can still
+ # sample <|content_thinking|> as its first token (the effort hint is only a
+ # soft signal), open a reasoning block, and burn the whole token budget before
+ # reaching <|content_text|> — which the splitter then strips to an empty
+ # answer. Ending the prompt at <|message_model|><|content_text|> forces content
+ # mode; it is exactly the sequence every non-thinking turn is trained on.
+ if eff == 0.0:
+ prompt.append("<|content_text|>")
return "".join(prompt)
def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
- """Render the text-only subset of the official GLM-5.2 chat template."""
+ tool_choice=None, add_generation_prompt=True):
+ """Render the text-only subset of the official GLM-5.2 chat template.
+
+ add_generation_prompt=False continues a trailing assistant turn. GLM has no per-turn
+ terminator (the next role token ends a turn), so the loop already renders that last message
+ as a past turn -- <|assistant|>{content} -- and suppressing the cue leaves
+ the prompt open on it, exactly as on glm53. Nothing to strip, unlike the ChatML families."""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
prompt = ["[gMASK]"]
@@ -1583,7 +1813,7 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No
"user query.\n\nYou are provided with function signatures within "
"XML tags:\n\n")
for tool in tools:
- fn = tool.get("function", tool) if isinstance(tool, dict) else {}
+ fn = _tool_function(tool)
clean = {k: v for k, v in fn.items() if k not in ("defer_loading", "strict")}
prompt.append(json.dumps(clean, ensure_ascii=False) + "\n")
prompt.append("\n\nFor each function call, output the function name and arguments "
@@ -1622,8 +1852,17 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No
args = json.loads(args)
except (json.JSONDecodeError, TypeError):
args = {}
+ if not isinstance(args, dict):
+ # `arguments` that is valid JSON but not an object ("[1,2]",
+ # "5", a bare list) reached .items() and raised
+ # AttributeError, which do_POST answers with HTTP 500. The
+ # same field is already tolerated when it does not parse at
+ # all, and every sibling renderer renders the call without
+ # arguments instead of failing; this is the one branch that
+ # was never completed.
+ args = {}
prompt.append(BOX_START + (fn.get("name") or ""))
- for key, value in (args or {}).items():
+ for key, value in args.items():
prompt.append(f"{key}"
+ (value if isinstance(value, str)
else json.dumps(value, ensure_ascii=False)) + "")
@@ -1636,8 +1875,9 @@ def render_chat(messages, enable_thinking=False, reasoning_effort=None, tools=No
raise APIError(400, f"Unsupported message role: {role!r}.",
f"messages.{index}.role", "unsupported_role")
prev_tool = (role == "tool")
- prompt.append("<|assistant|>" if enable_thinking else
- "<|assistant|>")
+ if add_generation_prompt:
+ prompt.append("<|assistant|>" if enable_thinking else
+ "<|assistant|>")
return "".join(prompt)
@@ -1899,8 +2139,7 @@ def _glm53_tool_block(tools):
somiglia a quello dell'addestramento non e' quello dell'addestramento."""
body = "".join(f"\n{_glm53_tool_json(tool)}\n\n"
for tool in tools
- if not (isinstance(tool, dict)
- and (tool.get("function", tool) or {}).get("defer_loading")))
+ if not _tool_function(tool).get("defer_loading"))
return GLM53_TOOL_PREAMBLE + body + GLM53_TOOL_EPILOGUE
@@ -1919,8 +2158,10 @@ def _glm53_tool_calls(calls):
arguments = json.loads(arguments)
except ValueError:
arguments = {}
+ if not isinstance(arguments, dict):
+ arguments = {} # same gap as render_chat above
pieces = [f"{name}"]
- for key, value in (arguments or {}).items():
+ for key, value in arguments.items():
rendered = value if isinstance(value, str) else json.dumps(value, ensure_ascii=False)
pieces.append(f"{key}{rendered}")
pieces.append("")
@@ -1929,7 +2170,7 @@ def _glm53_tool_calls(calls):
def render_chat_glm53(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
+ tool_choice=None, add_generation_prompt=True):
"""Render the text-only subset of the official GLM-5.3-Flash chat template.
Not a variant of the GLM-5.2 renderer above, and the differences are not
@@ -1944,7 +2185,7 @@ def render_chat_glm53(messages, enable_thinking=False, reasoning_effort=None, to
so the existing parser needs nothing added for this family.
The whole thing is pinned byte for byte against chat_template.jinja rendered
- with jinja2 (tests/test_glm53_chat_template.py). Getting the prompt nearly
+ with jinja2 (tests/glm53_chat_template_harness.py). Getting the prompt nearly
right is the failure mode worth guarding: the model answers either way.
"""
if not isinstance(messages, list) or not messages:
@@ -2033,7 +2274,14 @@ def render_chat_glm53(messages, enable_thinking=False, reasoning_effort=None, to
# su cui il modello e' addestrato" perche' il template lo scrive davanti a un
# TURNO PASSATO senza ragionamento. E' vero per un turno passato e falso per il
# prompt di generazione: la posizione da cui il modello scrive non e' mai quella.
- prompt.append("<|assistant|>")
+ #
+ # add_generation_prompt=False non e' una forma nostra: e' l'altro ramo di questo stesso
+ # `if` nel template. Il prompt finisce allora sull'ultimo turno assistant reso come
+ # turno PASSATO -- seguito dal contenuto -- e il modello lo prosegue
+ # invece di aprirne uno nuovo. Chi non chiede la prosecuzione non vede differenza:
+ # il ramo True e' invariato, byte per byte, ed e' quello che il test confronta.
+ if add_generation_prompt:
+ prompt.append("<|assistant|>")
return "".join(prompt)
@@ -2071,7 +2319,7 @@ def _dsv41_tools_block(tools):
"""V4.1 tool-declaration block, rendered by the vendored reference template."""
schemas = []
for tool in (tools or []):
- fn = tool.get("function", tool) if isinstance(tool, dict) else {}
+ fn = _tool_function(tool)
# Gateway-side scrub: OpenAI clients attach routing hints the model
# schema must not carry.
clean = {k: v for k, v in fn.items() if k not in ("defer_loading", "strict")}
@@ -2139,13 +2387,18 @@ def _dsv41_merge_turns(messages):
def render_chat_dsv41(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None):
+ tool_choice=None, add_generation_prompt=True):
"""encoding.py _encode_messages_text for one turn.
Tool use follows the checkpoint's own DSML format (encoding/encoding.py, vendored in
v41_dsml.py): schemas are declared at the end of the system message, assistant tool
calls are <|DSML| calls> blocks, and tool results are blocks merged
into the following user turn.
+
+ add_generation_prompt=False continues a trailing assistant turn: the last turn is rendered
+ open -- its block and content as a PAST turn, but without the closing
+ <|end▁of▁sentence|> and with no cue appended. That is the same drop-the-terminator move as
+ deepseek_v4, whose EOS this shares; the model resumes from the content it was handed.
"""
if not isinstance(messages, list) or not messages:
raise APIError(400, "`messages` must be a non-empty array.", "messages")
@@ -2205,27 +2458,157 @@ def render_chat_dsv41(messages, enable_thinking=False, reasoning_effort=None, to
prompt.append(turn["content"])
if turn.get("tool_calls"):
prompt.append(v41_dsml.render_tool_calls(turn["tool_calls"]))
- prompt.append(DSV41_EOS)
+ # A continued turn is the last message rendered open: no EOS, no cue below.
+ if add_generation_prompt or index != len(turns) - 1:
+ prompt.append(DSV41_EOS)
# the generation cue, exactly as render_message appends it after a user turn
- prompt.append(DSV41_ASSISTANT)
- prompt.append("" if enable_thinking and len(turns) - 1 >= last_user else "")
+ if add_generation_prompt:
+ prompt.append(DSV41_ASSISTANT)
+ prompt.append("" if enable_thinking and len(turns) - 1 >= last_user else "")
return "".join(prompt)
+# ---- continuing an unfinished assistant turn (COLI_CONTINUE_ASSISTANT) ----------------
+# A trailing `assistant` message means "continue writing this turn", not "here is a turn I
+# already finished". The official template says exactly that, and says it in one place --
+# {%- if add_generation_prompt -%}<|assistant|>{{- '' -}}{%- endif -%}
+# -- whose False branch every renderer in this file hard-codes to True. With the cue
+# suppressed the prompt ends mid-turn, on the shape the template writes in front of a PAST
+# assistant turn, which is a position the model saw all through training.
+#
+# That distinction is what makes this safe on GLM-5.3 specifically. #1327 measured that a
+# CLOSED, EMPTY at the end of a prompt is out of distribution and the model
+# keeps reasoning through it. The position here is a different one: followed
+# by real content, i.e. the past-turn shape, which is why a continuation must carry text.
+#
+# llama.cpp needs no switch for this because it runs the checkpoint's jinja at request time,
+# so `add_generation_prompt=False` costs it nothing. This gateway renders by hand, on purpose
+# and for speed (tests/glm53_chat_template_harness.py says why), and the bill for that choice is
+# exactly here: one template flag, one open-turn shape to derive per renderer. Each string
+# renderer derives its own, pinned byte-for-byte against the checkpoint's template;
+# CONTINUATION_FAMILIES is the set that has done so. Kimi K3 differs in WHERE its shape lives:
+# its prompt is framed engine-side (render_chat_kimi hands a K3CHAT1 record to kimi_k3.c, which
+# assembles the XTML tokens), so its open turn is a `C` record here plus a branch in that C path,
+# pinned by tests/test_k3_chat_tools.c against the tiny tokenizer rather than by a template diff.
+
+# Families whose renderer implements the add_generation_prompt=False (open-turn) branch. A
+# trailing assistant turn on a family NOT in this set falls through to the ordinary render
+# (the cue is appended, exactly as before this existed) rather than erroring -- continuation is
+# on by default, and a family without its open-turn shape yet must not start rejecting requests
+# nobody opted into. Each renderer adds itself here in the same commit that derives its shape.
+CONTINUATION_FAMILIES = {"glm53", "qwen38", "qwen36", "glm", "olmoe", "deepseek_v4", "inkling",
+ "kimi", "deepseek_v41"}
+
+
+def resolve_generation_prompt(messages, body):
+ """Does this prompt end on a generation cue, or on an assistant turn to continue?
+
+ Returns True for the ordinary case (append the cue) and False for a continuation, which is
+ the template's `add_generation_prompt=False`.
+
+ Continuation is ON by default. A message list ending in a non-empty assistant turn already
+ says "continue me" -- the same contract as Anthropic's API -- and no OpenAI-compatible
+ client sends a trailing assistant turn by accident. It is deliberately NOT a request field:
+ a client would have to know colibri specifically to send one, and the clients that most
+ want this -- anything pointed at an OpenAI- or Anthropic-compatible URL -- send a message
+ list and nothing else.
+
+ COLI_CONTINUE_ASSISTANT=0 is the off-switch, for a deployment that wants the old behaviour
+ (fold the trailing turn into a completed one and append a fresh cue). It is the only value
+ that turns this off; anything else, including unset, leaves it on.
+
+ A family whose renderer has no open-turn shape yet (ARCH not in CONTINUATION_FAMILIES)
+ falls through to the ordinary render rather than erroring: continuation defaults on, so a
+ family added before its open-turn shape must not start rejecting trailing-assistant
+ requests that worked before. Every shipped family is in the set today, Kimi K3 included --
+ its open turn is framed in kimi_k3.c (a `C` record), not derived in the renderer here.
+ """
+ continuing = os.environ.get("COLI_CONTINUE_ASSISTANT", "1") != "0"
+ last = messages[-1] if isinstance(messages, list) and messages else None
+ if not (isinstance(last, dict) and last.get("role") == "assistant"):
+ return True
+ where = f"messages.{len(messages) - 1}"
+ if not continuing:
+ return True
+ if ARCH not in CONTINUATION_FAMILIES:
+ return True # open-turn shape not derived for this family yet -- render as before
+ if body.get("tools") or body.get("functions"):
+ raise APIError(400, "A continued assistant turn cannot be combined with `tools`: "
+ "the tool-call parsers read an assistant turn from its start, and a "
+ "continuation can end anywhere -- including inside a "
+ "block.", "tools", "unsupported_parameter")
+ if last.get("tool_calls"):
+ raise APIError(400, "A continued `assistant` message cannot carry `tool_calls`.",
+ f"{where}.tool_calls", "unsupported_value")
+ if len(messages) < 2:
+ raise APIError(400, "A continued `assistant` turn needs a preceding turn to "
+ "continue from.", "messages")
+ raw = last.get("content")
+ if isinstance(raw, list): # multimodal parts: only the text counts
+ text = "".join(part.get("text", "") for part in raw
+ if isinstance(part, dict) and part.get("type") == "text")
+ elif raw is None:
+ text = ""
+ elif isinstance(raw, str):
+ text = raw
+ else:
+ raise APIError(400, "Message content must be a string or an array of blocks.",
+ f"{where}.content")
+ if not text.strip():
+ raise APIError(400, "A continued `assistant` turn needs text to continue. An empty "
+ "one ends the prompt on a closed, empty block, which "
+ "is the out-of-distribution position #1327 removed -- the model "
+ "reasons straight through it instead of answering.",
+ f"{where}.content", "invalid_value")
+ if text != text.rstrip():
+ raise APIError(400, "A continued `assistant` turn cannot end with whitespace: the "
+ "template strips it, so the model would resume from different bytes "
+ "than the ones sent. Put the space at the start of what you expect "
+ "back instead.", f"{where}.content", "invalid_value")
+ return False
+
+
def render_chat_for_arch(messages, enable_thinking=False, reasoning_effort=None, tools=None,
- tool_choice=None, audio_out=None):
- """Render a chat request with the active engine's native prompt contract."""
+ tool_choice=None, audio_out=None, add_generation_prompt=True):
+ """Render a chat request with the active engine's native prompt contract.
+
+ `add_generation_prompt=False` (a continued assistant turn) is implemented for the families
+ in CONTINUATION_FAMILIES. resolve_generation_prompt() passes any other family through with
+ the cue appended, so it never reaches here with the flag False; this stays as the backstop,
+ because silently appending a cue to a continuation is the exact failure this exists to remove.
+ """
+ if not add_generation_prompt and ARCH not in CONTINUATION_FAMILIES:
+ raise APIError(400, f"Continuing an assistant turn is not implemented for {ARCH!r}.",
+ "messages", "unsupported_parameter")
if ARCH == "inkling":
return render_chat_inkling(messages, enable_thinking, reasoning_effort, tools,
- tool_choice, audio_out=audio_out)
- renderer = (render_chat_glm53 if ARCH == "glm53" else
- render_chat_kimi if ARCH == "kimi" else
- render_chat_qwen if ARCH == "qwen36" else
- render_chat_qwen38 if ARCH == "qwen38" else
- render_chat_v4 if ARCH == "deepseek_v4" else
- render_chat_dsv41 if ARCH == "deepseek_v41" else
- render_chat_olmoe if ARCH == "olmoe" else render_chat)
- return renderer(messages, enable_thinking, reasoning_effort, tools, tool_choice)
+ tool_choice, audio_out=audio_out,
+ add_generation_prompt=add_generation_prompt)
+ if ARCH == "glm53":
+ return render_chat_glm53(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ if ARCH == "qwen38":
+ return render_chat_qwen38(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ if ARCH == "qwen36":
+ return render_chat_qwen(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ if ARCH == "glm":
+ return render_chat(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ if ARCH == "olmoe":
+ return render_chat_olmoe(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ if ARCH == "deepseek_v4":
+ return render_chat_v4(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ if ARCH == "kimi":
+ return render_chat_kimi(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt=add_generation_prompt)
+ if ARCH == "deepseek_v41":
+ return render_chat_dsv41(messages, enable_thinking, reasoning_effort, tools,
+ tool_choice, add_generation_prompt)
+ return render_chat(messages, enable_thinking, reasoning_effort, tools, tool_choice)
# ---- Anthropic Messages API (#343) --------------------------------------------------------
@@ -2237,7 +2620,7 @@ def render_chat_for_arch(messages, enable_thinking=False, reasoning_effort=None,
ANTHROPIC_LOCAL_SIGNATURE = "colibri-local" # opaque compatibility metadata, not a crypto proof
-def starts_in_reasoning(enable_thinking):
+def starts_in_reasoning(enable_thinking, add_generation_prompt=True):
"""Se l'uscita del modello comincia DENTRO al blocco di ragionamento.
Dipende da come il prompt lo ha lasciato, e ogni famiglia lo lascia come
@@ -2250,8 +2633,20 @@ def starts_in_reasoning(enable_thinking):
ha un interruttore, render_chat_glm53 apre SEMPRE, e "thinking
spento" vuol dire solo effort Low. L'uscita comincia dentro al blocco in
ogni caso; partire in modalita' testo perche' il client ha detto False e'
- esattamente il ragionamento incollato davanti alla risposta di #1278."""
- return enable_thinking or ARCH == "glm53"
+ esattamente il ragionamento incollato davanti alla risposta di #1278.
+
+ Il turno proseguito (add_generation_prompt=False) e' il terzo stato, e non
+ lo dice l'interruttore: il prompt finisce sull'ultimo turno assistant reso
+ come turno PASSATO, quindi GIA' CHIUSO seguito dal
+ contenuto, col ragionamento acceso o spento che sia. Il modello riprende in
+ modalita' testo; se lo splitter parte in modalita' ragionamento aspetta un
+ che e' gia' passato, e archivia come ragionamento tutta la
+ risposta -- content vuoto, reasoning_content pieno, stop pulito. Misurato
+ su glm53 int4, CPU: 10 e 109 caratteri di ragionamento contro
+ zero di risposta, col prompt corretto sul filo. Vale anche per glm53: la
+ regola di famiglia sopra dice dove comincia un turno NUOVO, e il turno
+ proseguito non ne apre nessuno."""
+ return (enable_thinking or ARCH == "glm53") and add_generation_prompt
class ThinkingStreamSplit:
@@ -2304,10 +2699,12 @@ def finish(self):
close = finish # interface parity with InklingStreamSplit in the streaming path
-def split_thinking_reply(text, enable_thinking=True):
+def split_thinking_reply(text, enable_thinking=True, add_generation_prompt=True):
"""Return the marker-free (thinking, answer) portions of one GLM reply."""
thinking, answer = [], []
- split = ThinkingStreamSplit(thinking.append, answer.append, initial_thinking=starts_in_reasoning(enable_thinking))
+ split = ThinkingStreamSplit(thinking.append, answer.append,
+ initial_thinking=starts_in_reasoning(enable_thinking,
+ add_generation_prompt))
split.feed(text)
split.finish()
return "".join(thinking), "".join(answer)
@@ -2483,6 +2880,11 @@ def anthropic_tools(body):
DEFAULT_CHAT_STOP_SEQUENCES = ("<|user|>", "<|observation|>")
+# Seconds to wait for the engine to exit on its own after stdin EOF (its
+# atexit teardown writes HEAT_FILE). EOF is only observed between turns,
+# so an in-flight generation delays exit; override for impatient scripts.
+_ENGINE_DRAIN_S = float(os.environ.get("COLI_ENGINE_DRAIN_S", "30"))
+
def parse_stop_sequences(body):
value = body.get("stop")
@@ -2690,8 +3092,8 @@ def generation_options(body, limit):
raise APIError(400, "Log probabilities are not supported yet.", "logprobs", "unsupported_parameter")
if body.get("frequency_penalty", 0) or body.get("presence_penalty", 0):
raise APIError(400, "Token penalties are not supported yet.", None, "unsupported_parameter")
- if body.get("seed") is not None:
- raise APIError(400, "Per-request seeds are not supported yet.", "seed", "unsupported_parameter")
+ # `seed` is accepted for request-shape compatibility and silently discarded:
+ # this server puts no per-request seed on the wire, at any temperature.
# response_format -> optional per-request grammar for the engine's grammar-forced
# draft source (#70/#148). NEVER a sampling constraint: drafts are verified, so a
# schema the engine cannot compile degrades to "no speedup", not to an error and
@@ -2797,7 +3199,7 @@ def model_arch(model):
return resolve_model(model).descriptor.id
-def cap_for_arch(arch, cap, env=None):
+def cap_for_arch(arch, cap, env=None, model=None):
"""Cap-sentinel shim (#379): CURRENT-STATE CALIBRATION, not durable core.
An absent cap (None) means different things across today's engines --
@@ -2836,6 +3238,22 @@ def cap_for_arch(arch, cap, env=None):
planned = 0
if planned >= 1:
return planned
+ if arch == "deepseek_v41" and model is not None:
+ # V4.1 only reads its argv cap, not RAM_GB. Without --auto-tier the
+ # legacy eight slots silently discarded both --ram and RAM_GB (#1666).
+ from resource_plan import build_plan
+ settings = env if env is not None else os.environ
+ ram = settings.get("RAM_GB", "0")
+ limits = family_by_id(arch).limits
+ plan = build_plan(model, ram_gb=0 if ram == "auto" else float(ram),
+ context=int(settings.get(limits.context_env, limits.default_context)),
+ gpu_indices=[])
+ slots = plan["tiers"]["ram"]["cache_slots_per_layer"]
+ if slots < 1:
+ raise ValueError("DeepSeek V4.1 RAM budget cannot hold one expert slot per layer")
+ print(f"[v41] RAM plan: {slots} expert cache slots/layer; --cap overrides",
+ file=sys.stderr)
+ return slots
return family_by_id(arch).limits.implicit_cap
@@ -2954,6 +3372,35 @@ class JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure):
return None # never let process bookkeeping break starting the engine
+def _write_all(stream, data, frame):
+ """Write every byte of `data` to `stream`, looping on short writes.
+
+ The production engine stdin is a raw, unbuffered pipe (bufsize=0 ->
+ io.FileIO), whose write() is a single os.write() and may transfer fewer
+ bytes than it was given (a signal landing mid-write, a full pipe buffer
+ on a large IMAGE frame). Discarding the return value would leave the
+ tail of a frame unsent and desynchronize the engine's stdin framing, so
+ the remainder is re-offered until it is all consumed.
+
+ Neither `None` nor 0 is progress. `RawIOBase.write` answers `None` when
+ the stream is non-blocking and could not take a single byte, and 0 says
+ the same thing with a count; re-offering the buffer after either would
+ spin forever, so both fail closed as the named engine-write error a
+ broken pipe raises."""
+ written = 0
+ total = len(data)
+ view = memoryview(data)
+ while written < total:
+ sent = stream.write(view[written:])
+ # None is RawIOBase's "not one byte went out", not an uncounted
+ # full write, so it fails closed exactly as a zero count does.
+ if sent is None or sent <= 0:
+ raise RuntimeError(
+ f"failed to write {frame} to the engine "
+ f"(stdin took {written} of {total} bytes)")
+ written += sent
+
+
class Engine:
# cap=None = "not explicitly set": a glm-arch model's engine resolves the
# 0 sentinel (8 historically, 1 on Metal+darwin+fast SSD -- colibri.c
@@ -2976,12 +3423,18 @@ def __init__(self, executable, model, cap=None, max_tokens=1024, env=None, kv_sl
child_env = dict(env or os.environ, SNAP=str(model), SERVE="1", SERVE_BATCH="1",
NGEN=str(max_tokens), KV_SLOTS=str(kv_slots))
tune_child_env(child_env, arch)
- resolved_cap = cap_for_arch(arch, cap, child_env)
+ resolved_cap = cap_for_arch(arch, cap, child_env, model=model)
child_env.pop("COLI_PROFILE_CAP", None)
child_env.pop("COLI_PLAN_CAP", None)
+ # Own process group on Windows: a CTRL_BREAK sent to the serve
+ # process group (the graceful stop, handled as SIGBREAK above) must
+ # not reach the engine — the C runtime's default would kill it
+ # before its stdin-EOF teardown (atexit -> HEAT_FILE save) can run.
+ spawn_flags = subprocess.CREATE_NEW_PROCESS_GROUP if sys.platform == "win32" else 0
self.process = subprocess.Popen(
[str(executable), str(resolved_cap)], env=child_env,
stdin=subprocess.PIPE, stdout=subprocess.PIPE, bufsize=0,
+ creationflags=spawn_flags,
)
# Keep the job handle on the instance: KILL_ON_JOB_CLOSE fires when the
# LAST handle closes, so this reference is what ties the engine (and the
@@ -3030,6 +3483,32 @@ def _fail_pending(self, error):
for events in requests:
events.put(("error", error))
+ def _write_frame(self, request_id, data, frame):
+ """Checked server->engine protocol write for CANCEL/STOP: the write
+ and its flush happen under one write_lock acquisition. Any failure
+ here -- an OSError from the pipe itself, or _write_all's own
+ fail-closed RuntimeError on a None/zero-progress write -- drops this
+ request's pending-map entry: the dispatcher only does that on this
+ id's own DONE/ERROR frame, and neither arrives when the write that
+ would have solicited one never reached the engine. An OSError is
+ additionally re-raised as a named RuntimeError rather than left as
+ itself: BrokenPipeError is a ConnectionError subclass, so an
+ unwrapped failure here would fall into do_POST's client-hangup
+ handler (`except ConnectionError: pass`) and the client would see a
+ silent connection close instead of the 500 engine_error the failure
+ actually is. _write_all's own RuntimeError is already the named
+ error this raises for an OSError, so it is re-raised as-is."""
+ try:
+ with self.write_lock:
+ _write_all(self.process.stdin, data, frame)
+ self.process.stdin.flush()
+ except Exception as error:
+ with self.pending_lock:
+ self.pending.pop(request_id, None)
+ if isinstance(error, OSError):
+ raise RuntimeError(f"failed to write {frame} to the engine ({error})") from error
+ raise
+
def _read_exact(self, size):
chunks = []
remaining = size
@@ -3250,14 +3729,21 @@ def decode_tool(data):
# annunciato subito prima del SUBMIT a cui appartengono. Deve
# partire dentro lo stesso lock, o un'altra richiesta potrebbe
# infilarsi in mezzo e prendersi l'immagine di questa.
- if image is not None:
- patches, grid_h, grid_w = image
- blob = patches.tobytes() if hasattr(patches, "tobytes") else patches
- self.process.stdin.write(
- f"IMAGE {request_id} {len(blob)} {grid_h} {grid_w}\n".encode()
- + blob + b"\n")
- self.process.stdin.write(header + payload + xpayload + b"\n")
- self.process.stdin.flush()
+ try:
+ if image is not None:
+ patches, grid_h, grid_w = image
+ blob = patches.tobytes() if hasattr(patches, "tobytes") else patches
+ try:
+ _write_all(
+ self.process.stdin,
+ f"IMAGE {request_id} {len(blob)} {grid_h} {grid_w}\n".encode()
+ + blob + b"\n", "IMAGE")
+ except OSError as error:
+ raise RuntimeError(f"failed to write IMAGE to the engine ({error})") from error
+ _write_all(self.process.stdin, header + payload + xpayload + b"\n", "SUBMIT")
+ self.process.stdin.flush()
+ except OSError as error:
+ raise RuntimeError(f"failed to write SUBMIT to the engine ({error})") from error
except Exception:
with self.pending_lock:
self.pending.pop(request_id, None)
@@ -3297,9 +3783,7 @@ def _accept(info):
# DONE frame; ClientCancelled is raised when it arrives.
if not cancel_sent and not stop_sent and cancelled and cancelled():
cancel_sent = True
- with self.write_lock:
- self.process.stdin.write(f"CANCEL {request_id}\n".encode())
- self.process.stdin.flush()
+ self._write_frame(request_id, f"CANCEL {request_id}\n".encode(), "CANCEL")
continue
if kind == "accept":
if accepted:
@@ -3311,17 +3795,13 @@ def _accept(info):
decode(value)
if stopped and stopped():
stop_sent = True
- with self.write_lock:
- self.process.stdin.write(f"STOP {request_id}\n".encode())
- self.process.stdin.flush()
+ self._write_frame(request_id, f"STOP {request_id}\n".encode(), "STOP")
elif cancelled and cancelled():
# Same admission-holding rule as the idle branch above:
# send CANCEL, then keep consuming frames until the
# engine acknowledges with ERROR CANCELLED or DONE.
cancel_sent = True
- with self.write_lock:
- self.process.stdin.write(f"CANCEL {request_id}\n".encode())
- self.process.stdin.flush()
+ self._write_frame(request_id, f"CANCEL {request_id}\n".encode(), "CANCEL")
elif kind == "echo":
# Lettura del prefill: arriva PRIMA di ogni DATA e non e' testo
# generato, quindi non passa da decode() e non entra nella
@@ -3335,14 +3815,10 @@ def _accept(info):
decode_tool(value)
if stopped and stopped():
stop_sent = True
- with self.write_lock:
- self.process.stdin.write(f"STOP {request_id}\n".encode())
- self.process.stdin.flush()
+ self._write_frame(request_id, f"STOP {request_id}\n".encode(), "STOP")
elif cancelled and cancelled():
cancel_sent = True
- with self.write_lock:
- self.process.stdin.write(f"CANCEL {request_id}\n".encode())
- self.process.stdin.flush()
+ self._write_frame(request_id, f"CANCEL {request_id}\n".encode(), "CANCEL")
elif kind == "done":
_accept({"prompt_tokens": None})
if cancel_sent:
@@ -3370,21 +3846,38 @@ def close(self):
self.closed = True
self._fail_pending(RuntimeError("colibri engine is shutting down"))
if self.process.poll() is None:
- self.process.terminate()
+ # Graceful drain first: the engine's serve loop reads requests
+ # from stdin, and EOF there is the one portable path to its
+ # atexit teardown (qt_shutdown -> HEAT_FILE save). EOF only
+ # lands between turns, so the drain wait must be generous.
+ # poll() (not the absence of TimeoutExpired) decides whether
+ # the hard-stop ladder below still needs to run: wait() may
+ # simply return None for a process (or test double) that only
+ # "terminates" when asked.
+ try:
+ self.process.stdin.close()
+ except (OSError, ValueError, AttributeError):
+ pass
try:
- self.process.wait(timeout=5)
+ self.process.wait(timeout=_ENGINE_DRAIN_S)
except subprocess.TimeoutExpired:
- # A large resident cache (e.g. 111 GB at --memory-gb 126) can
- # take longer than the grace period to unmap and free on
- # SIGTERM. SIGKILL cannot be caught, so the process is already
- # on its way out; a second timeout only means the reap has not
- # landed yet. Teardown is best-effort: never raise from here, or
- # a completed measurement is lost to a shutdown that succeeded.
- self.process.kill()
+ pass
+ if self.process.poll() is None:
+ self.process.terminate()
try:
self.process.wait(timeout=5)
except subprocess.TimeoutExpired:
- pass
+ # A large resident cache (e.g. 111 GB at --memory-gb 126) can
+ # take longer than the grace period to unmap and free on
+ # SIGTERM. SIGKILL cannot be caught, so the process is already
+ # on its way out; a second timeout only means the reap has not
+ # landed yet. Teardown is best-effort: never raise from here, or
+ # a completed measurement is lost to a shutdown that succeeded.
+ self.process.kill()
+ try:
+ self.process.wait(timeout=5)
+ except subprocess.TimeoutExpired:
+ pass
if self.dispatcher is not threading.current_thread():
self.dispatcher.join(timeout=5)
@@ -3447,6 +3940,27 @@ def __init__(self, address, engine, model_id, api_key=None, max_tokens=1024,
self._conn_by_ip = {}
self._conn_owner = {}
+ def generate(self, prompt, max_tokens, temperature, top_p, on_text, *args, **kwargs):
+ started = time.monotonic()
+ first_output = False
+
+ def measured(callback):
+ def feed(text):
+ nonlocal first_output
+ if text and not first_output:
+ first_output = True
+ self.scheduler.observe_timing("first_output_seconds", time.monotonic() - started)
+ return callback(text)
+ return feed
+
+ if kwargs.get("on_tool") is not None:
+ kwargs["on_tool"] = measured(kwargs["on_tool"])
+ try:
+ return self.engine.generate(prompt, max_tokens, temperature, top_p,
+ measured(on_text), *args, **kwargs)
+ finally:
+ self.scheduler.observe_timing("engine_call_seconds", time.monotonic() - started)
+
def process_request(self, request, client_address):
"""Refuse past the caps instead of spawning an unbounded thread."""
peer = client_address[0] if client_address else "?"
@@ -3791,6 +4305,16 @@ def do_GET(self):
try:
self._check_host()
path = urlsplit(self.path).path
+ if path == "/metrics":
+ self.require_auth()
+ data = self.server.scheduler.prometheus().encode("utf-8")
+ self.send_response(200)
+ self.send_header("Content-Type", "text/plain; version=0.0.4; charset=utf-8")
+ self.send_header("Content-Length", str(len(data)))
+ self.send_header("Cache-Control", "no-store")
+ self.end_headers()
+ self.wfile.write(data)
+ return
if path == "/health":
# Liveness is always public; hardware/scheduler internals only when a
# request is authed (or no key set), so a configured key isn't leaked
@@ -3799,6 +4323,7 @@ def do_GET(self):
if self._is_authed():
payload["scheduler"] = self.server.scheduler.snapshot()
payload["kv_slots"] = self.server.kv_slots
+ payload["continue_assistant"] = os.environ.get("COLI_CONTINUE_ASSISTANT", "1") != "0" and ARCH in CONTINUATION_FAMILIES
tiers = getattr(self.server.engine, "tiers", None) if self.server.engine else None
if tiers: payload["tiers"] = tiers
hwinfo = getattr(self.server.engine, "hwinfo", None) if self.server.engine else None
@@ -3861,14 +4386,19 @@ def do_POST(self):
self._check_host()
self.require_auth()
body = self.read_json()
- self.check_model(body)
path = urlsplit(self.path).path
+ # A client written for Jev sends "jev-latest": on that route the
+ # served model answers whatever name was asked for.
+ if path != "/v1/systemone":
+ self.check_model(body)
if path == "/v1/chat/completions":
self.chat_completion(body, request_id)
elif path == "/v1/completions":
self.completion(body, request_id)
elif path == "/v1/brio":
self.brio(body, request_id)
+ elif path == "/v1/systemone":
+ self.systemone(body, request_id)
elif path == "/v1/messages":
self.anthropic_messages(body, request_id)
else:
@@ -3914,11 +4444,11 @@ def do_POST(self):
# malformato perche' non lo scrive il modello. Prima queste due forme
# esistevano solo come script di misura: chi integrava doveva riscriverle.
@staticmethod
- def _brio_options(options, where):
+ def _brio_options(options, where, limit=64):
if not isinstance(options, list) or not options:
raise APIError(400, f"`{where}` must be a non-empty array of strings.", where)
- if len(options) > 64:
- raise APIError(400, f"`{where}` accepts at most 64 entries.", where)
+ if len(options) > limit:
+ raise APIError(400, f"`{where}` accepts at most {limit} entries.", where)
seen = set()
for option in options:
if not isinstance(option, str) or not option.strip():
@@ -3930,7 +4460,11 @@ def _brio_options(options, where):
raise APIError(400, f"`{where}` needs at least two options to choose between.", where)
return options
- def brio(self, body, request_id):
+ def brio(self, body, request_id, send=True):
+ # `send=False` returns the result instead of writing it: /v1/systemone
+ # builds a `questions` request and re-shapes the answer. `_max_options`
+ # is that caller's word too (Jev allows 255 labels); clamped.
+ option_limit = min(int(body.get("_max_options", 64) or 64), 255)
forms = [k for k in ("options", "questions", "schema") if body.get(k) is not None]
if len(forms) != 1:
raise APIError(400, "Provide exactly one of `options`, `questions` or `schema`.",
@@ -3960,7 +4494,8 @@ def brio(self, body, request_id):
if per not in ("mean", "sum"):
raise APIError(400, "`normalize` must be \"mean\" or \"sum\".", "normalize")
questions.append((text, self._brio_options(entry.get("options"),
- f"questions[{i}].options"), per))
+ f"questions[{i}].options",
+ option_limit), per))
else:
raw = body["schema"]
if not isinstance(raw, dict) or not raw:
@@ -4037,7 +4572,7 @@ def score(text, pin):
def on_accept(value):
accepted.update(value)
- self.server.engine.generate(
+ self.server.generate(
text, 0, 0.0, 1.0, lambda _chunk: None, cache_slot,
self.client_disconnected, logprobs=1, pin=pin,
on_echo=echoes.append, on_accept=on_accept)
@@ -4086,7 +4621,18 @@ def choose(prefix, choices, norm):
# le domande (o tutte le caselle) condividono. Con un livello solo
# la domanda si rilegge una volta per opzione; con due, 176 token
# invece di 496 su quattro item (misurato).
- if state_prefix and form != "options":
+ #
+ # Vale anche per la forma `options`: dentro una singola richiesta lo
+ # stato si legge comunque una volta (lo snapshot dello stato viene
+ # ripristinato quando `choose` fotografa il prefisso completo), ma
+ # il punto di ritorno sullo stato condiviso serve TRA richieste. La
+ # pagina web manda una domanda per richiesta sullo stesso documento;
+ # senza questa fotografia ogni domanda rifarebbe il prefill di tutto
+ # il documento, buttando via il "read once" che e' il senso della
+ # modalita. Con essa, ogni domanda successiva paga solo i propri
+ # token. Il costo e' uno snapshot in piu' su una richiesta one-shot,
+ # riusato o sfrattato.
+ if state_prefix:
n_state, _ = score(state_prefix, True)
prompt_max = max(prompt_max, n_state)
@@ -4136,9 +4682,143 @@ def choose(prefix, choices, norm):
"read_tokens": read_total,
"total_tokens": prompt_max + read_total},
})
- self.send_json(200, result, request_id,
- {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000)),
- "x-colibri-elapsed-ms": str(round((time.time() - started) * 1000))})
+ headers = {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000)),
+ "x-colibri-elapsed-ms": str(round((time.time() - started) * 1000))}
+ if not send:
+ result["_headers"] = headers
+ return result
+ self.send_json(200, result, request_id, headers)
+
+ # ------------------------------------------------------------ Jev-compatible
+ #
+ # POST /v1/systemone speaks the request and the reply of TypeSafe's Jev
+ # API (docs.typesafe.ai/api): a client written for it points at colibri
+ # and changes the base URL, nothing else. The three primitives map onto
+ # the `questions` form of /v1/brio, the same channel: the state is
+ # photographed once and every question pays only its own tokens.
+ #
+ # noul -> one yes/no question. `noul` is the probability of yes. The
+ # optional criteria (what true and false mean) go into the
+ # question text.
+ # choice -> the labels of `criteria` are the options; their descriptions
+ # go into the question text, because a label alone ("billing")
+ # does not say what it means. `confidence` follows their
+ # documented formula, (n * peak - 1) / (n - 1).
+ # score -> the levels of `criteria` are the options "1".."n"; `score`
+ # is the expected value under the distribution, `legend` the
+ # levels by number, `confidence` as for choice.
+ #
+ # What differs, stated rather than hidden: `model` echoes the served
+ # model, not "jev-latest"; `usage.output_tokens` counts the option tokens
+ # READ, since this engine generates nothing; validation errors are 422 as
+ # theirs are, with this server's error envelope. docs/brio.md has the
+ # mapping table.
+ _SYSTEMONE_MAX_QUESTIONS = 64
+
+ @staticmethod
+ def _systemone_text(value, where):
+ """Jev's EntryType: a string, or JSON given as an object or an array."""
+ if value is None:
+ return None
+ if isinstance(value, str):
+ return value.strip() or None
+ if isinstance(value, (dict, list)):
+ return json.dumps(value, ensure_ascii=False, indent=2)
+ raise APIError(422, f"`{where}` must be a string, an object or an array.", where)
+
+ @staticmethod
+ def _systemone_confidence(probabilities):
+ """(n * peak - 1) / (n - 1): 1 when all the mass is on one label, 0 when flat."""
+ values = list(probabilities)
+ n = len(values)
+ if n < 2:
+ return 1.0
+ return round(max(0.0, (n * max(values) - 1.0) / (n - 1)), 6)
+
+ def systemone(self, body, request_id):
+ state = self._systemone_text(body.get("state"), "state")
+ if state is None:
+ raise APIError(422, "`state` is required: the content the questions are about.", "state")
+ raw = body.get("questions")
+ if not isinstance(raw, dict) or not raw:
+ raise APIError(422, "`questions` must be a non-empty object of id: question.", "questions")
+ if len(raw) > self._SYSTEMONE_MAX_QUESTIONS:
+ raise APIError(422, f"`questions` accepts at most {self._SYSTEMONE_MAX_QUESTIONS} entries.",
+ "questions")
+ plan = [] # (id, kind, text, options, levels)
+ for qid, question in raw.items():
+ where = f"questions.{qid}"
+ if not isinstance(qid, str) or not qid.strip():
+ raise APIError(422, "Every question id must be a non-empty string.", "questions")
+ if not isinstance(question, dict):
+ raise APIError(422, f"`{where}` must be an object.", where)
+ kind = question.get("type")
+ instructions = self._systemone_text(question.get("instructions"), f"{where}.instructions")
+ criteria = question.get("criteria")
+ if kind == "noul":
+ if criteria is not None and not isinstance(criteria, dict):
+ raise APIError(422, f"`{where}.criteria` must be an object with `true` and/or `false`.",
+ f"{where}.criteria")
+ yes = self._systemone_text((criteria or {}).get("true"), f"{where}.criteria.true")
+ no = self._systemone_text((criteria or {}).get("false"), f"{where}.criteria.false")
+ text = instructions or "Is this true?"
+ if yes:
+ text += f"\nyes: {yes}"
+ if no:
+ text += f"\nno: {no}"
+ plan.append((qid, "noul", text + "\nAnswer yes or no.", ["yes", "no"], None))
+ elif kind == "choice":
+ if not isinstance(criteria, dict) or not criteria:
+ raise APIError(422, f"`{where}.criteria` must be a non-empty object of label: description.",
+ f"{where}.criteria")
+ if len(criteria) > 255:
+ raise APIError(422, f"`{where}.criteria` accepts at most 255 labels.", f"{where}.criteria")
+ labels, lines = [], []
+ for label, description in criteria.items():
+ if not isinstance(label, str) or not label.strip():
+ raise APIError(422, f"Every label of `{where}.criteria` must be a non-empty string.",
+ f"{where}.criteria")
+ labels.append(label)
+ text = self._systemone_text(description, f"{where}.criteria.{label}")
+ lines.append(f"- {label}: {text}" if text else f"- {label}")
+ if len(labels) < 2:
+ raise APIError(422, f"`{where}.criteria` needs at least two labels.", f"{where}.criteria")
+ text = (instructions or "Which of the following applies?") + "\nOptions:\n" + "\n".join(lines)
+ plan.append((qid, "choice", text + "\nAnswer with one of the options.", labels, None))
+ elif kind == "score":
+ if not isinstance(criteria, list) or not 2 <= len(criteria) <= 10:
+ raise APIError(422, f"`{where}.criteria` must be an array of 2 to 10 level descriptions.",
+ f"{where}.criteria")
+ levels = [self._systemone_text(c, f"{where}.criteria[{i}]") or f"level {i + 1}"
+ for i, c in enumerate(criteria)]
+ text = (instructions or "Rate this on the scale below.") + "\nScale:\n" + \
+ "\n".join(f"{i + 1}: {d}" for i, d in enumerate(levels))
+ plan.append((qid, "score", text + "\nAnswer with the number.",
+ [str(i + 1) for i in range(len(levels))], levels))
+ else:
+ raise APIError(422, f"`{where}.type` must be \"noul\", \"choice\" or \"score\".", f"{where}.type")
+ inner = {"state": state, "_max_options": 255,
+ "questions": [{"question": text, "options": options} for _, _, text, options, _ in plan]}
+ result = self.brio(inner, request_id, send=False)
+ answers = {}
+ for (qid, kind, _, options, levels), got in zip(plan, result["answers"]):
+ p = {c["option"]: c["p"] for c in got["choices"]}
+ if kind == "noul":
+ answers[qid] = {"type": "noul", "noul": round(p.get("yes", 0.0), 6)}
+ elif kind == "choice":
+ answers[qid] = {"type": "choice", "choice": got["answer"],
+ "probabilities": {o: round(p[o], 6) for o in options},
+ "confidence": self._systemone_confidence(p.values())}
+ else:
+ answers[qid] = {"type": "score",
+ "score": round(sum(int(k) * v for k, v in p.items()), 6),
+ "legend": {str(i + 1): d for i, d in enumerate(levels)},
+ "probabilities": {o: round(p[o], 6) for o in options},
+ "confidence": self._systemone_confidence(p.values())}
+ reply = {"model": self.server.model_id, "answers": answers,
+ "usage": {"input_tokens": result["usage"]["prompt_tokens"],
+ "output_tokens": result["usage"]["read_tokens"]}}
+ self.send_json(200, reply, request_id, result.get("_headers"))
def _fail(self, error, request_id):
"""Report an error, unless the response is already on the wire. Once a streaming 200
@@ -4157,7 +4837,8 @@ def error_body(self, error):
return {"type": "error", "error": {"type": error.error_type, "message": error.message}}
def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=None,
- enable_thinking=False, audio=None, image=None):
+ enable_thinking=False, audio=None, image=None,
+ add_generation_prompt=True):
# COLI_DEBUG tees the engine transaction to stderr: 1 = decoded output stream only,
# 2 = both sides (rendered prompt + output). render_chat already folds prior turns and
# tool results into `prompt`, so level 2 is the full conversation the engine saw.
@@ -4205,7 +4886,8 @@ def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=Non
completion_id = id_prefix + uuid.uuid4().hex
created = int(time.time())
- with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission:
+ with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission, \
+ contextlib.ExitStack() as stream_cleanup:
queue_wait, cache_slot = admission
queue_headers = {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000))}
if not stream:
@@ -4217,7 +4899,7 @@ def generation(self, body, prompt, request_id, chat, tools=None, tool_choice=Non
def generation_stopped():
return stop_filter.stopped() or sideband.stopped()
- stats = self.server.engine.generate(
+ stats = self.server.generate(
prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot,
self.client_disconnected, grammar=grammar, stopped=generation_stopped,
**({"on_tool": sideband.feed} if sideband.enabled else {}),
@@ -4233,7 +4915,8 @@ def generation_stopped():
# #597 item 4: GLM emits reasoning then then the answer. Route the
# reasoning to reasoning_content instead of dumping it (or the raw )
# into the visible answer / tool-call parser.
- reasoning, text = split_thinking_reply(text, enable_thinking)
+ reasoning, text = split_thinking_reply(text, enable_thinking,
+ add_generation_prompt)
length_finish = "length" if stats["length_limited"] else "stop"
if chat and tools:
content, calls = parse_arch_tool_calls(text, tools, sideband.reply())
@@ -4356,6 +5039,8 @@ def start_stream(_accept_info=None):
"logprobs": None, "finish_reason": None}])
ka_thread[0] = threading.Thread(target=_keepalive, daemon=True)
ka_thread[0].start()
+ stream_cleanup.callback(ka_thread[0].join, timeout=2)
+ stream_cleanup.callback(ka_stop.set)
if chat and tools:
# Suppress tool-call markers from the streamed content and parse the authoritative
# calls from the FULL reply after generation. Hold back a marker-length tail so a
@@ -4388,7 +5073,8 @@ def feed_content(chunk): # answer text only (post-)
# #597: keep GLM reasoning out of the tool-call buffer — a think splitter sends it
# to reasoning_content and passes only the answer text on to feed_content/parser.
think = (ThinkingStreamSplit(emit_reasoning, feed_content,
- initial_thinking=starts_in_reasoning(enable_thinking))
+ initial_thinking=starts_in_reasoning(
+ enable_thinking, add_generation_prompt))
if glm_think else None)
def emit_tools(chunk):
if dbg_echo:
@@ -4399,7 +5085,7 @@ def emit_tools(chunk):
def generation_stopped():
return stop_filter.stopped() or sideband.stopped()
- stats = self.server.engine.generate(
+ stats = self.server.generate(
prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot,
self.client_disconnected, grammar=grammar, stopped=generation_stopped,
**({"on_tool": sideband.feed} if sideband.enabled else {}),
@@ -4423,8 +5109,10 @@ def generation_stopped():
if splitter is not None: # inkling content/marker splitter
content_split = splitter
elif glm_think: # GLM reasoning → reasoning_content
- content_split = ThinkingStreamSplit(emit_reasoning, emit,
- initial_thinking=starts_in_reasoning(enable_thinking))
+ content_split = ThinkingStreamSplit(
+ emit_reasoning, emit,
+ initial_thinking=starts_in_reasoning(enable_thinking,
+ add_generation_prompt))
else:
content_split = None
def emit_plain(chunk):
@@ -4432,7 +5120,7 @@ def emit_plain(chunk):
sys.stderr.write(chunk); sys.stderr.flush()
(content_split.feed if content_split else emit)(chunk)
stop_filter = StopFilter(stop_sequences, emit_plain, ignore_leading_stop)
- stats = self.server.engine.generate(
+ stats = self.server.generate(
prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot,
self.client_disconnected, grammar=grammar, stopped=stop_filter.stopped,
on_accept=start_stream, **({"audio": audio} if audio else {}),
@@ -4539,10 +5227,13 @@ def chat_completion(self, body, request_id):
raise APIError(400, "one image per request for now; the engine "
"holds a single pending image.", "messages")
image = images[0] if images else None
+ add_generation_prompt = resolve_generation_prompt(messages, body)
prompt = render_chat_for_arch(messages, enable_thinking, reasoning_effort,
- tools, tool_choice, audio_out=audio_clips)
+ tools, tool_choice, audio_out=audio_clips,
+ add_generation_prompt=add_generation_prompt)
self.generation(body, prompt, request_id, True, tools, tool_choice,
enable_thinking=enable_thinking,
+ add_generation_prompt=add_generation_prompt,
audio=b"".join(audio_clips) if audio_clips else None,
image=image)
@@ -4581,12 +5272,16 @@ def anthropic_messages(self, body, request_id):
if tool_choice == "none":
tools = None
default_effort = "xhigh" if ARCH == "qwen38" and thinking is None else "high"
+ add_generation_prompt = resolve_generation_prompt(messages, body)
prompt = render_chat_for_arch(messages, enable_thinking,
default_effort if enable_thinking else None,
- tools, tool_choice)
- self.anthropic_generation(translated, prompt, request_id, tools, enable_thinking)
+ tools, tool_choice,
+ add_generation_prompt=add_generation_prompt)
+ self.anthropic_generation(translated, prompt, request_id, tools, enable_thinking,
+ add_generation_prompt)
- def anthropic_generation(self, body, prompt, request_id, tools, enable_thinking):
+ def anthropic_generation(self, body, prompt, request_id, tools, enable_thinking,
+ add_generation_prompt=True):
maximum, temperature, top_p, grammar, _stop_sequences = generation_options(
body, self.server.max_tokens)
# Same policy as /v1/chat/completions: `body` is the translated OpenAI-shaped
@@ -4618,7 +5313,8 @@ def blocks_and_stop(text, stats, tool_reply=None):
if ARCH == "inkling":
text, reasoning = split_inkling(text)
elif enable_thinking:
- reasoning, text = split_thinking_reply(text)
+ reasoning, text = split_thinking_reply(text, enable_thinking,
+ add_generation_prompt)
if enable_thinking:
content.append({"type": "thinking", "thinking": reasoning,
"signature": ANTHROPIC_LOCAL_SIGNATURE})
@@ -4638,7 +5334,8 @@ def blocks_and_stop(text, stats, tool_reply=None):
reason = "tool_calls" if calls else ("length" if stats["length_limited"] else "stop")
return content, self.ANTHROPIC_STOP[reason]
- with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission:
+ with self.server.scheduler.admit(self.client_disconnected, cache_slot) as admission, \
+ contextlib.ExitStack() as stream_cleanup:
queue_wait, cache_slot = admission
queue_headers = {"x-colibri-queue-wait-ms": str(round(queue_wait * 1000))}
if not stream:
@@ -4650,7 +5347,7 @@ def blocks_and_stop(text, stats, tool_reply=None):
def generation_stopped():
return stop_filter.stopped() or sideband.stopped()
- stats = self.server.engine.generate(
+ stats = self.server.generate(
prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot,
self.client_disconnected, grammar=grammar, stopped=generation_stopped,
**({"on_tool": sideband.feed} if sideband.enabled else {}))
@@ -4719,6 +5416,8 @@ def keepalive():
"content_block": {"type": "text", "text": ""}})
ka_thread = threading.Thread(target=keepalive, daemon=True)
ka_thread.start()
+ stream_cleanup.callback(ka_thread.join, timeout=2)
+ stream_cleanup.callback(ka_stop.set)
raw = []
sideband = ToolSideband(ARCH == "kimi" and bool(tools), stop_sequences,
@@ -4780,7 +5479,8 @@ def close_thinking():
# lo splitter serve pure col ragionamento "spento", o il
# pensiero finisce incollato davanti alla risposta.
split = (ThinkingStreamSplit(emit_thinking, emit_answer, close_thinking)
- if starts_in_reasoning(enable_thinking) else None)
+ if starts_in_reasoning(enable_thinking, add_generation_prompt)
+ else None)
def on_text(chunk):
raw.append(chunk)
@@ -4791,7 +5491,7 @@ def on_text(chunk):
def generation_stopped():
return stop_filter.stopped() or sideband.stopped()
- stats = self.server.engine.generate(
+ stats = self.server.generate(
prompt, maximum, temperature, top_p, stop_filter.feed, cache_slot,
lambda: not connected[0], grammar=grammar, stopped=generation_stopped,
**({"on_tool": sideband.feed} if sideband.enabled else {}))
@@ -4887,6 +5587,16 @@ def serve(model, host="127.0.0.1", port=8000, model_id=None, api_key=None,
server.engine = runtime
print(f"OpenAI-compatible API listening on http://{host}:{port}/v1", file=sys.stderr)
signal.signal(signal.SIGTERM, lambda *_: threading.Thread(target=server.shutdown, daemon=True).start())
+ # On Windows SIGTERM is never delivered (os.kill is TerminateProcess);
+ # CTRL_BREAK — the one console signal a controller CAN target at this
+ # process group — arrives as SIGBREAK. Without this handler it kills
+ # the serve loop outright, skipping the finally that drains the
+ # engine (stdin EOF -> atexit -> HEAT_FILE save). The engine child
+ # runs in its own process group (see Engine.__init__) and does not
+ # receive this event.
+ if hasattr(signal, "SIGBREAK"):
+ signal.signal(signal.SIGBREAK,
+ lambda *_: threading.Thread(target=server.shutdown, daemon=True).start())
try:
server.serve_forever()
except KeyboardInterrupt:
diff --git a/c/oracle.h b/c/oracle.h
new file mode 100644
index 000000000..7790cb92a
--- /dev/null
+++ b/c/oracle.h
@@ -0,0 +1,97 @@
+/* Validation for the GLM reference file; no model or Python dependency. */
+#ifndef COLI_ORACLE_H
+#define COLI_ORACLE_H
+#include
+#include
+#include "json.h"
+
+typedef struct {
+ int *prompt, *full, *tf;
+ int np, nfull;
+} OracleRef;
+
+static void oracle_ref_free(OracleRef *r) {
+ free(r->prompt); free(r->full); free(r->tf);
+ memset(r, 0, sizeof(*r));
+}
+
+static int *oracle_read_ids(jval *root, const char *key, int vocab, int *n) {
+ jval *a=json_get(root,key);
+ *n=0;
+ if (!a || a->t!=J_ARR || a->len<1) {
+ fprintf(stderr,"[ORACLE] %s must be a nonempty token array\n",key);
+ return NULL;
+ }
+ int *ids=malloc((size_t)a->len*sizeof(*ids));
+ if (!ids) { fprintf(stderr,"[ORACLE] out of memory reading %s\n",key); return NULL; }
+ for (int i=0; ilen; i++) {
+ jval *v=a->kids[i];
+ if (v->t!=J_NUM || !isfinite(v->num) || v->num<0 || v->num>=vocab ||
+ v->num!=floor(v->num)) {
+ fprintf(stderr,"[ORACLE] %s[%d] must be an integer token in [0,%d)\n",key,i,vocab);
+ free(ids); return NULL;
+ }
+ ids[i]=(int)v->num;
+ }
+ *n=a->len;
+ return ids;
+}
+
+static int oracle_ref_parse(const char *text, int vocab, int teacher_forcing, OracleRef *r) {
+ memset(r,0,sizeof(*r));
+ jval *root=json_parse_checked(text);
+ if (!root || root->t!=J_OBJ || vocab<1) {
+ fprintf(stderr,"[ORACLE] invalid reference JSON or vocabulary\n");
+ json_free(root); return 0;
+ }
+ /* Duplicate keys could otherwise silently select a stale prediction array. */
+ const char *keys[]={"prompt_ids","full_ids","tf_pred"};
+ for (int k=0; k<3; k++) {
+ int count=0;
+ for (int i=0; ilen; i++) if (!strcmp(root->keys[i],keys[k])) count++;
+ if (count>1) {
+ fprintf(stderr,"[ORACLE] duplicate reference field %s\n",keys[k]);
+ goto fail;
+ }
+ }
+ r->prompt=oracle_read_ids(root,"prompt_ids",vocab,&r->np);
+ r->full=oracle_read_ids(root,"full_ids",vocab,&r->nfull);
+ if (!r->prompt || !r->full) goto fail;
+ if (r->nfullnp || (!teacher_forcing && r->nfull==r->np) ||
+ memcmp(r->prompt,r->full,(size_t)r->np*sizeof(int))) {
+ fprintf(stderr,"[ORACLE] full_ids must start with prompt_ids and include the compared tokens\n");
+ goto fail;
+ }
+ if (teacher_forcing) {
+ int ntf=0;
+ r->tf=oracle_read_ids(root,"tf_pred",vocab,&ntf);
+ if (!r->tf) goto fail;
+ if (ntf!=r->nfull) {
+ fprintf(stderr,"[ORACLE] tf_pred length %d != full_ids length %d\n",ntf,r->nfull);
+ goto fail;
+ }
+ }
+ json_free(root); return 1;
+fail:
+ json_free(root); oracle_ref_free(r); return 0;
+}
+
+static int oracle_logits_finite(const float *lo, int vocab) {
+ for (int i=0; i'9') return 0;
+ char *end;
+ errno=0;
+ long value=strtol(text,&end,10);
+ if (errno==ERANGE || *end || value<0 || value>=total) return 0;
+ *allowed=(int)value;
+ return 1;
+}
+#endif
diff --git a/c/qgemv.h b/c/qgemv.h
new file mode 100644
index 000000000..62f53d7f1
--- /dev/null
+++ b/c/qgemv.h
@@ -0,0 +1,147 @@
+#ifndef COLIBRI_QGEMV_H
+#define COLIBRI_QGEMV_H
+/* Plain (non-group-scaled) int8 GEMV: one f32 scale per output row, q[O,I]
+ * int8 row-major. Its own header so tests/test_qgemv.c can link the very
+ * kernel the engine runs; qwen36.c carries a main and cannot be linked into
+ * a test.
+ *
+ * This is qwen36's dense-projection kernel (lm_head and any non-expert
+ * quantized matmul); the group-scaled sibling used for expert matmuls lives
+ * in gsgemv.h. Its float operations must stay in exactly this order -- float
+ * addition is not associative and the engine's token stream is required to
+ * be byte-identical to the reference. tests/test_qgemv.c holds the
+ * pre-restructure kernel verbatim and compares raw float bits, so any
+ * reassociation fails there rather than surfacing as drifted text much
+ * later.
+ *
+ * The SSE4.1 tier below is the exception to that byte-identical rule, by
+ * design: it is new code with no pre-existing output to match, and it is
+ * checked by tests/test_qgemv.c against a tolerance, not memcmp. It routes
+ * its FMA and float loads through sse41_kernels.h -- the same shared
+ * primitives header olmoe.c uses -- rather than inlining its own copy. */
+#include
+#include
+#include
+#if defined(__ARM_NEON)
+#include
+#endif
+#if (defined(__AVX2__) && defined(__FMA__)) || defined(__SSE4_1__)
+#include
+#endif
+#if defined(__SSE4_1__)
+#include "sse41_kernels.h"
+#endif
+
+#if defined(__ARM_NEON)
+static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) {
+ int32x4_t acc = vdupq_n_s32(0);
+ int8x16_t va = vld1q_s8(a), vb = vld1q_s8(b);
+#if defined(__ARM_FEATURE_DOTPROD)
+ acc = vdotq_s32(acc, va, vb);
+#else
+ acc = vpadalq_s16(acc, vmull_s8(vget_low_s8(va), vget_low_s8(vb)));
+ acc = vpadalq_s16(acc, vmull_s8(vget_high_s8(va), vget_high_s8(vb)));
+#endif
+ return vaddvq_s32(acc);
+}
+#endif
+
+static void matmul_q(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) {
+#if defined(__ARM_NEON)
+ /* IDOT is opt-in, not default-on: this path quantizes the ACTIVATIONS to
+ * Q8_0 per 16-element block, which the scalar path does not, so the two are
+ * not numerically equivalent. olmoe shipped it default-on and it cost
+ * token-exactness end to end (#1044, fixed in af48fe8 by making it opt-in);
+ * qwen36 inherited the same default from the same family of kernels. The
+ * tiny-oracle gate would not have caught it -- that job runs on x86. */
+ static int idot = -1;
+ if (idot < 0) { const char *e = getenv("IDOT"); idot = (e && atoi(e)); }
+ if (idot && I % 16 == 0 && I <= 4096) {
+ int nb = I / 16; int8_t xi[4096]; float xs[256];
+ for (int b = 0; b < nb; b++) {
+ const float *xb = x + b*16;
+ float am = 0.f; for (int i = 0; i < 16; i++) { float a = fabsf(xb[i]); if (a > am) am = a; }
+ float s = am/127.f; if (s < 1e-12f) s = 1e-12f;
+ xs[b] = s; float inv = 1.f/s;
+ for (int i = 0; i < 16; i++) xi[b*16+i] = (int8_t)lrintf(xb[i]*inv);
+ }
+ #pragma omp parallel for schedule(static)
+ for (int o = 0; o < O; o++) {
+ const int8_t *w = q + (int64_t)o * I;
+ float acc = 0.f;
+ for (int b = 0; b < nb; b++) acc += xs[b]*(float)dot_i8_16(xi+b*16, w+b*16);
+ y[o] = acc * scale[o];
+ }
+ return;
+ }
+#endif
+#if defined(__AVX2__) && defined(__FMA__)
+ /* Hand-vectorized int8->f32 GEMV (gcc does not auto-vectorize the
+ * convert+accumulate chain). 32 weights per iteration, FMA accumulate. */
+ #pragma omp parallel for schedule(static) if(O >= 256)
+ for (int o = 0; o < O; o++) {
+ const int8_t *w = q + (int64_t)o * I;
+ __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps();
+ __m256 a2 = _mm256_setzero_ps(), a3 = _mm256_setzero_ps();
+ int i = 0;
+ for (; i + 32 <= I; i += 32) {
+ __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i));
+ __m128i b1 = _mm_loadu_si128((const __m128i*)(w + i + 16));
+ a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0);
+ a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1);
+ a2 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+16), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a2);
+ a3 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+24), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a3);
+ }
+ a0 = _mm256_add_ps(_mm256_add_ps(a0,a1), _mm256_add_ps(a2,a3));
+ __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1));
+ s = _mm_add_ps(s, _mm_movehl_ps(s,s));
+ s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1));
+ float acc = _mm_cvtss_f32(s);
+ for (; i < I; i++) acc += x[i] * (float)w[i];
+ y[o] = acc * scale[o];
+ }
+#elif defined(__SSE4_1__)
+ /* SSE4.1 tier: one rung below the AVX2 branch above, same shape --
+ * single output row per iteration, no row-interleaving and no runtime
+ * gate (matmul_q has no group scale to gate on, unlike matmul_q_gs's
+ * SSE4.1 branch in gsgemv.h). 128-bit lanes instead of 256-bit: four
+ * accumulators of 4 floats each, 16 weights per iteration where AVX2
+ * manages 32. The multiply-add and the float loads route through
+ * sse41_kernels.h (COLIBRI_FMA / colibri_sse41_loadu_ps), the same
+ * shared primitives olmoe.c's canary uses, instead of an inline copy of
+ * the intrinsics. This tier is new code with no pre-existing
+ * byte-identical output to match, so tests/test_qgemv.c checks it
+ * against a tolerance, not memcmp -- unlike the AVX2 and scalar tiers
+ * above/below. */
+ #pragma omp parallel for schedule(static) if(O >= 256)
+ for (int o = 0; o < O; o++) {
+ const int8_t *w = q + (int64_t)o * I;
+ __m128 a0 = _mm_setzero_ps(), a1 = _mm_setzero_ps();
+ __m128 a2 = _mm_setzero_ps(), a3 = _mm_setzero_ps();
+ int i = 0;
+ for (; i + 16 <= I; i += 16) {
+ __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i));
+ a0 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(b0)), a0);
+ a1 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+4), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,4))), a1);
+ a2 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+8), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,8))), a2);
+ a3 = COLIBRI_FMA(colibri_sse41_loadu_ps(x+i+12), _mm_cvtepi32_ps(_mm_cvtepi8_epi32(_mm_srli_si128(b0,12))), a3);
+ }
+ a0 = _mm_add_ps(_mm_add_ps(a0,a1), _mm_add_ps(a2,a3));
+ a0 = _mm_add_ps(a0, _mm_movehl_ps(a0,a0));
+ a0 = _mm_add_ss(a0, _mm_shuffle_ps(a0,a0,1));
+ float acc = _mm_cvtss_f32(a0);
+ for (; i < I; i++) acc += x[i] * (float)w[i];
+ y[o] = acc * scale[o];
+ }
+#else
+ #pragma omp parallel for schedule(static)
+ for (int o = 0; o < O; o++) {
+ const int8_t *w = q + (int64_t)o * I;
+ float acc = 0.f;
+ for (int i = 0; i < I; i++) acc += x[i] * (float)w[i];
+ y[o] = acc * scale[o];
+ }
+#endif
+}
+
+#endif /* COLIBRI_QGEMV_H */
diff --git a/c/quant.h b/c/quant.h
index f53ad4360..a62c36ffb 100644
--- a/c/quant.h
+++ b/c/quant.h
@@ -14,33 +14,10 @@
#include
#endif
-/* ---- SIMD includes -------------------------------------------------------- */
-#ifdef __AVX2__
-#include
-static inline float hsum256(__m256 v){
- __m128 lo=_mm256_castps256_ps128(v), hi=_mm256_extractf128_ps(v,1);
- lo=_mm_add_ps(lo,hi); __m128 sh=_mm_movehl_ps(lo,lo); lo=_mm_add_ps(lo,sh);
- sh=_mm_shuffle_ps(lo,lo,1); lo=_mm_add_ss(lo,sh); return _mm_cvtss_f32(lo);
-}
-static inline int hsum256_i32(__m256i v){
- __m128i lo=_mm256_castsi256_si128(v), hi=_mm256_extracti128_si256(v,1);
- lo=_mm_add_epi32(lo,hi); lo=_mm_hadd_epi32(lo,lo); lo=_mm_hadd_epi32(lo,lo);
- return _mm_cvtsi128_si32(lo);
-}
-#endif
-#if defined(__AVXVNNI__) && defined(__AVX2__)
-static inline int hsum128_i32(__m128i v){
- v=_mm_hadd_epi32(v,v); v=_mm_hadd_epi32(v,v); return _mm_cvtsi128_si32(v);
-}
-#endif
-#ifdef __ARM_NEON
-#include
-#endif
-#ifdef __VSX__
-#include
-#undef vector
-#undef pixel
-#undef bool
+#include "idot.h" /* SIMD prelude + the integer dot kernels, shared with qwen36 */
+
+#if defined(__SSE4_1__)
+#include "sse41_kernels.h"
#endif
/* ---- AVX-512 int4->float accumulator -------------------------------------- */
@@ -168,8 +145,17 @@ static void matmul_i4(float *y, const float *x, const uint8_t *q4, const float *
static void matmul_i4_grouped(float *y, const float *x, const uint8_t *q4, const float *scale,
int S, int I, int O, int gs){
int rb=(I+1)/2; int ng=(I+gs-1)/gs;
+ int o0=0;
+#if defined(__SSE4_1__) && !defined(__AVX2__)
+ /* Even group sizes keep every group start on a low-nibble boundary. */
+ if(!(gs&1)){
+ o0=O&~3;
+ if(o0) matmul_i4_grouped_sse41_rows4(y,x,q4,scale,S,I,O,gs,rb,ng,o0);
+ if(o0==O) return;
+ }
+#endif
#pragma omp parallel for schedule(static)
- for(int o=0;oI) blen=I-base;
+ float acc0=0,acc1=0,acc2=0,acc3=0;
+ for(int i=base;iamax)amax=a; }
- float s=amax/127.f; if(s<1e-12f) s=1e-12f; float inv=1.f/s;
- for(int i=0;i bit-identico. Stessa struttura
- * dei 4 accumulatori del ramo NEON piu' sotto.
- * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer
- * adds are associative, so the result is bit-identical (mirrors the NEON path). */
- __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128();
- for(;i+64<=I;i+=64){
- __m128i w0=_mm_loadu_si128((const __m128i*)(w+i)), x0=_mm_loadu_si128((const __m128i*)(x+i));
- __m128i w1=_mm_loadu_si128((const __m128i*)(w+i+16)), x1=_mm_loadu_si128((const __m128i*)(x+i+16));
- __m128i w2=_mm_loadu_si128((const __m128i*)(w+i+32)), x2=_mm_loadu_si128((const __m128i*)(x+i+32));
- __m128i w3=_mm_loadu_si128((const __m128i*)(w+i+48)), x3=_mm_loadu_si128((const __m128i*)(x+i+48));
- a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
- a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
- a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2));
- a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3));
- }
- __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3));
- for(;i+16<=I;i+=16){
- __m128i wv=_mm_loadu_si128((const __m128i*)(w+i));
- __m128i xv=_mm_loadu_si128((const __m128i*)(x+i));
- acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(wv),_mm_sign_epi8(xv,wv));
- }
- sum=hsum128_i32(acc);
-#elif defined(__AVX2__)
- __m256i acc=_mm256_setzero_si256(); const __m256i ones=_mm256_set1_epi16(1);
- for(;i+32<=I;i+=32){
- __m256i wv=_mm256_loadu_si256((const __m256i*)(w+i));
- __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
- __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
- acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
- }
- sum=hsum256_i32(acc);
-#elif defined(__ARM_NEON)
-#if defined(__ARM_FEATURE_DOTPROD)
- int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
- for(;i+64<=I;i+=64){
- a0=vdotq_s32(a0,vld1q_s8(w+i), vld1q_s8(x+i));
- a1=vdotq_s32(a1,vld1q_s8(w+i+16),vld1q_s8(x+i+16));
- a2=vdotq_s32(a2,vld1q_s8(w+i+32),vld1q_s8(x+i+32));
- a3=vdotq_s32(a3,vld1q_s8(w+i+48),vld1q_s8(x+i+48));
- }
- int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
- for(;i+16<=I;i+=16) acc=vdotq_s32(acc,vld1q_s8(w+i),vld1q_s8(x+i));
- sum=vaddvq_s32(acc);
-#else
- int32x4_t acc=vdupq_n_s32(0);
- for(;i+16<=I;i+=16){
- int8x16_t wv=vld1q_s8(w+i), xv=vld1q_s8(x+i);
- int16x8_t p=vmull_s8(vget_low_s8(wv),vget_low_s8(xv));
- p=vmlal_s8(p,vget_high_s8(wv),vget_high_s8(xv));
- acc=vpadalq_s16(acc,p);
- }
- sum=vaddvq_s32(acc);
-#endif
-#elif defined(__VSX__)
- __vector signed int acc=vec_splats(0);
- const __vector signed char vz=vec_splats((signed char)0);
- for(;i+16<=I;i+=16){
- __vector signed char wv=vec_xl(0,(const signed char*)(w+i));
- __vector signed char xv=vec_xl(0,(const signed char*)(x+i));
- __vector __bool char neg=vec_cmplt(wv,vz);
- __vector signed char xs=vec_sel(xv,vec_sub(vz,xv),neg);
- __vector unsigned char wa=(__vector unsigned char)vec_sel(wv,vec_sub(vz,wv),neg);
- acc=vec_msum(xs,wa,acc);
- }
- sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
-#endif
- for(;i>1)));
- __m256i lo=_mm256_and_si256(by,m4v), hi=_mm256_and_si256(_mm256_srli_epi16(by,4),m4v);
- __m256i z0=_mm256_unpacklo_epi8(lo,hi), z1=_mm256_unpackhi_epi8(lo,hi);
- __m512i wv=_mm512_sub_epi8(_mm512_inserti64x4(_mm512_castsi256_si512(z0),z1,1),b8v);
- __m512i xv=_mm512_permutexvar_epi64(xidx,_mm512_loadu_si512((const void*)(x+i)));
- __mmask64 neg=_mm512_movepi8_mask(wv);
- __m512i xs=_mm512_mask_sub_epi8(xv,neg,_mm512_setzero_si512(),xv);
- acc=_mm512_dpbusd_epi32(acc,_mm512_abs_epi8(wv),xs);
- }
- sum=_mm512_reduce_add_epi32(acc);
-#elif defined(__AVXVNNI__) && defined(__AVX2__)
- /* 4 accumulatori indipendenti (64 elementi = 32 byte packed/iter): un solo acc
- * incatena i vpdpbusd (latenza-bound ~5c). Somme intere associative -> bit-identico.
- * Stessa struttura dei 4 accumulatori del ramo NEON piu' sotto.
- * EN: four independent accumulators break the serial vpdpbusd->acc chain; integer
- * adds are associative, so the result is bit-identical (mirrors the NEON path). */
- const __m128i m4=_mm_set1_epi8(0x0F); const __m128i b8=_mm_set1_epi8(8);
- __m128i a0=_mm_setzero_si128(),a1=_mm_setzero_si128(),a2=_mm_setzero_si128(),a3=_mm_setzero_si128();
- for(;i+64<=I;i+=64){
- __m128i by0=_mm_loadu_si128((const __m128i*)(w4+(i>>1))); /* elem i..i+31 */
- __m128i by1=_mm_loadu_si128((const __m128i*)(w4+(i>>1)+16)); /* elem i+32..i+63 */
- __m128i lo0=_mm_and_si128(by0,m4), hi0=_mm_and_si128(_mm_srli_epi16(by0,4),m4);
- __m128i lo1=_mm_and_si128(by1,m4), hi1=_mm_and_si128(_mm_srli_epi16(by1,4),m4);
- __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo0,hi0),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo0,hi0),b8);
- __m128i w2=_mm_sub_epi8(_mm_unpacklo_epi8(lo1,hi1),b8), w3=_mm_sub_epi8(_mm_unpackhi_epi8(lo1,hi1),b8);
- __m128i x0=_mm_loadu_si128((const __m128i*)(x+i)), x1=_mm_loadu_si128((const __m128i*)(x+i+16));
- __m128i x2=_mm_loadu_si128((const __m128i*)(x+i+32)), x3=_mm_loadu_si128((const __m128i*)(x+i+48));
- a0=_mm_dpbusd_epi32(a0,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
- a1=_mm_dpbusd_epi32(a1,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
- a2=_mm_dpbusd_epi32(a2,_mm_abs_epi8(w2),_mm_sign_epi8(x2,w2));
- a3=_mm_dpbusd_epi32(a3,_mm_abs_epi8(w3),_mm_sign_epi8(x3,w3));
- }
- __m128i acc=_mm_add_epi32(_mm_add_epi32(a0,a1),_mm_add_epi32(a2,a3));
- for(;i+32<=I;i+=32){ /* 32-nibble remainder: 2 dpbusd, same unpack */
- __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
- __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
- __m128i w0=_mm_sub_epi8(_mm_unpacklo_epi8(lo,hi),b8), w1=_mm_sub_epi8(_mm_unpackhi_epi8(lo,hi),b8);
- __m128i x0=_mm_loadu_si128((const __m128i*)(x+i));
- __m128i x1=_mm_loadu_si128((const __m128i*)(x+i+16));
- acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w0),_mm_sign_epi8(x0,w0));
- acc=_mm_dpbusd_epi32(acc,_mm_abs_epi8(w1),_mm_sign_epi8(x1,w1));
- }
- sum=hsum128_i32(acc);
-#elif defined(__AVX2__)
- const __m128i m4=_mm_set1_epi8(0x0F); const __m256i b8=_mm256_set1_epi8(8);
- const __m256i ones=_mm256_set1_epi16(1);
- __m256i acc=_mm256_setzero_si256();
- for(;i+32<=I;i+=32){
- __m128i by=_mm_loadu_si128((const __m128i*)(w4+(i>>1)));
- __m128i lo=_mm_and_si128(by,m4), hi=_mm_and_si128(_mm_srli_epi16(by,4),m4);
- __m128i n0=_mm_unpacklo_epi8(lo,hi), n1=_mm_unpackhi_epi8(lo,hi);
- __m256i wv=_mm256_sub_epi8(_mm256_set_m128i(n1,n0),b8);
- __m256i xv=_mm256_loadu_si256((const __m256i*)(x+i));
- __m256i p=_mm256_maddubs_epi16(_mm256_sign_epi8(wv,wv),_mm256_sign_epi8(xv,wv));
- acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p,ones));
- }
- sum=hsum256_i32(acc);
-#elif defined(__ARM_NEON)
- const uint8x16_t m4q=vdupq_n_u8(0x0F); const int8x16_t b8q=vdupq_n_s8(8);
-#if defined(__ARM_FEATURE_DOTPROD)
- int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0);
- for(;i+64<=I;i+=64){
- uint8x16_t byA=vld1q_u8(w4+(i>>1)), byB=vld1q_u8(w4+(i>>1)+16);
- uint8x16x2_t zA=vzipq_u8(vandq_u8(byA,m4q), vshrq_n_u8(byA,4));
- uint8x16x2_t zB=vzipq_u8(vandq_u8(byB,m4q), vshrq_n_u8(byB,4));
- a0=vdotq_s32(a0,vsubq_s8(vreinterpretq_s8_u8(zA.val[0]),b8q),vld1q_s8(x+i));
- a1=vdotq_s32(a1,vsubq_s8(vreinterpretq_s8_u8(zA.val[1]),b8q),vld1q_s8(x+i+16));
- a2=vdotq_s32(a2,vsubq_s8(vreinterpretq_s8_u8(zB.val[0]),b8q),vld1q_s8(x+i+32));
- a3=vdotq_s32(a3,vsubq_s8(vreinterpretq_s8_u8(zB.val[1]),b8q),vld1q_s8(x+i+48));
- }
- int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
- for(;i+32<=I;i+=32){
- uint8x16_t by=vld1q_u8(w4+(i>>1));
- uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
- acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q),vld1q_s8(x+i));
- acc=vdotq_s32(acc,vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q),vld1q_s8(x+i+16));
- }
- sum=vaddvq_s32(acc);
-#else
- int32x4_t acc=vdupq_n_s32(0);
- for(;i+32<=I;i+=32){
- uint8x16_t by=vld1q_u8(w4+(i>>1));
- uint8x16x2_t z=vzipq_u8(vandq_u8(by,m4q), vshrq_n_u8(by,4));
- int8x16_t w0=vsubq_s8(vreinterpretq_s8_u8(z.val[0]),b8q);
- int8x16_t w1=vsubq_s8(vreinterpretq_s8_u8(z.val[1]),b8q);
- int8x16_t x0=vld1q_s8(x+i), x1=vld1q_s8(x+i+16);
- int16x8_t p=vmull_s8(vget_low_s8(w0),vget_low_s8(x0));
- p=vmlal_s8(p,vget_high_s8(w0),vget_high_s8(x0));
- acc=vpadalq_s16(acc,p);
- p=vmull_s8(vget_low_s8(w1),vget_low_s8(x1));
- p=vmlal_s8(p,vget_high_s8(w1),vget_high_s8(x1));
- acc=vpadalq_s16(acc,p);
- }
- sum=vaddvq_s32(acc);
-#endif
-#elif defined(__VSX__)
- const __vector unsigned char m4v=vec_splats((unsigned char)0x0F);
- const __vector unsigned char sh4=vec_splats((unsigned char)4);
- const __vector signed char b8v=vec_splats((signed char)8);
- const __vector signed char vz=vec_splats((signed char)0);
- __vector signed int acc=vec_splats(0);
- for(;i+32<=I;i+=32){
- __vector unsigned char by=vec_xl(0,w4+(i>>1));
- __vector unsigned char lo=vec_and(by,m4v), hi=vec_sr(by,sh4);
- __vector signed char w0=vec_sub((__vector signed char)vec_mergeh(lo,hi),b8v);
- __vector signed char w1=vec_sub((__vector signed char)vec_mergel(lo,hi),b8v);
- __vector signed char x0=vec_xl(0,(const signed char*)(x+i));
- __vector signed char x1=vec_xl(0,(const signed char*)(x+i+16));
- __vector __bool char n0=vec_cmplt(w0,vz), n1=vec_cmplt(w1,vz);
- acc=vec_msum(vec_sel(x0,vec_sub(vz,x0),n0),
- (__vector unsigned char)vec_sel(w0,vec_sub(vz,w0),n0),acc);
- acc=vec_msum(vec_sel(x1,vec_sub(vz,x1),n1),
- (__vector unsigned char)vec_sel(w1,vec_sub(vz,w1),n1),acc);
- }
- sum=vec_extract(acc,0)+vec_extract(acc,1)+vec_extract(acc,2)+vec_extract(acc,3);
#endif
- for(;i+1>1]; sum+=((int)(b&0xF)-8)*x[i]+((int)(b>>4)-8)*x[i+1]; }
- if(i>1]; sum+=((int)(b&0xF)-8)*x[i]; }
- return sum;
-}
-/* ---- ARM i8mm SMMLA tiled kernels ---------------------------------------- */
-#if defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
-static inline int32x4_t mm_tile16(int32x4_t acc, int8x16_t wo, int8x16_t wo1,
- int8x16_t xs, int8x16_t xs1){
- acc=vmmlaq_s32(acc, vcombine_s8(vget_low_s8(wo), vget_low_s8(wo1)),
- vcombine_s8(vget_low_s8(xs), vget_low_s8(xs1)));
- return vmmlaq_s32(acc, vcombine_s8(vget_high_s8(wo), vget_high_s8(wo1)),
- vcombine_s8(vget_high_s8(xs), vget_high_s8(xs1)));
-}
-static void matmul_q_idot_mm(float *y, const int8_t *xq, const float *sx, const int8_t *q,
- const float *scale, int S, int I, int O){
- #pragma omp parallel for schedule(static)
- for(int o=0;o<(O&~1);o+=2){
- const int8_t *wo=q+(int64_t)o*I, *wo1=q+(int64_t)(o+1)*I;
- float sc0=scale[o], sc1=scale[o+1];
- for(int s=0;s<(S&~1);s+=2){
- const int8_t *xs=xq+(int64_t)s*I, *xs1=xq+(int64_t)(s+1)*I;
- int32x4_t a0=vdupq_n_s32(0),a1=vdupq_n_s32(0),a2=vdupq_n_s32(0),a3=vdupq_n_s32(0); int i=0;
- for(;i+64<=I;i+=64){
- a0=mm_tile16(a0,vld1q_s8(wo+i), vld1q_s8(wo1+i), vld1q_s8(xs+i), vld1q_s8(xs1+i));
- a1=mm_tile16(a1,vld1q_s8(wo+i+16),vld1q_s8(wo1+i+16),vld1q_s8(xs+i+16),vld1q_s8(xs1+i+16));
- a2=mm_tile16(a2,vld1q_s8(wo+i+32),vld1q_s8(wo1+i+32),vld1q_s8(xs+i+32),vld1q_s8(xs1+i+32));
- a3=mm_tile16(a3,vld1q_s8(wo+i+48),vld1q_s8(wo1+i+48),vld1q_s8(xs+i+48),vld1q_s8(xs1+i+48));
- }
- for(;i+16<=I;i+=16)
- a0=mm_tile16(a0,vld1q_s8(wo+i),vld1q_s8(wo1+i),vld1q_s8(xs+i),vld1q_s8(xs1+i));
- int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
- int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
- int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
- for(;i>1)), byo1=vld1q_u8(wo1+(i>>1));
- uint8x16_t cyo=vld1q_u8(wo+(i>>1)+16), cyo1=vld1q_u8(wo1+(i>>1)+16);
- uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
- uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
- uint8x16x2_t ko =vzipq_u8(vandq_u8(cyo, m4q), vshrq_n_u8(cyo, 4));
- uint8x16x2_t ko1=vzipq_u8(vandq_u8(cyo1,m4q), vshrq_n_u8(cyo1,4));
- a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
- vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
- vld1q_s8(xs+i), vld1q_s8(xs1+i));
- a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
- vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
- vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
- a2=mm_tile16(a2, vsubq_s8(vreinterpretq_s8_u8(ko.val[0]),b8q),
- vsubq_s8(vreinterpretq_s8_u8(ko1.val[0]),b8q),
- vld1q_s8(xs+i+32), vld1q_s8(xs1+i+32));
- a3=mm_tile16(a3, vsubq_s8(vreinterpretq_s8_u8(ko.val[1]),b8q),
- vsubq_s8(vreinterpretq_s8_u8(ko1.val[1]),b8q),
- vld1q_s8(xs+i+48), vld1q_s8(xs1+i+48));
- }
- for(;i+32<=I;i+=32){
- uint8x16_t byo=vld1q_u8(wo+(i>>1)), byo1=vld1q_u8(wo1+(i>>1));
- uint8x16x2_t zo =vzipq_u8(vandq_u8(byo, m4q), vshrq_n_u8(byo, 4));
- uint8x16x2_t zo1=vzipq_u8(vandq_u8(byo1,m4q), vshrq_n_u8(byo1,4));
- a0=mm_tile16(a0, vsubq_s8(vreinterpretq_s8_u8(zo.val[0]),b8q),
- vsubq_s8(vreinterpretq_s8_u8(zo1.val[0]),b8q),
- vld1q_s8(xs+i), vld1q_s8(xs1+i));
- a1=mm_tile16(a1, vsubq_s8(vreinterpretq_s8_u8(zo.val[1]),b8q),
- vsubq_s8(vreinterpretq_s8_u8(zo1.val[1]),b8q),
- vld1q_s8(xs+i+16), vld1q_s8(xs1+i+16));
- }
- int32x4_t acc=vaddq_s32(vaddq_s32(a0,a1),vaddq_s32(a2,a3));
- int32_t d00=vgetq_lane_s32(acc,0), d01=vgetq_lane_s32(acc,1);
- int32_t d10=vgetq_lane_s32(acc,2), d11=vgetq_lane_s32(acc,3);
- for(;i+1>1], bo1=wo1[i>>1];
- int a0=(int)(bo&0xF)-8, a1=(int)(bo>>4)-8, b0=(int)(bo1&0xF)-8, b1=(int)(bo1>>4)-8;
- int u0=xs[i],u1=xs[i+1],v0=xs1[i],v1=xs1[i+1];
- d00+=a0*u0+a1*u1; d01+=a0*v0+a1*v1; d10+=b0*u0+b1*u1; d11+=b0*v0+b1*v1; }
- if(i>1], bo1=wo1[i>>1];
- int a0=(int)(bo&0xF)-8, b0=(int)(bo1&0xF)-8;
- d00+=a0*xs[i]; d01+=a0*xs1[i]; d10+=b0*xs[i]; d11+=b0*xs1[i]; }
- y[(int64_t)s*O+o] =(float)d00*sc0*sx[s];
- y[(int64_t)s*O+(o+1)] =(float)d10*sc1*sx[s];
- y[(int64_t)(s+1)*O+o] =(float)d01*sc0*sx[s+1];
- y[(int64_t)(s+1)*O+(o+1)]=(float)d11*sc1*sx[s+1];
- }
- if(S&1){ int s=S-1; const int8_t *xs=xq+(int64_t)s*I;
- y[(int64_t)s*O+o] =(float)dot_i4i8(wo, xs,I)*sc0*sx[s];
- y[(int64_t)s*O+(o+1)]=(float)dot_i4i8(wo1,xs,I)*sc1*sx[s]; }
- }
- if(O&1){ int o=O-1; const uint8_t *w=q4+(int64_t)o*rb; float sc=scale[o];
- #pragma omp parallel for schedule(static)
- for(int s=0;s=2){ matmul_q_idot_mm(y,xq,sx,q,scale,S,I,O); return; }
-#endif
- #pragma omp parallel for schedule(static)
- for(int o=0;o=2){ matmul_i4_idot_mm(y,xq,sx,q4,scale,S,I,O); return; }
-#endif
- #pragma omp parallel for schedule(static)
- for(int o=0;oplanar repack. */
-static void planarize_i4_row(uint8_t *row, int I){
- uint8_t tmp[32];
- int nb=I/64;
- for(int b=0;b>1]>>((src_lo&1)*4))&0xF;
- uint8_t nib_hi=(blk[src_hi>>1]>>((src_hi&1)*4))&0xF;
- tmp[k]=(uint8_t)(nib_lo|(nib_hi<<4));
- }
- memcpy(blk,tmp,32);
- }
-}
-static void planarize_i4(uint8_t *q4, int O, int I){
- int rb=(I+1)/2;
- #pragma omp parallel for schedule(static)
- for(int o=0;o>1)));
- __m256i b1=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)+32));
- a0=coli_dpbusd256(a0,_mm256_and_si256(b0,m4),
- _mm256_loadu_si256((const __m256i*)(x+i)));
- a1=coli_dpbusd256(a1,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4),
- _mm256_loadu_si256((const __m256i*)(x+i+32)));
- a2=coli_dpbusd256(a2,_mm256_and_si256(b1,m4),
- _mm256_loadu_si256((const __m256i*)(x+i+64)));
- a3=coli_dpbusd256(a3,_mm256_and_si256(_mm256_srli_epi16(b1,4),m4),
- _mm256_loadu_si256((const __m256i*)(x+i+96)));
- }
- __m256i acc=_mm256_add_epi32(_mm256_add_epi32(a0,a1),_mm256_add_epi32(a2,a3));
- for(;i+64<=I;i+=64){
- __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)));
- acc=coli_dpbusd256(acc,_mm256_and_si256(b0,m4),
- _mm256_loadu_si256((const __m256i*)(x+i)));
- acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(b0,4),m4),
- _mm256_loadu_si256((const __m256i*)(x+i+32)));
- }
- sum=hsum256_i32(acc);
-#elif defined(__AVX2__)
- const __m256i m4=_mm256_set1_epi8(0x0F);
- const __m256i ones=_mm256_set1_epi16(1);
- __m256i acc=_mm256_setzero_si256();
- for(;i+64<=I;i+=64){
- __m256i b0=_mm256_loadu_si256((const __m256i*)(w4+(i>>1)));
- /* maddubs(u8, s8): u<=15, |x|<=127 -> coppia <= 3810, int16 sicuro */
- __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(b0,m4),
- _mm256_loadu_si256((const __m256i*)(x+i)));
- __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(b0,4),m4),
- _mm256_loadu_si256((const __m256i*)(x+i+32)));
- acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p0,ones));
- acc=_mm256_add_epi32(acc,_mm256_madd_epi16(p1,ones));
- }
- sum=hsum256_i32(acc);
-#elif defined(__ARM_NEON)
- int32x4_t acc=vdupq_n_s32(0);
- for(;i+64<=I;i+=64){
- uint8x16_t b0=vld1q_u8(w4+(i>>1)), b1=vld1q_u8(w4+(i>>1)+16);
- int8x16_t lo0=vreinterpretq_s8_u8(vandq_u8(b0,vdupq_n_u8(0x0F)));
- int8x16_t lo1=vreinterpretq_s8_u8(vandq_u8(b1,vdupq_n_u8(0x0F)));
- int8x16_t hi0=vreinterpretq_s8_u8(vshrq_n_u8(b0,4));
- int8x16_t hi1=vreinterpretq_s8_u8(vshrq_n_u8(b1,4));
-#if defined(__ARM_FEATURE_DOTPROD)
- acc=vdotq_s32(acc,lo0,vld1q_s8(x+i));
- acc=vdotq_s32(acc,lo1,vld1q_s8(x+i+16));
- acc=vdotq_s32(acc,hi0,vld1q_s8(x+i+32));
- acc=vdotq_s32(acc,hi1,vld1q_s8(x+i+48));
-#else
- int8x16_t xs0=vld1q_s8(x+i), xs1=vld1q_s8(x+i+16);
- int8x16_t xs2=vld1q_s8(x+i+32), xs3=vld1q_s8(x+i+48);
- int16x8_t m;
- m=vmull_s8(vget_low_s8(lo0),vget_low_s8(xs0)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_high_s8(lo0),vget_high_s8(xs0)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_low_s8(lo1),vget_low_s8(xs1)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_high_s8(lo1),vget_high_s8(xs1)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_low_s8(hi0),vget_low_s8(xs2)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_high_s8(hi0),vget_high_s8(xs2)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_low_s8(hi1),vget_low_s8(xs3)); acc=vpadalq_s16(acc,m);
- m=vmull_s8(vget_high_s8(hi1),vget_high_s8(xs3)); acc=vpadalq_s16(acc,m);
-#endif
- }
- sum=vaddvq_s32(acc);
-#endif
- for(;i+64<=I;i+=64){ /* fallback scalare sui blocchi planari */
- const uint8_t *blk=w4+(i>>1);
- for(int k=0;k<32;k++){
- sum+=(int32_t)(blk[k]&0xF)*x[i+k];
- sum+=(int32_t)(blk[k]>>4)*x[i+k+32];
- }
- }
- for(;i>1];
- sum+=(int32_t)(byte&0xF)*x[i];
- if(i+1>4)*x[i+1];
- }
- return sum;
-}
-
-/* ---- K1b (OPT-IN, IDOT_GS=1): IDOT planare A GRUPPI (fmt=4, gs%64==0) -----
- * Con gs=64 il gruppo di scala COINCIDE col blocco-piano da 64 elementi: il
- * dot unsigned del blocco (2 dpbusd) -> int32 di gruppo, meno 8*somma(x) del
- * gruppo, per la scala f32 del gruppo. Attivazioni int8 (stessa famiglia
- * qrow_i8 del resto dell'IDOT): NON bit-identico al kernel f32 a gruppi --
- * per questo e' dietro flag, in attesa dell'ablazione. xsg = somme int32
- * per (riga, gruppo), calcolate dal chiamante in una passata esatta.
- * EN: grouped planar IDOT, opt-in. With gs=64 the scale group IS the plane
- * block; per-group unsigned dot minus 8*group-sum, times the group scale.
- * int8 activations: not bit-identical to the f32 grouped kernel, hence the
- * flag until the ablation blesses a default. */
-static void matmul_i4p_grouped_idot(float *y, const int8_t *xq, const float *sx,
- const int32_t *xsg, const uint8_t *q4,
- const float *scale, int S, int I, int O, int gs){
- int rb=(I+1)/2, ng=(I+gs-1)/gs, bpg=gs/64; /* blocchi-piano per gruppo */
- #pragma omp parallel for schedule(static)
- for(int o=0;o>1);
- const int8_t *xb=xr+base;
-#if defined(coli_dpbusd256)
- const __m256i m4=_mm256_set1_epi8(0x0F);
- __m256i bb=_mm256_loadu_si256((const __m256i*)blk);
- __m256i acc=_mm256_setzero_si256();
- acc=coli_dpbusd256(acc,_mm256_and_si256(bb,m4),
- _mm256_loadu_si256((const __m256i*)xb));
- acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(bb,4),m4),
- _mm256_loadu_si256((const __m256i*)(xb+32)));
- d+=hsum256_i32(acc);
-#elif defined(__AVX2__)
- const __m256i m4=_mm256_set1_epi8(0x0F);
- const __m256i ones=_mm256_set1_epi16(1);
- __m256i bb=_mm256_loadu_si256((const __m256i*)blk);
- __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(bb,m4),
- _mm256_loadu_si256((const __m256i*)xb));
- __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(bb,4),m4),
- _mm256_loadu_si256((const __m256i*)(xb+32)));
- __m256i acc=_mm256_add_epi32(_mm256_madd_epi16(p0,ones),
- _mm256_madd_epi16(p1,ones));
- d+=hsum256_i32(acc);
-#else
- for(int k=0;k<32;k++){
- d+=(int32_t)(blk[k]&0xF)*xb[k];
- d+=(int32_t)(blk[k]>>4)*xb[k+32];
- }
-#endif
- }
- a=fmaf((float)(d-8*xg[g]),scl[g],a);
- }
- if(g*gs>1];
- d+=(int32_t)((i&1)?(byte>>4):(byte&0xF))*xr[i];
- }
- a=fmaf((float)(d-8*xg[g]),scl[g],a);
- }
- y[(int64_t)s*O+o]=a*sx[s];
- }
- }
-}
-
-/* matmul IDOT planare (fmt=2): y = (dot_u - 8*xsum[s]) * scale[o] * sx[s].
- * Bit-identico a matmul_i4_idot: somme intere, identita' esatta. */
-static void matmul_i4p_idot(float *y, const int8_t *xq, const float *sx, const int32_t *xsum,
- const uint8_t *q4, const float *scale, int S, int I, int O){
- int rb=(I+1)/2;
- #pragma omp parallel for schedule(static)
- for(int o=0;o
- * bit-identico al path per-riga per associativita'.
- * EN: 1x4 register tile — the weight block's load+mask cost is paid
- * once per 4 activation rows. Integer sums: bit-identical to the
- * per-row path by associativity. */
- const __m256i m4t=_mm256_set1_epi8(0x0F);
- for(;s+4<=S;s+=4){
- const int8_t *x0=xq+(int64_t)s*I, *x1=x0+I, *x2=x1+I, *x3=x2+I;
- __m256i a0=_mm256_setzero_si256(), a1=_mm256_setzero_si256();
- __m256i a2=_mm256_setzero_si256(), a3=_mm256_setzero_si256();
- int i=0;
- for(;i+64<=I;i+=64){
- __m256i b =_mm256_loadu_si256((const __m256i*)(w+(i>>1)));
- __m256i lo=_mm256_and_si256(b,m4t);
- __m256i hi=_mm256_and_si256(_mm256_srli_epi16(b,4),m4t);
- a0=coli_dpbusd256(a0,lo,_mm256_loadu_si256((const __m256i*)(x0+i)));
- a0=coli_dpbusd256(a0,hi,_mm256_loadu_si256((const __m256i*)(x0+i+32)));
- a1=coli_dpbusd256(a1,lo,_mm256_loadu_si256((const __m256i*)(x1+i)));
- a1=coli_dpbusd256(a1,hi,_mm256_loadu_si256((const __m256i*)(x1+i+32)));
- a2=coli_dpbusd256(a2,lo,_mm256_loadu_si256((const __m256i*)(x2+i)));
- a2=coli_dpbusd256(a2,hi,_mm256_loadu_si256((const __m256i*)(x2+i+32)));
- a3=coli_dpbusd256(a3,lo,_mm256_loadu_si256((const __m256i*)(x3+i)));
- a3=coli_dpbusd256(a3,hi,_mm256_loadu_si256((const __m256i*)(x3+i+32)));
- }
- int32_t d0=hsum256_i32(a0), d1=hsum256_i32(a1);
- int32_t d2=hsum256_i32(a2), d3=hsum256_i32(a3);
- /* coda a coppie, unsigned: stessa identita' -8*xsum del per-riga */
- for(;i>1];
- d0+=(int32_t)(byte&0xF)*x0[i]; d1+=(int32_t)(byte&0xF)*x1[i];
- d2+=(int32_t)(byte&0xF)*x2[i]; d3+=(int32_t)(byte&0xF)*x3[i];
- if(i+1>4)*x0[i+1]; d1+=(int32_t)(byte>>4)*x1[i+1];
- d2+=(int32_t)(byte>>4)*x2[i+1]; d3+=(int32_t)(byte>>4)*x3[i+1];
- }
- }
- y[(int64_t)(s+0)*O+o]=(float)(d0-8*xsum[s+0])*sc*sx[s+0];
- y[(int64_t)(s+1)*O+o]=(float)(d1-8*xsum[s+1])*sc*sx[s+1];
- y[(int64_t)(s+2)*O+o]=(float)(d2-8*xsum[s+2])*sc*sx[s+2];
- y[(int64_t)(s+3)*O+o]=(float)(d3-8*xsum[s+3])*sc*sx[s+3];
- }
-#endif
- for(;s bit-identico. */
diff --git a/c/qwen36.c b/c/qwen36.c
index 1b74becfe..e7515014e 100644
--- a/c/qwen36.c
+++ b/c/qwen36.c
@@ -67,6 +67,7 @@ static int qwen36_max_ctx(void) {
#include "json.h" /* tokenizer.json parsing (reuse minimal parser) */
#include "qwen36_tier.h" /* optional CUDA VRAM expert tier */
#include "expert_ffn.h" /* routed experts: planar int4 kernel + layer runner */
+#include "idot.h" /* integer dot kernels for the dense trunk (COLI_DENSE_IDOT, COLI_DENSE_BITS) */
#ifdef COLI_SEGMENT_ADAPTER
#include "segment_runtime.h"
#include "segment_adapters.h"
@@ -188,7 +189,15 @@ static void build_byte_sym(void){
g_unmap[cp]=(short)b; /* reverse: mapped codepoint -> original byte */
}
}
-static void push_id(int **ids,int *n,int *cap,int v){ if(*n==*cap){*cap*=2; *ids=realloc(*ids,*cap*sizeof(int));} (*ids)[(*n)++]=v; }
+static void push_id(int **ids,int *n,int *cap,int v){
+ if(*n==*cap){
+ *cap*=2;
+ int *tmp=realloc(*ids,*cap*sizeof(int));
+ if(!tmp){ fprintf(stderr,"qwen36: OOM reallocating token id buffer (%d entries)\n",*cap); exit(1); }
+ *ids=tmp;
+ }
+ (*ids)[(*n)++]=v;
+}
static int try_special(const char *s,int i,int n,int *id_out){
int best_len=0,best_id=-1;
@@ -227,7 +236,22 @@ static int pretok_end(const char *s,int i,int n){
if(s[i]==' '&&i+10) return nl_end;
+ /* rule6: \s+(?!\S) -- a run followed by a non-space keeps its last char
+ * for the next piece, which then takes it as " x" or " ." (HF: " <" is
+ * " " then " <", not " " then "<"). A single char cannot back off and
+ * falls to rule7, \s+, which takes it whole. */
+ if(ki) return last;
+ return k;
+ }
return i+adv;
}
static void bpe_piece(const char *piece,int len,int **ids,int *n,int *cap){
@@ -236,7 +260,13 @@ static void bpe_piece(const char *piece,int len,int **ids,int *n,int *cap){
for(int b=0;b1){
int best=-1,besti=-1;
@@ -256,22 +286,39 @@ static void bpe_piece(const char *piece,int len,int **ids,int *n,int *cap){
for(int k=0;k`
+ * together with the `.` before it, so the special was never at a piece start and
+ * got encoded as text (#1653: `X.<|im_end|>` was 7 tokens instead of 3, and
+ * every chat turn ending in punctuation paid +4). qwen38.c already splits this
+ * way; this is the same shape. */
+static int next_special(const char *s,int i,int n){
+ for(int k=i;k0) return k; }
+ return n;
+}
static void encode_text(const char *text,int **out_ids,int *out_n){
int cap=1024,n=0; int *ids=malloc(cap*sizeof(int));
int tlen=(int)strlen(text); int i=0;
while(i0){ push_id(&ids,&n,&cap,sid); i+=L; continue; }
- int j=pretok_end(text,i,tlen); if(j<=i) j=i+utf8_adv(text,i,tlen);
- if(j>tlen) j=tlen;
- bpe_piece(text+i,j-i,&ids,&n,&cap);
- i=j;
+ /* Ordinary text runs to the next added token, and the pre-tokenizer
+ * sees that boundary as the end of its input, exactly as HF's does. */
+ int end=next_special(text,i+1,tlen);
+ while(iend) j=end;
+ bpe_piece(text+i,j-i,&ids,&n,&cap);
+ i=j;
+ }
}
*out_ids=ids; *out_n=n;
}
-/* Load Qwen tokenizer.json and build an id->piece table. Only needs the
- * "model.vocab" map (piece string -> id); merges are irrelevant for decoding. */
+/* Load Qwen tokenizer.json and build an id->piece table from model.vocab
+ * and added_tokens. Merges are irrelevant for decoding. */
static void load_tokenizer(const char *path){
FILE *f = fopen(path, "rb");
if (!f) { fprintf(stderr, "[tok] cannot open %s\n", path); return; }
@@ -285,18 +332,43 @@ static void load_tokenizer(const char *path){
jval *vocab = json_get(model, "vocab");
if (!vocab) vocab = json_get(model, "tokens");
if (!vocab) { fprintf(stderr, "[tok] no model.vocab/tokens in %s\n", path); free(buf); return; }
+ jval *adds = json_get(root, "added_tokens");
+
int mx = 0;
if (vocab->t == J_OBJ){
for (int i=0;ilen;i++){ int id=(int)vocab->kids[i]->num; if(id>mx)mx=id; }
} else {
mx = vocab->len - 1;
}
+ if (adds && adds->t==J_ARR){
+ for (int k=0;klen;k++){
+ jval *t = adds->kids[k];
+ int id = (int)jnum(t,"id");
+ if (id > mx) mx = id;
+ }
+ }
+
g_tok = calloc((size_t)mx+1, sizeof(char*));
if (vocab->t == J_OBJ){
for (int i=0;ilen;i++){ int id=(int)vocab->kids[i]->num; if(id>=0 && id<=mx) g_tok[id]=strdup(vocab->keys[i]); }
} else {
for (int i=0;ilen;i++){ if(vocab->kids[i] && vocab->kids[i]->t==J_STR) g_tok[i]=strdup(vocab->kids[i]->str); }
}
+ if (adds && adds->t==J_ARR){
+ for (int k=0;klen;k++){
+ jval *t = adds->kids[k];
+ const char *c = jstr(t,"content");
+ int id = (int)jnum(t,"id");
+ /* Only the non-special ones: , , ,
+ * are text the gateway parses. Special tokens
+ * (<|im_start|>, <|endoftext|>, ...) keep decoding to nothing,
+ * as reference decoding does with skip_special_tokens. */
+ jval *sp = json_get(t,"special");
+ if (sp && sp->t==J_BOOL && sp->boolean) continue;
+ if (c && id>=0 && id<=mx && !g_tok[id])
+ g_tok[id]=strdup(c);
+ }
+ }
g_tok_n = mx+1;
/* ---- encoder tables (text -> ids) ---- */
@@ -333,7 +405,6 @@ static void load_tokenizer(const char *path){
smap_put(&g_merge, key, r);
}
}
- jval *adds = json_get(root, "added_tokens");
if (adds && adds->t==J_ARR && g_nspecial==0){
g_nspecial = adds->len;
g_sp_str = malloc(g_nspecial*sizeof(char*));
@@ -439,7 +510,7 @@ static void sse_chunk(const char *json){
* <0xXX> byte-fallback tokens emit the raw byte directly. */
static void decode_id_to_bytes(int id, unsigned char *out, int *outn){
*outn = 0;
- if (!g_tok || id<0 || id>=g_tok_n) return;
+ if (!g_tok || id<0 || id>=g_tok_n || !g_tok[id]) return;
const unsigned char *pc = (const unsigned char*)g_tok[id];
/* byte-fallback token: <0xXX> -> raw byte */
if (pc[0]=='<' && pc[1]=='0' && pc[2]=='x' && pc[5]=='>'){
@@ -613,18 +684,47 @@ typedef struct {
/* Gated DeltaNet (linear_attention) dims, read from qwen36_meta.json. */
int dn_vheads, dn_kheads, dn_kdim, dn_vdim, dn_convk, dn_conv_dim;
int expert_gs; /* expert scale group size along input dim; 0 = per-row */
+ /* Mixed expert layout (convert_qwen36.py --down-bits): gate/up stay int4
+ * (ebits, expert_gs), down_proj is int8 with its own group size. One slab
+ * per expert, [gate int4 packed | up int4 packed | down int8], 2*inter*hidden
+ * bytes -- told apart from int4 (1.5x) and int8 (3x) by size, like today. */
+ int expert_down_bits, expert_down_gs;
} Cfg;
+/* ---- Dense int8: a dense matrix that is quantized to int8 during load
+ * (load_tq, below matmul_d) instead of loaded as f32 and quantized in a
+ * separate pass afterward -- so the f32 staging buffer for THIS matrix alone
+ * is what's briefly resident, not every dense matrix in the model at once.
+ * `w` is the f32 copy: kept (and `q`/`sc` left NULL) when COLI_DENSE_I8=0,
+ * the reference/parity path; freed once `q`/`sc` are populated otherwise
+ * (COLI_KEEP_F32=1 keeps it alongside them, for debugging). matmul_d
+ * dispatches on q!=NULL directly -- no pointer-keyed scan. */
+/* q/sc: int8 rows with one scale per row (the classic copy, what the VRAM
+ * tier uploads). q4/sg: the same matrix as int4 planar blocks of 64 with one
+ * scale per group (COLI_DENSE_BITS=4), the layout the K1b grouped kernel
+ * reads; ng = I/64 groups per row. */
+typedef struct { const float *w; int8_t *q; float *sc; int I, O; uint8_t *q4; float *sg; int ng; } QW;
+static void qw_free(QW *w) {
+ free((void*)w->w); free(w->q); free(w->sc); free(w->q4); free(w->sg);
+ w->w = NULL; w->q = NULL; w->sc = NULL; w->q4 = NULL; w->sg = NULL; w->ng = 0;
+}
+
/* ---------- per-layer dense weights ---------- */
typedef struct {
- float *in_ln, *post_ln, *q, *k, *v, *o, *qn, *kn, *gate, *gate_bias;
- float *sh_g, *sh_u, *sh_d, *sh_gate; /* shared expert (dense f32) + shared_expert_gate */
+ float *in_ln, *post_ln, *qn, *kn, *gate_bias;
+ QW q, k, v, o, gate;
+ QW sh_g, sh_u, sh_d; float *sh_gate; /* shared expert (dense, int8-during-load) + shared_expert_gate */
/* Gated DeltaNet (linear_attention) dense weights (f16->f32 via st_read_f32). */
- float *dn_qkv, *dn_z, *dn_b, *dn_a; /* in_proj_qkv/z/b/a */
+ QW dn_qkv, dn_z; float *dn_b, *dn_a; /* in_proj_qkv/z (int8-during-load), b/a (not dense-matmul'd) */
float *dn_conv; /* conv1d.weight [conv_dim, convk] (groups=conv_dim) */
float *dn_dtbias, *dn_alog; /* dt_bias[vh], A_log[vh] */
float *dn_norm; /* RMSNormGated weight [vdim] */
- float *dn_out; /* out_proj [hidden, value_dim] */
+ QW dn_out; /* out_proj [hidden, value_dim] */
+ /* VRAM copies the tier placed (qt_dense handle + 1, 0 = stays on the CPU):
+ * the DeltaNet out_proj, the attention q/k/v/o and the shared expert's
+ * three matrices. Offered per layer as "dnout", "attnproj", "shexp";
+ * see trunk_offer_dense / trunk_place_dense. */
+ int qth_dnout, qth_q, qth_k, qth_v, qth_o, qth_shg, qth_shu, qth_shd;
} Layer;
/* ---------- LRU expert cache (int8 weights + per-row float scales) ---------- */
@@ -638,17 +738,27 @@ typedef struct {
int n, cap;
} LCache;
+/* CACHE_ROUTE telemetry (docs/CACHE_ROUTE.md): the lever changes which experts
+ * run, so it carries its own meters. Only touched when the lever is on. */
+typedef struct {
+ uint64_t slots, swaps, swaps_vram; /* chosen slots; not in the true top-K; of those, VRAM-resident */
+ uint64_t agree_hit, agree_tot; /* |chosen ∩ true top-K| summed, K summed */
+ double kl_sum; uint64_t kl_n; /* mean KL(true top-K mass || chosen mass) */
+} RouteStats;
+
typedef struct {
Cfg c;
shards S;
int quant_bits;
- float *embed, *lm_head, *final_norm;
+ float *embed, *final_norm;
+ QW lm_head;
Layer *L;
LCache *cache; /* [n_layers] */
int *active_of; /* [n_layers] original->active idx (Phase 2: identity for all layers) */
float **DN_rec; /* [n_layers] recurrent state S[h]=[kdim,vdim] for DeltaNet layers (NULL for attn) */
float **DN_conv; /* [n_layers] conv ring [conv_dim, convk-1] for DeltaNet layers (NULL for attn) */
uint64_t clock, hits, miss;
+ RouteStats route; /* CACHE_ROUTE / ROUTE_AGREE meters */
/* Telemetria per la dashboard (Brain/Profile): tempo di lettura esperti
* accumulato dall'avvio, e bitmap degli esperti toccati nel turno. */
double t_disk;
@@ -680,6 +790,13 @@ static volatile unsigned pilot_r = 0, pilot_w = 0;
static Model *pilot_m = NULL;
static int g_pilot = 0;
static int g_wide = 1;
+/* CACHE_ROUTE family, same names and defaults as the GLM engine (docs/CACHE_ROUTE.md). */
+static int g_cache_route = 0;
+static int g_route_j = 2;
+static int g_route_m = 12;
+static float g_route_p = 0.f;
+static float g_route_alpha = 1.f;
+static int g_route_agree = 0;
static void pilot_prefetch(Model *m, int lnext, const float *x, int S);
static void *pilot_worker(void *arg);
@@ -810,85 +927,11 @@ static void matmul(float *y, const float *x, const float *W, int S, int I, int O
}
}
-/* y[1,O] = x[1,I] @ W^T with W quantized: q[O,I] int8 + scale per row. */
-#if defined(__ARM_NEON)
-#include
-static inline int32_t dot_i8_16(const int8_t *a, const int8_t *b) {
- int32x4_t acc = vdupq_n_s32(0);
- int8x16_t va = vld1q_s8(a), vb = vld1q_s8(b);
-#if defined(__ARM_FEATURE_DOTPROD)
- acc = vdotq_s32(acc, va, vb);
-#else
- acc = vpadalq_s16(acc, vmull_s8(vget_low_s8(va), vget_low_s8(vb)));
- acc = vpadalq_s16(acc, vmull_s8(vget_high_s8(va), vget_high_s8(vb)));
-#endif
- return vaddvq_s32(acc);
-}
-#endif
-static void matmul_q(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) {
-#if defined(__ARM_NEON)
- /* IDOT is opt-in, not default-on: this path quantizes the ACTIVATIONS to
- * Q8_0 per 16-element block, which the scalar path does not, so the two are
- * not numerically equivalent. olmoe shipped it default-on and it cost
- * token-exactness end to end (#1044, fixed in af48fe8 by making it opt-in);
- * qwen36 inherited the same default from the same family of kernels. The
- * tiny-oracle gate would not have caught it -- that job runs on x86. */
- static int idot = -1;
- if (idot < 0) { const char *e = getenv("IDOT"); idot = (e && atoi(e)); }
- if (idot && I % 16 == 0 && I <= 4096) {
- int nb = I / 16; int8_t xi[4096]; float xs[256];
- for (int b = 0; b < nb; b++) {
- const float *xb = x + b*16;
- float am = 0.f; for (int i = 0; i < 16; i++) { float a = fabsf(xb[i]); if (a > am) am = a; }
- float s = am/127.f; if (s < 1e-12f) s = 1e-12f;
- xs[b] = s; float inv = 1.f/s;
- for (int i = 0; i < 16; i++) xi[b*16+i] = (int8_t)lrintf(xb[i]*inv);
- }
- #pragma omp parallel for schedule(static)
- for (int o = 0; o < O; o++) {
- const int8_t *w = q + (int64_t)o * I;
- float acc = 0.f;
- for (int b = 0; b < nb; b++) acc += xs[b]*(float)dot_i8_16(xi+b*16, w+b*16);
- y[o] = acc * scale[o];
- }
- return;
- }
-#endif
-#if defined(__AVX2__) && defined(__FMA__)
- /* Hand-vectorized int8->f32 GEMV (gcc does not auto-vectorize the
- * convert+accumulate chain). 32 weights per iteration, FMA accumulate. */
- #pragma omp parallel for schedule(static) if(O >= 256)
- for (int o = 0; o < O; o++) {
- const int8_t *w = q + (int64_t)o * I;
- __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps();
- __m256 a2 = _mm256_setzero_ps(), a3 = _mm256_setzero_ps();
- int i = 0;
- for (; i + 32 <= I; i += 32) {
- __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i));
- __m128i b1 = _mm_loadu_si128((const __m128i*)(w + i + 16));
- a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0);
- a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1);
- a2 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+16), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b1)), a2);
- a3 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+24), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b1,8))), a3);
- }
- a0 = _mm256_add_ps(_mm256_add_ps(a0,a1), _mm256_add_ps(a2,a3));
- __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1));
- s = _mm_add_ps(s, _mm_movehl_ps(s,s));
- s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1));
- float acc = _mm_cvtss_f32(s);
- for (; i < I; i++) acc += x[i] * (float)w[i];
- y[o] = acc * scale[o];
- }
-#else
- #pragma omp parallel for schedule(static)
- for (int o = 0; o < O; o++) {
- const int8_t *w = q + (int64_t)o * I;
- float acc = 0.f;
- for (int i = 0; i < I; i++) acc += x[i] * (float)w[i];
- y[o] = acc * scale[o];
- }
-#endif
-}
+/* y[1,O] = x[1,I] @ W^T with W quantized: q[O,I] int8 + scale per row.
+ * matmul_q lives in qgemv.h so tests/test_qgemv.c can link the exact kernel
+ * the engine runs (qwen36.c has a main() and cannot itself be linked into a
+ * test binary). */
+#include "qgemv.h"
/* Multi-row dense-int8 prefill kernel. matmul_q() above is deliberately kept
* as the S=1 decode implementation: its four AVX accumulators stay in
@@ -978,8 +1021,14 @@ static void matmul_q_batch(float *y, const float *x, const int8_t *q,
}
/* Group-scaled int8 GEMV: one f32 scale per `gs` input elements per row
- * (gs64 expert containers). Row layout of `scale`: [O][I/gs] row-major. */
+ * (gs64 expert containers). Row layout of `scale`: [O][I/gs] row-major.
+ * matmul_q_gs lives in gsgemv.h so tests/test_gsgemv.c can link the exact
+ * kernel the engine runs. */
static int g_expert_gs = 0; /* set from qwen36_meta.json (expert_gs) at load */
+#include "gsgemv.h"
+
+static int g_expert_mixed = 0; /* mixed layout on disk (int4 gate/up, int8 down) */
+static int g_expert_down_gs = 0; /* down_proj's group size in the mixed layout (0 = per row) */
/* 1 = expert container packs int4 (tier fmt=4); 0 = int8 per-row (tier fmt=1).
* Same signal main's nbytes probe and tier_warmstart receive; the decode path
* needs it to offer int8 experts (#1391): on an int8 container e->g4 is NULL. */
@@ -992,6 +1041,12 @@ static int g_expert_is_int4 = 1;
* int8 copy) and with QWEN_EXPERT_KERNEL=0, which keeps the historical
* unpack-to-int8 path for A/Bs. Decided once from the container itself. */
static int container_layer_is_int4(Model *m, int layer);
+/* The routed experts take the same integer path as the dense trunk
+ * (expert_ffn.h mode 1: activation to int8 once per row, dpbusd against the
+ * planar nibbles). Measured on the 35B: expert compute 22.7 to 15.9
+ * ms/token for +0.1% perplexity. QWEN_EXPERT_ACT=f32 restores mode 0, f32
+ * activations and the bit-identical contract with the pair kernels. */
+static int xf_act_mode(void){ static int v=-1; if(v<0){ const char *e=getenv("QWEN_EXPERT_ACT"); v=(e&&!strcmp(e,"f32"))?0:1; } return v; }
static int xf_mode(Model *m) {
static int v = -1;
if (v >= 0) return v;
@@ -1030,61 +1085,93 @@ static void tier_offer_slot(int layer, int eid, const Slot *s) {
qt_note(layer, eid, (const uint8_t *)s->g, (const uint8_t *)s->u,
(const uint8_t *)s->d, s->gs, s->us, s->ds);
}
-static void matmul_q_gs(float *y, const float *x, const int8_t *q, const float *scale,
- int I, int O, int gs) {
- int ng = (I + gs - 1) / gs;
-#if defined(__AVX2__) && defined(__FMA__)
- if ((gs & 31) == 0) {
- #pragma omp parallel for schedule(static) if(O >= 256)
- for (int o = 0; o < O; o++) {
- const int8_t *w = q + (int64_t)o * I;
- const float *sc = scale + (int64_t)o * ng;
- float acc = 0.f;
- for (int gi = 0; gi < ng; gi++) {
- __m256 a0 = _mm256_setzero_ps(), a1 = _mm256_setzero_ps();
- int base = gi * gs, end = base + gs; if (end > I) end = I;
- for (int i = base; i + 16 <= end; i += 16) {
- __m128i b0 = _mm_loadu_si128((const __m128i*)(w + i));
- a0 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(b0)), a0);
- a1 = _mm256_fmadd_ps(_mm256_loadu_ps(x+i+8), _mm256_cvtepi32_ps(_mm256_cvtepi8_epi32(_mm_srli_si128(b0,8))), a1);
- }
- a0 = _mm256_add_ps(a0, a1);
- __m128 s = _mm_add_ps(_mm256_castps256_ps128(a0), _mm256_extractf128_ps(a0,1));
- s = _mm_add_ps(s, _mm_movehl_ps(s,s));
- s = _mm_add_ss(s, _mm_shuffle_ps(s,s,1));
- acc += _mm_cvtss_f32(s) * sc[gi];
- }
- y[o] = acc;
- }
- return;
- }
-#endif
- #pragma omp parallel for schedule(static) if(O >= 256)
- for (int o = 0; o < O; o++) {
- const int8_t *w = q + (int64_t)o * I;
- const float *sc = scale + (int64_t)o * ng;
- float acc = 0.f;
- for (int gi = 0; gi < ng; gi++) {
- int base = gi * gs, end = base + gs; if (end > I) end = I;
- float part = 0.f;
- for (int i = base; i < end; i++) part += x[i] * (float)w[i];
- acc += part * sc[gi];
- }
- y[o] = acc;
- }
-}
/* Expert-GEMV dispatch: per-row scales (classic) or grouped (gs64 container). */
static void matmul_qe(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) {
if (g_expert_gs) matmul_q_gs(y, x, q, scale, I, O, g_expert_gs);
else matmul_q(y, x, q, scale, I, O);
}
+/* down_proj: in the mixed layout it carries its own scale layout (int8, per
+ * row or expert_down_gs), everywhere else it is matmul_qe. */
+static void matmul_qd(float *y, const float *x, const int8_t *q, const float *scale, int I, int O) {
+ if (!g_expert_mixed) { matmul_qe(y, x, q, scale, I, O); return; }
+ if (g_expert_down_gs) matmul_q_gs(y, x, q, scale, I, O, g_expert_down_gs);
+ else matmul_q(y, x, q, scale, I, O);
+}
/* ---- Dense int8: per-row quantized copies of the large f32 matrices.
- * matmul_d dispatches via pointer lookup to matmul_q; COLI_DENSE_I8=0 falls
- * back to f32 (reference path for parity tests). ~4x less memory traffic. */
-#define QDW_MAX 1024
-static struct { const float *w; int8_t *q; float *sc; int I, O; } g_qdw[QDW_MAX];
-static int g_qdw_n = 0;
+ * matmul_d dispatches directly off QW.q (no pointer-keyed scan -- see QW,
+ * above Layer); COLI_DENSE_I8=0 falls back to f32 (QW.w, reference path for
+ * parity tests). ~4x less memory traffic. */
+/* ---- Dense trunk, integer dot products ------------------------------------
+ *
+ * Every dense GEMV of a token (DeltaNet projections and out_proj, attention
+ * q/k/v/o, the shared expert, lm_head) used to multiply int8 weights by f32
+ * activations: each weight byte converted to f32 and fed to an FMA, eight
+ * weights per instruction. Measured on lm_head (248320 x 2048 int8, 508 MB)
+ * that runs at 29 GB/s on a 16-core AVX-512 host whose memory bus does 80:
+ * the kernel, not the bus, was the limit, and the dense part of a token is
+ * 1.9 GB of int8 on the 35B, three times the routed experts.
+ *
+ * The activation is now quantized to int8 once per call (one scale,
+ * amax/127, the qrow_i8 contract the expert IDOT already uses) and the dot
+ * is integer: 32 weights per instruction on AVX2 (maddubs), 64 on AVX-512
+ * VNNI, exact int32 sums scaled once per output. Not bit-identical to the
+ * f32 path (the activation is rounded): measured on the 35B, +1.0%
+ * perplexity on 4 x 512 tokens, lm_head 12.6 to 10.2 ms/token, decode
+ * 6.71 to 7.35 tok/s alone and 8.23 with the experts' int8 activations.
+ * COLI_DENSE_IDOT=0 restores the f32-activation kernel.
+ *
+ * COLI_DENSE_BITS=4 additionally stores the dense matrices as int4 in blocks
+ * of 64 with one scale per block, the K1b planar layout, halving the bytes
+ * the token reads; lm_head alone goes from 508 to 254 MB. It implies the
+ * integer dot (that layout has no f32 kernel). Same gate: measured. */
+static int dense_idot_on(void){ static int v=-1; if(v<0){ const char *e=getenv("COLI_DENSE_IDOT"); v=!(e&&*e=='0'); } return v; }
+static int dense_bits(void){ static int v=-1; if(v<0){ const char *e=getenv("COLI_DENSE_BITS"); v=(e&&atoi(e)==4)?4:8; } return v; }
+
+/* f32 rows -> int4 in blocks of 64 with one f32 scale per block, packed as the
+ * K1b planar layout (unsigned nibbles v+8, block b: lo nibbles = elements
+ * b*64..b*64+31, hi = b*64+32..b*64+63). The quantizer is the symmetric
+ * absmax/7 the expert containers use. */
+static void pack_int4_g64_planar(const float *w, uint8_t *q4, float *sg, int O, int I){
+ int rb = I / 2, ng = I / 64;
+ #pragma omp parallel for schedule(static)
+ for (int o = 0; o < O; o++) {
+ const float *wr = w + (int64_t)o * I;
+ uint8_t *row = q4 + (int64_t)o * rb;
+ float *sr = sg + (int64_t)o * ng;
+ for (int g = 0; g < ng; g++) {
+ const float *blk = wr + g * 64;
+ float amax = 0.f;
+ for (int k = 0; k < 64; k++) { float a = fabsf(blk[k]); if (a > amax) amax = a; }
+ /* absmax/7 is where the search starts; the scale is then refined
+ * by least squares against the rounded codes (s = w.q / q.q) and
+ * the candidate with the smallest squared error wins. Three
+ * rounds: measured on the 35B this recovers a third of the
+ * perplexity absmax alone loses at 4 bits. */
+ float best_s = amax / 7.f; if (best_s < 1e-8f) best_s = 1e-8f;
+ int best_q[64]; double best_err = 1e30;
+ float s = best_s;
+ for (int round = 0; round < 4; round++) {
+ int q[64]; double err = 0, wq = 0, qq = 0;
+ float inv = 1.f / s;
+ for (int k = 0; k < 64; k++) {
+ int v = (int)lrintf(blk[k] * inv); if (v > 7) v = 7; if (v < -8) v = -8;
+ q[k] = v; double d = (double)blk[k] - (double)v * s; err += d * d;
+ wq += (double)blk[k] * v; qq += (double)v * v;
+ }
+ if (err < best_err) { best_err = err; best_s = s; memcpy(best_q, q, sizeof q); }
+ if (qq <= 0) break;
+ float ns = (float)(wq / qq); /* least-squares scale for these codes */
+ if (ns <= 0.f || ns == s) break;
+ s = ns;
+ }
+ sr[g] = best_s;
+ uint8_t *dst = row + g * 32;
+ for (int k = 0; k < 32; k++)
+ dst[k] = (uint8_t)((best_q[k] + 8) | ((best_q[k + 32] + 8) << 4));
+ }
+ }
+}
#ifdef COLI_QWEN_BATCH_TEST
static uint64_t g_qwen_matmul_d_calls;
#endif
@@ -1094,10 +1181,36 @@ static int dense_i8_on(void){ static int v=-1; if(v<0){ const char *e=getenv("CO
* the time. */
static int kv_prefix_off(void){ const char *e=getenv("COLI_KV_PREFIX"); return e && *e=='0'; }
static int dense_batch_on(void){ const char *e=getenv("QWEN_DENSE_BATCH"); return !(e&&*e=='0'); }
-static void qdw_register(const float *W, int I, int O){
- if (!W || !dense_i8_on() || g_qdw_n >= QDW_MAX) return;
+/* COLI_DENSE_INT4 names which components take the int4 copy when
+ * COLI_DENSE_BITS=4: a comma list of lmhead, dnproj, dnout, attn, shexp,
+ * router; unset means all of them. The parts of the trunk pay 4 bits
+ * differently (measured: the attention projections and the DeltaNet input
+ * projections cost the most perplexity, lm_head the least per byte saved),
+ * so the default is chosen per component from the numbers, not for the
+ * whole trunk at once. */
+static int dense_int4_wanted(const char *tag){
+ if (dense_bits() != 4 || !tag) return 0;
+ const char *e = getenv("COLI_DENSE_INT4");
+ if (!e || !*e) return 1;
+ size_t n = strlen(tag);
+ for (const char *p = e; *p; ) {
+ while (*p == ',' || *p == ' ') p++;
+ const char *q = p; while (*q && *q != ',' && *q != ' ') q++;
+ if ((size_t)(q - p) == n && !strncmp(p, tag, n)) return 1;
+ p = q;
+ }
+ return 0;
+}
+/* Per-row max-abs / round-clamp int8 quantization of an in-memory f32 matrix
+ * W [O][I] row-major -- same math load_tq runs during streamed load, factored
+ * out so it can also run on a caller-owned buffer directly (tests that
+ * synthesize weights in memory, without a shard file to load from). Does not
+ * touch out->w -- the caller sets that (or leaves it, e.g. load_tq frees it
+ * right after). `tag` selects the int4 planar copy per dense_int4_wanted
+ * (NULL/COLI_DENSE_BITS!=4: skipped, out->q4 stays NULL). */
+static void qw_quantize(const float *W, int I, int O, const char *tag, QW *out) {
int8_t *q = malloc((size_t)O*I); float *sc = malloc((size_t)O*sizeof(float));
- if (!q || !sc) { free(q); free(sc); return; }
+ if (!q || !sc) { fprintf(stderr, "OOM qw_quantize\n"); exit(1); }
#pragma omp parallel for schedule(static)
for (int o = 0; o < O; o++) {
const float *r = W + (int64_t)o*I; float am = 0.f;
@@ -1106,20 +1219,65 @@ static void qdw_register(const float *W, int I, int O){
int8_t *d = q + (int64_t)o*I;
for (int i = 0; i < I; i++) { int v = (int)lrintf(r[i]*inv); if (v>127) v=127; if (v<-127) v=-127; d[i] = (int8_t)v; }
}
- g_qdw[g_qdw_n].w=W; g_qdw[g_qdw_n].q=q; g_qdw[g_qdw_n].sc=sc; g_qdw[g_qdw_n].I=I; g_qdw[g_qdw_n].O=O; g_qdw_n++;
+ out->q = q; out->sc = sc; out->I = I; out->O = O;
+ out->q4 = NULL; out->sg = NULL; out->ng = 0;
+ if (dense_int4_wanted(tag) && I % 64 == 0) {
+ uint8_t *q4 = malloc((size_t)O*(I/2)); float *sg = malloc((size_t)O*(I/64)*sizeof(float));
+ if (q4 && sg) {
+ pack_int4_g64_planar(W, q4, sg, O, I);
+ out->q4 = q4; out->sg = sg; out->ng = I/64;
+ } else { free(q4); free(sg); }
+ }
}
-static void matmul_d(float *y, const float *x, const float *W, int S, int I, int O){
+static void matmul_d(float *y, const float *x, const QW *w, int S, int I, int O){
#ifdef COLI_QWEN_BATCH_TEST
g_qwen_matmul_d_calls++;
#endif
- for (int i = 0; i < g_qdw_n; i++) if (g_qdw[i].w == W && g_qdw[i].I == I) {
+ if (w->q) {
+ if (w->q4 || dense_idot_on()) {
+ /* integer dot: the activation rows to int8 once, then the K1b
+ * grouped kernel (int4 planar) or the per-row int8 kernel */
+ int ng = I / 64;
+ int8_t *xq = malloc((size_t)S * I);
+ float *sx = malloc((size_t)S * sizeof(float));
+ int32_t *xsg = w->q4 ? malloc((size_t)S * ng * sizeof(int32_t)) : NULL;
+ if (xq && sx && (!w->q4 || xsg)) {
+ for (int s = 0; s < S; s++)
+ sx[s] = dense_act_i8(x + (int64_t)s * I, I, xq + (int64_t)s * I, xsg ? xsg + (int64_t)s * ng : NULL);
+ if (w->q4) matmul_i4p_grouped_idot(y, xq, sx, xsg, w->q4, w->sg, S, I, O, 64);
+ else matmul_q_idot(y, xq, sx, w->q, w->sc, S, I, O);
+ free(xq); free(sx); free(xsg);
+ return;
+ }
+ free(xq); free(sx); free(xsg); /* out of memory: the f32 path below */
+ }
if (S > 1 && dense_batch_on())
- matmul_q_batch(y, x, g_qdw[i].q, g_qdw[i].sc, S, I, O);
+ matmul_q_batch(y, x, w->q, w->sc, S, I, O);
else
- for (int s = 0; s < S; s++) matmul_q(y+(int64_t)s*O, x+(int64_t)s*I, g_qdw[i].q, g_qdw[i].sc, I, O);
+ for (int s = 0; s < S; s++) matmul_q(y+(int64_t)s*O, x+(int64_t)s*I, w->q, w->sc, I, O);
return;
}
- matmul(y, x, W, S, I, O);
+ matmul(y, x, w->w, S, I, O);
+}
+/* A dense matrix the tier placed in VRAM (handle+1 kept in the Layer, 0 = CPU):
+ * a device matmul, or 0 and the caller runs matmul_d as before. The
+ * tier turns a failing handle off itself, so the fallback is permanent. */
+static inline int qtd_batch(int hp1, float *y, const float *x, int S, int I, int O){
+ return hp1 > 0 && qt_dense_matmul_batch(hp1 - 1, y, x, S, I, O);
+}
+static inline int qtd(int hp1, float *y, const float *x, int I, int O){
+ return hp1 > 0 && qt_dense_matmul(hp1 - 1, y, x, I, O);
+}
+/* Bytes of w's dense-i8 copy (int8 rows + per-row scales), 0 when there is
+ * none (COLI_DENSE_I8=0): nothing to offer, the CPU path stands. */
+static size_t qdw_bytes(const QW *w){
+ return w->q ? (size_t)w->I * w->O + (size_t)w->O * sizeof(float) : 0;
+}
+/* Upload w's dense-i8 copy to `dev`; handle+1, or 0 when it stays on the CPU. */
+static int qdw_place(const QW *w, int dev){
+ if (dev == QT_PLACE_CPU || !w->q) return 0;
+ int h = qt_dense_init(w->q, w->sc, w->I, w->O, dev);
+ return h >= 0 ? h + 1 : 0;
}
/* rmsnorm over a row of length D (in-place capable: out may == x).
@@ -1266,6 +1424,7 @@ static void load_meta(Cfg *c, const char *snap) {
G("q_head_dim", q_head_dim); G("k_head_dim", k_head_dim); G("v_head_dim", v_head_dim);
G("o_in", o_in); G("rope_dim", rope_dim); G("qk_rope_head_dim", rope_dim);
G("expert_gs", expert_gs);
+ G("expert_down_bits", expert_down_bits); G("expert_down_gs", expert_down_gs);
G("num_experts", n_experts); G("topk", topk);
G("moe_inter", inter); G("shared_inter", shared_inter);
G("n_group", n_group); G("topk_group", topk_group);
@@ -1337,6 +1496,27 @@ static float *load_t_n(Model *m, const char *name, int64_t want) {
return p;
}
+/* Dense matrix load, quantized to int8 (+ int4 planar per `tag`, see
+ * dense_int4_wanted) DURING loading rather than in a separate pass over the
+ * whole model afterward (see QW, above Layer): reads `name` (I*O elements,
+ * same size discipline as load_t_n), and when `quantize` && COLI_DENSE_I8 is
+ * on, quantizes it via qw_quantize and frees the f32 staging buffer right
+ * away -- so at most one dense matrix's f32 copy is ever resident at a time,
+ * not the whole model's. `quantize` is false for loaders that never ran
+ * through the old post-hoc qdw_register pass either (the Segment/Edge
+ * adapters build partial or auxiliary models straight off
+ * model_init_range/load_t_n, never main()'s dense-i8 block) -- passing it
+ * through keeps their f32-only behavior exactly as it was; `tag` is unused
+ * on that path. */
+static void load_tq(Model *m, const char *name, int I, int O, int quantize, const char *tag, QW *out) {
+ float *p = load_t_n(m, name, (int64_t)I * O);
+ out->w = p; out->q = NULL; out->sc = NULL; out->I = I; out->O = O;
+ out->q4 = NULL; out->sg = NULL; out->ng = 0;
+ if (!quantize || !dense_i8_on()) return;
+ qw_quantize(p, I, O, tag, out);
+ if (getenv("COLI_KEEP_F32")) out->w = p; else { free(p); out->w = NULL; }
+}
+
static void model_init_range(Model *m, const char *snap, int cap, int bits,
int layer_begin, int layer_end,
int load_boundaries, int allocate_state) {
@@ -1362,9 +1542,17 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
exit(1);
}
double t0 = now_s();
+ /* Quantize during load only for the full-model path (main()'s static Model,
+ * load_boundaries=1): the Segment/Edge adapters build partial or auxiliary
+ * models straight off this same loop and never ran the old post-hoc
+ * qdw_register pass either, so gating on load_boundaries keeps their
+ * f32-only numerics exactly as they were. */
+ int quantize_dense = load_boundaries && dense_i8_on();
+ int qcount = 0; double qfreed = 0;
if (load_boundaries) {
- m->embed = load_t_n(m, "model.embed_tokens.weight", (int64_t)c->vocab * c->hidden);
- m->lm_head = load_t_n(m, "lm_head.weight", (int64_t)c->vocab * c->hidden);
+ m->embed = load_t_n(m, "model.embed_tokens.weight", (int64_t)c->vocab * c->hidden);
+ load_tq(m, "lm_head.weight", c->hidden, c->vocab, quantize_dense, "lmhead", &m->lm_head);
+ if (m->lm_head.q) { qcount++; qfreed += (double)c->hidden * c->vocab * sizeof(float); }
m->final_norm = load_t_n(m, "model.norm.weight", c->hidden);
}
m->L = calloc((size_t)c->n_layers, sizeof(Layer));
@@ -1374,15 +1562,19 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
m->active_of = malloc((size_t)c->n_layers * sizeof(int));
for (int i = 0; i < c->n_layers; i++) m->active_of[i] = i;
char nm[256];
+ int q_out = c->q_heads * c->q_head_dim, kv_out = c->kv_heads * c->k_head_dim;
+ #define QCOUNT(field) do { if ((field).q) { qcount++; qfreed += (double)(field).I * (field).O * sizeof(float); } } while (0)
for (int i = layer_begin; i < layer_end; i++) {
int ai = m->active_of[i]; /* == i for Phase 2 */
Layer *l = &m->L[i];
- /* input/post layernorms + MoE exist for every layer */
+ /* input/post layernorms exist for every layer */
#define LD(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d." suffix,ai); l->field = load_t_n(m,nm,(want))
LD(in_ln, "input_layernorm.weight", c->hidden);
LD(post_ln,"post_attention_layernorm.weight", c->hidden);
- LD(gate, "mlp.gate.weight", (int64_t)c->n_experts * c->hidden);
#undef LD
+ snprintf(nm,sizeof(nm),"model.layers.%d.mlp.gate.weight", ai);
+ load_tq(m, nm, c->hidden, c->n_experts, quantize_dense, "router", &l->gate);
+ QCOUNT(l->gate);
/* q/k norms are per-head [head_dim]; only on attention layers, load if present */
if (c->has_qk_norm) {
snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.q_norm.weight", ai);
@@ -1394,42 +1586,51 @@ static void model_init_range(Model *m, const char *snap, int cap, int bits,
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.gate.e_score_correction_bias", ai);
if (st_has(&m->S, nm)) { l->gate_bias = falloc(c->n_experts); st_read_f32(&m->S, nm, l->gate_bias, 0); }
else l->gate_bias = NULL;
- /* shared expert (dense f32) */
- #define LD2(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert." suffix,ai); l->field = load_t_n(m,nm,(want))
- LD2(sh_g, "gate_proj.weight", (int64_t)c->shared_inter * c->hidden);
- LD2(sh_u, "up_proj.weight", (int64_t)c->shared_inter * c->hidden);
- LD2(sh_d, "down_proj.weight", (int64_t)c->hidden * c->shared_inter);
- #undef LD2
+ /* shared expert (dense, int8-during-load) */
+ snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert.gate_proj.weight", ai);
+ load_tq(m, nm, c->hidden, c->shared_inter, quantize_dense, "shexp", &l->sh_g); QCOUNT(l->sh_g);
+ snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert.up_proj.weight", ai);
+ load_tq(m, nm, c->hidden, c->shared_inter, quantize_dense, "shexp", &l->sh_u); QCOUNT(l->sh_u);
+ snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert.down_proj.weight", ai);
+ load_tq(m, nm, c->shared_inter, c->hidden, quantize_dense, "shexp", &l->sh_d); QCOUNT(l->sh_d);
/* shared_expert_gate: Linear(hidden -> 1), sigmoid-gated shared expert */
snprintf(nm,sizeof(nm),"model.layers.%d.mlp.shared_expert_gate.weight", ai);
l->sh_gate = st_has(&m->S, nm) ? load_t_n(m, nm, c->hidden) : NULL;
if (c->is_attn[i]) {
- /* Gated Attention (full_attention) layer */
- #define LD3(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d.self_attn." suffix,ai); l->field = load_t_n(m,nm,(want))
- LD3(q, "q_proj.weight", (int64_t)c->q_heads * c->q_head_dim * c->hidden);
- LD3(k, "k_proj.weight", (int64_t)c->kv_heads * c->k_head_dim * c->hidden);
- LD3(v, "v_proj.weight", (int64_t)c->kv_heads * c->v_head_dim * c->hidden);
- LD3(o, "o_proj.weight", (int64_t)c->hidden * c->o_in);
- #undef LD3
- l->dn_qkv=l->dn_z=l->dn_b=l->dn_a=l->dn_conv=NULL;
- l->dn_dtbias=l->dn_alog=l->dn_norm=l->dn_out=NULL;
+ /* Gated Attention (full_attention) layer, dense projections int8-during-load */
+ snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.q_proj.weight", ai);
+ load_tq(m, nm, c->hidden, q_out, quantize_dense, "attn", &l->q); QCOUNT(l->q);
+ snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.k_proj.weight", ai);
+ load_tq(m, nm, c->hidden, kv_out, quantize_dense, "attn", &l->k); QCOUNT(l->k);
+ snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.v_proj.weight", ai);
+ load_tq(m, nm, c->hidden, kv_out, quantize_dense, "attn", &l->v); QCOUNT(l->v);
+ snprintf(nm,sizeof(nm),"model.layers.%d.self_attn.o_proj.weight", ai);
+ load_tq(m, nm, c->o_in, c->hidden, quantize_dense, "attn", &l->o); QCOUNT(l->o);
+ l->dn_qkv=l->dn_z=(QW){0}; l->dn_b=l->dn_a=l->dn_conv=NULL;
+ l->dn_dtbias=l->dn_alog=l->dn_norm=NULL; l->dn_out=(QW){0};
} else {
/* Gated DeltaNet (linear_attention) layer */
- l->q=l->k=l->v=l->o=NULL;
+ l->q=l->k=l->v=l->o=(QW){0};
#define LD4(field, suffix, want) snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn." suffix,ai); l->field = load_t_n(m,nm,(want))
int64_t vdim_tot = (int64_t)c->dn_vheads * c->dn_vdim;
- LD4(dn_qkv, "in_proj_qkv.weight", (int64_t)c->dn_conv_dim * c->hidden);
- LD4(dn_z, "in_proj_z.weight", vdim_tot * c->hidden);
+ snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn.in_proj_qkv.weight", ai);
+ load_tq(m, nm, c->hidden, c->dn_conv_dim, quantize_dense, "dnproj", &l->dn_qkv); QCOUNT(l->dn_qkv);
+ snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn.in_proj_z.weight", ai);
+ load_tq(m, nm, c->hidden, (int)vdim_tot, quantize_dense, "dnproj", &l->dn_z); QCOUNT(l->dn_z);
LD4(dn_b, "in_proj_b.weight", (int64_t)c->dn_vheads * c->hidden);
LD4(dn_a, "in_proj_a.weight", (int64_t)c->dn_vheads * c->hidden);
LD4(dn_conv,"conv1d.weight", (int64_t)c->dn_conv_dim * c->dn_convk);
LD4(dn_dtbias, "dt_bias", c->dn_vheads);
LD4(dn_alog,"A_log", c->dn_vheads);
LD4(dn_norm, "norm.weight", c->dn_vdim);
- LD4(dn_out, "out_proj.weight", (int64_t)c->hidden * vdim_tot);
#undef LD4
+ snprintf(nm,sizeof(nm),"model.layers.%d.linear_attn.out_proj.weight", ai);
+ load_tq(m, nm, (int)vdim_tot, c->hidden, quantize_dense, "dnout", &l->dn_out); QCOUNT(l->dn_out);
}
}
+ #undef QCOUNT
+ if (quantize_dense)
+ fprintf(stderr, "[dense-i8] %d matrices quantized during load, %.1f GB f32 freed\n", qcount, qfreed/1073741824.0);
m->cache = calloc((size_t)c->n_layers, sizeof(LCache));
for (int i = layer_begin; i < layer_end; i++) {
m->cache[i].cap = cap;
@@ -1474,7 +1675,8 @@ static void model_init(Model *m, const char *snap, int cap, int bits) {
/* scale counts per expert matrix: per-row (gs=0) or grouped along input dim */
static int64_t scale_count_gu(const Cfg *c){ return c->expert_gs ? (int64_t)c->inter * ((c->hidden + c->expert_gs - 1) / c->expert_gs) : c->inter; }
-static int64_t scale_count_d (const Cfg *c){ return c->expert_gs ? (int64_t)c->hidden * ((c->inter + c->expert_gs - 1) / c->expert_gs) : c->hidden; }
+static int down_gs_of(const Cfg *c){ return c->expert_down_bits ? c->expert_down_gs : c->expert_gs; }
+static int64_t scale_count_d (const Cfg *c){ int gs = down_gs_of(c); return gs ? (int64_t)c->hidden * ((c->inter + gs - 1) / gs) : c->hidden; }
static void slot_ensure_allocated(Model *m, Slot *s) {
if (s->g || s->pw) return;
@@ -1572,9 +1774,10 @@ static void load_expert_merged(Model *m, int layer, int eid, Slot *s) {
int64_t want_w = ng + ng + nd;
int64_t want_s = 2*scale_count_gu(cc) + scale_count_d(cc);
st_tensor *tw = st_find(&m->S, nm), *ts = st_find(&m->S, qsnm);
- if (!tw || (tw->nbytes != want_w && tw->nbytes != want_w / 2)) {
- fprintf(stderr, "%s: expert weight is %lld bytes — expected %lld (int8) or %lld (int4)\n",
- nm, (long long)(tw ? tw->nbytes : -1), (long long)want_w, (long long)(want_w / 2)); exit(1); }
+ int64_t want_mixed = ng + nd; /* gate|up packed int4 (ng bytes) + down int8 (nd bytes) */
+ if (!tw || (tw->nbytes != want_w && tw->nbytes != want_w / 2 && tw->nbytes != want_mixed)) {
+ fprintf(stderr, "%s: expert weight is %lld bytes — expected %lld (int8), %lld (int4) or %lld (int4 gate/up + int8 down)\n",
+ nm, (long long)(tw ? tw->nbytes : -1), (long long)want_w, (long long)(want_w / 2), (long long)want_mixed); exit(1); }
if (!ts || ts->numel != want_s) {
fprintf(stderr, "%s: scale array is %lld elems — expected %lld (refusing)\n",
qsnm, (long long)(ts ? ts->numel : -1), (long long)want_s); exit(1); }
@@ -1584,6 +1787,23 @@ static void load_expert_merged(Model *m, int layer, int eid, Slot *s) {
rest of the MoE path (matmul_q) is unchanged. Nibble convention (must match
c/tools/convert_qwen36.py pack_int4): LOW nibble = element 2k, HIGH nibble = 2k+1;
each nibble is signed 4-bit (sign-extend if bit3 set). */
+ if (tw->nbytes == want_mixed) {
+ /* mixed layout: unpack gate|up (2*ng int4 elements in ng bytes) into the
+ * slot's g|u block, copy down's int8 rows behind them. No packed copy is
+ * kept: the tier does not take this layout yet (main refuses it). */
+ static int noted_m = 0;
+ if (!noted_m) { fprintf(stderr, "[qwen36] mixed expert layout detected (int4 gate/up, int8 down) — unpacking gate/up to int8 in slot\n"); noted_m = 1; }
+ uint8_t *raw = (uint8_t *)malloc((size_t)want_mixed);
+ if (!raw) { fprintf(stderr, "OOM reading mixed expert %s\n", nm); exit(1); }
+ st_read_raw(&m->S, nm, raw, 1);
+ unpack_int4_to_int8(s->g, raw, ng + ng); /* 2*ng elements from ng bytes */
+ memcpy(s->d, raw + ng, (size_t)nd);
+ free(raw);
+ s->is_int4 = 0;
+ free(s->g4); free(s->u4); free(s->d4); s->g4 = s->u4 = s->d4 = NULL;
+ st_read_f32(&m->S, qsnm, s->gs, 0);
+ return;
+ }
if (tw->nbytes == want_w / 2) {
static int noted = 0;
if (!noted) { fprintf(stderr, "[qwen36] int4 packed weights detected — %s\n", s->pw ? "kept int4, repacked planar for expert_ffn.h" : "unpacking to int8 in slot"); noted = 1; }
@@ -1656,19 +1876,15 @@ static void slot_ensure_int8(Model *m, Slot *s) {
int64_t ng = (int64_t)c->inter * c->hidden, nd = (int64_t)c->hidden * c->inter;
int8_t *w = malloc((size_t)(ng + ng + nd));
if (!w) { fprintf(stderr, "OOM slot_ensure_int8\n"); exit(1); }
- const uint8_t *src4[3] = { s->g4, s->u4, s->d4 };
- int64_t lens[3] = { ng, ng, nd };
- int8_t *dst = w;
- for (int t = 0; t < 3; t++) {
- const uint8_t *p = src4[t];
- for (int64_t i = 0; i < lens[t]; i += 2) {
- uint8_t b = p[i >> 1];
- int8_t lo = (int8_t)(b & 0xF); if (lo & 8) lo -= 16;
- int8_t hi = (int8_t)((b >> 4) & 0xF); if (hi & 8) hi -= 16;
- dst[i] = lo; dst[i + 1] = hi;
- }
- dst += lens[t];
- }
+ /* #1271's unpack_int4_to_int8 (branchless, AVX2/NEON/scalar -- see its
+ * definition above load_expert_merged, which already uses it for the
+ * container's int4 read) instead of this function's own separate scalar
+ * copy of the same nibble-unpack math. g4/u4/d4 are three independently
+ * malloc'd packed buffers here (unlike load_expert_merged's single
+ * contiguous `raw`), so one call per segment. */
+ unpack_int4_to_int8(w, s->g4, ng);
+ unpack_int4_to_int8(w + ng, s->u4, ng);
+ unpack_int4_to_int8(w + ng + ng, s->d4, nd);
s->g = w; s->u = w + ng; s->d = w + ng + ng;
}
@@ -1879,9 +2095,11 @@ static void attention(Model *m, Layer *l, int layer, float *x, int S, int pos_ba
float *q = falloc((int64_t)S*q_out);
float *k = falloc((int64_t)S*kv_out);
float *vv= falloc((int64_t)S*kv_out);
- matmul_d(q, x, l->q, S, D, q_out);
- matmul_d(k, x, l->k, S, D, kv_out);
- matmul_d(vv, x, l->v, S, D, kv_out);
+ /* The projections the tier placed answer from VRAM for the whole batch,
+ * with one backend call per matrix; unavailable handles use CPU matmul. */
+ if (!qtd_batch(l->qth_q, q, x, S, D, q_out)) matmul_d(q, x, &l->q, S, D, q_out);
+ if (!qtd_batch(l->qth_k, k, x, S, D, kv_out)) matmul_d(k, x, &l->k, S, D, kv_out);
+ if (!qtd_batch(l->qth_v, vv, x, S, D, kv_out)) matmul_d(vv, x, &l->v, S, D, kv_out);
/* split q into query (first hd) and gate (next gate_dim), both per head */
float *query = falloc((int64_t)S*H*hd);
float *gate = falloc((int64_t)S*H*gate_dim);
@@ -1943,7 +2161,7 @@ static void attention(Model *m, Layer *l, int layer, float *x, int S, int pos_ba
float g = gate_dim ? gate[o] : 0.f;
ag[o] = ctx[o] * (1.f / (1.f + expf(-g)));
}
- matmul_d(out, ag, l->o, S, H*hd, D);
+ if (!qtd_batch(l->qth_o, out, ag, S, H*hd, D)) matmul_d(out, ag, &l->o, S, H*hd, D);
free(q); free(k); free(vv); free(query); free(gate); free(ctx); free(ag);
}
@@ -1976,10 +2194,10 @@ static void qwen_shared_experts_cpu(Model *m, Layer *l, const float *x, int S,
if (B == 1) {
for (int s=0;ssh_g,1,D,I);
- matmul_d(u,xs,l->sh_u,1,D,I);
+ if(!qtd(l->qth_shg,g,xs,D,I)) matmul_d(g,xs,&l->sh_g,1,D,I);
+ if(!qtd(l->qth_shu,u,xs,D,I)) matmul_d(u,xs,&l->sh_u,1,D,I);
for(int i=0;ish_d,1,I,D);
+ if(!qtd(l->qth_shd,hh,g,I,D)) matmul_d(hh,g,&l->sh_d,1,I,D);
float sgate=1.f;
if(l->sh_gate){float sg=0.f;for(int i=0;ish_gate[i];sgate=1.f/(1.f+expf(-sg));}
float *os=out+(int64_t)s*D;
@@ -1990,10 +2208,10 @@ static void qwen_shared_experts_cpu(Model *m, Layer *l, const float *x, int S,
float *bh=falloc((int64_t)B*D);
for(int base=0;basesh_g,rows,D,I);
- matmul_d(bu,x+(int64_t)base*D,l->sh_u,rows,D,I);
+ matmul_d(bg,x+(int64_t)base*D,&l->sh_g,rows,D,I);
+ matmul_d(bu,x+(int64_t)base*D,&l->sh_u,rows,D,I);
for(int64_t q=0;q<(int64_t)rows*I;q++){float sv=bg[q];bg[q]=(sv/(1.f+expf(-sv)))*bu[q];}
- matmul_d(bh,bg,l->sh_d,rows,I,D);
+ matmul_d(bh,bg,&l->sh_d,rows,I,D);
for(int s=0;s ROUTE_RANK_MAX) K = ROUTE_RANK_MAX;
+ if (J < 0) J = 0; if (J > K) J = K;
+ int cap = (P > 0.f && P < 1.f) ? (M > 4*K ? M : 4*K) : (M > K ? M : K);
+ if (cap > E) cap = E;
+ if (cap > ROUTE_RANK_MAX) cap = ROUTE_RANK_MAX;
+ int rank[ROUTE_RANK_MAX]; float rw[ROUTE_RANK_MAX]; int8_t rl[ROUTE_RANK_MAX];
+ int n = 0;
+ for (int r = 0; r < cap; r++) {
+ int best = -1; float bv = -1e30f;
+ for (int e = 0; e < E; e++) {
+ if (keep && !keep[e]) continue;
+ int taken = 0; for (int j = 0; j < n; j++) if (rank[j] == e) { taken = 1; break; }
+ if (!taken && pr[e] > bv) { bv = pr[e]; best = e; }
+ }
+ if (best < 0) break;
+ rank[n] = best; rw[n] = bv; n++;
+ }
+ int Kt = K < n ? K : n; /* the true top-K is rank[0..Kt) */
+ int win = n;
+ if (P > 0.f && P < 1.f) { /* cumulative-mass window: grow past K until P of the ranked mass */
+ float tot = 1e-20f; for (int r = 0; r < n; r++) tot += rw[r] > 0 ? rw[r] : 0;
+ float cum = 0; win = Kt;
+ for (int r = 0; r < n; r++) { cum += rw[r] > 0 ? rw[r] : 0; win = r + 1; if (cum >= P * tot) break; }
+ if (win < Kt) win = Kt;
+ }
+ for (int r = 0; r < n; r++) rl[r] = (r >= J && r < win) ? (int8_t)lvl(ctx, rank[r]) : 0;
+ int chosen = 0, pos[ROUTE_RANK_MAX]; uint8_t used[ROUTE_RANK_MAX] = {0};
+ for (int r = 0; r < J && r < n && chosen < K; r++) { pos[chosen++] = r; used[r] = 1; }
+ for (int level = 2; level >= 1; level--)
+ for (int r = J; r < win && chosen < K; r++)
+ if (!used[r] && rl[r] == level) { pos[chosen++] = r; used[r] = 1; }
+ for (int r = 0; r < n && chosen < K; r++)
+ if (!used[r]) { pos[chosen++] = r; used[r] = 1; }
+ for (int kk = 0; kk < chosen; kk++) {
+ int r = pos[kk]; idx[kk] = rank[r]; val[kk] = rw[r];
+ if (r >= Kt && alpha > 0.f && alpha < 1.f) val[kk] *= alpha;
+ }
+ for (int kk = chosen; kk < K; kk++) { idx[kk] = -1; val[kk] = 0.f; } /* fewer eligible than K: keep mask */
+ if (!st) return;
+ st->slots += (uint64_t)chosen; st->agree_tot += (uint64_t)chosen;
+ float tsum = 1e-20f, csum = 1e-20f;
+ for (int t = 0; t < Kt; t++) tsum += rw[t] > 0 ? rw[t] : 0;
+ for (int kk = 0; kk < chosen; kk++) {
+ csum += val[kk] > 0 ? val[kk] : 0;
+ if (pos[kk] < Kt) st->agree_hit++;
+ else { st->swaps++; if (rl[pos[kk]] == 2) st->swaps_vram++; }
+ }
+ double kl = 0; /* KL(true top-K mass || chosen mass), as the GLM meter */
+ for (int t = 0; t < Kt; t++) {
+ double pt = (rw[t] > 0 ? rw[t] : 0) / tsum; if (pt <= 0) continue;
+ double pc = 1e-12;
+ for (int kk = 0; kk < chosen; kk++) if (pos[kk] == t) { pc = (val[kk] > 0 ? val[kk] : 0) / csum; break; }
+ kl += pt * log(pt / pc);
+ }
+ st->kl_sum += kl; st->kl_n++;
+}
+
+/* Residency levels for route_select: 2 = in the VRAM tier, 1 = in this
+ * layer's RAM cache (pinned or LRU), 0 = would be read from disk. */
+typedef struct { Model *m; int layer; } RouteCtx;
+static int route_level(void *vctx, int e) {
+ RouteCtx *rc = (RouteCtx *)vctx;
+ if (qt_is_resident(rc->layer, e)) return 2;
+ pthread_mutex_lock(&g_pilot_mx);
+ Slot *s = slot_indexed(rc->m, rc->layer, e);
+ pthread_mutex_unlock(&g_pilot_mx);
+ return s ? 1 : 0;
+}
+
+static void route_footer(FILE *f, const Model *m) {
+ if (g_cache_route && m->route.slots)
+ fprintf(f, "CACHE_ROUTE J=%d M=%d P=%.2f alpha=%.2f | swap %.1f%% (%llu/%llu, %llu to VRAM)\n",
+ g_route_j, g_route_m, g_route_p, g_route_alpha,
+ 100.0*m->route.swaps/m->route.slots, (unsigned long long)m->route.swaps,
+ (unsigned long long)m->route.slots, (unsigned long long)m->route.swaps_vram);
+ if (m->route.agree_tot)
+ fprintf(f, "route_agree %.1f%% | route_kl %.4f\n", 100.0*m->route.agree_hit/m->route.agree_tot,
+ m->route.kl_n ? m->route.kl_sum/(double)m->route.kl_n : 0.0);
+}
+
static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
Cfg *c = &m->c; int D = c->hidden, E = c->n_experts, K = c->topk, I = c->inter;
float *logits = falloc((int64_t)S*E);
double _tr = tm_now();
- matmul_d(logits, x, l->gate, S, D, E);
+ matmul_d(logits, x, &l->gate, S, D, E);
tm_add(S, 4, tm_now()-_tr);
if (c->has_bias && l->gate_bias) {
for (int s = 0; s < S; s++) { float *pr = logits + (int64_t)s*E; for (int e = 0; e < E; e++) pr[e] += l->gate_bias[e]; }
@@ -2102,14 +2417,23 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
for (int e = 0; e < Ec; e++) keep[e] = 1;
}
int idx[256]; float val[256];
- for (int kk = 0; kk < K; kk++) {
- int best = -1; float bv = -1e30f;
- for (int e = 0; e < E; e++) {
- if (!keep[e]) continue;
- int taken = 0; for (int j = 0; j < kk; j++) if (idx[j]==e){taken=1;break;}
- if (!taken && pr[e] > bv) { bv = pr[e]; best = e; }
+ if (g_cache_route) {
+ RouteCtx rc = { m, layer };
+ route_select(pr, keep, E, K, g_route_j, g_route_m, g_route_p, g_route_alpha,
+ route_level, &rc, idx, val, &m->route);
+ } else {
+ for (int kk = 0; kk < K; kk++) {
+ int best = -1; float bv = -1e30f;
+ for (int e = 0; e < E; e++) {
+ if (!keep[e]) continue;
+ int taken = 0; for (int j = 0; j < kk; j++) if (idx[j]==e){taken=1;break;}
+ if (!taken && pr[e] > bv) { bv = pr[e]; best = e; }
+ }
+ idx[kk] = best; val[kk] = bv;
+ }
+ if (g_route_agree) { /* plain routing: full agreement by construction */
+ m->route.agree_hit += (uint64_t)K; m->route.agree_tot += (uint64_t)K; m->route.kl_n++;
}
- idx[kk] = best; val[kk] = bv;
}
if (m->resident_collecting) {
for (int kk = 0; kk < K; kk++) if (idx[kk] >= 0) m->seen[(int64_t)layer * E + idx[kk]] = 1;
@@ -2146,7 +2470,7 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
matmul_qe(g, xs, e->g, e->gs, D, I);
matmul_qe(u, xs, e->u, e->us, D, I);
for (int i = 0; i < I; i++) { float gv = g[i]; g[i] = (gv / (1.f + expf(-gv))) * u[i]; }
- matmul_qe(hh, g, e->d, e->ds, I, D);
+ matmul_qd(hh, g, e->d, e->ds, I, D);
float w = val[kk]; float *os = out + (int64_t)s*D;
for (int d = 0; d < D; d++) os[d] += w * hh[d];
}
@@ -2155,10 +2479,10 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
{
double _ts2 = tm_now();
int Ish = c->shared_inter;
- matmul_d(sh, xs, l->sh_g, 1, D, Ish);
- matmul_d(shu, xs, l->sh_u, 1, D, Ish);
+ if (!qtd(l->qth_shg, sh, xs, D, Ish)) matmul_d(sh, xs, &l->sh_g, 1, D, Ish);
+ if (!qtd(l->qth_shu, shu, xs, D, Ish)) matmul_d(shu, xs, &l->sh_u, 1, D, Ish);
for (int i = 0; i < Ish; i++) { float sv = sh[i]; sh[i] = (sv / (1.f + expf(-sv))) * shu[i]; }
- matmul_d(shd, sh, l->sh_d, 1, Ish, D);
+ if (!qtd(l->qth_shd, shd, sh, Ish, D)) matmul_d(shd, sh, &l->sh_d, 1, Ish, D);
float sgate = 1.f;
if (l->sh_gate) {
float sg = 0.f; const float *wg = l->sh_gate;
@@ -2170,7 +2494,10 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
tm_add(S, 3, tm_now()-_ts2);
}
double _q2 = tm_now();
- qt_take(qmask, val, K, out + (int64_t)s*D);
+ if(!qt_take(qmask, val, K, out + (int64_t)s*D)){
+ fprintf(stderr,"qwen36: CUDA expert collection failed at layer %d; stopping inference\n",layer);
+ exit(1);
+ }
if (tm_on() && S==1) {
extern double g_qt_iss, g_qt_cpu, g_qt_tak;
g_qt_iss += _q1-_q0; g_qt_cpu += _q2-_q1; g_qt_tak += tm_now()-_q2;
@@ -2182,7 +2509,7 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
matmul_qe(g, xs, e->g, e->gs, D, I);
matmul_qe(u, xs, e->u, e->us, D, I);
for (int i = 0; i < I; i++) { float gv = g[i]; g[i] = (gv / (1.f + expf(-gv))) * u[i]; }
- matmul_qe(hh, g, e->d, e->ds, I, D);
+ matmul_qd(hh, g, e->d, e->ds, I, D);
float w = val[kk];
float *os = out + (int64_t)s*D;
for (int d = 0; d < D; d++) os[d] += w * hh[d];
@@ -2209,6 +2536,14 @@ static void moe(Model *m, Layer *l, int layer, float *x, int S, float *out) {
* split conv_out -> q_in/k_in/v_in; repeat_interleave q,k by rep; l2norm
* (q scaled by 1/sqrt(kdim)); recurrence S[h]*=exp(g); kv=k@S; delta=(v-kv)*beta;
* S+=k (x) delta; out=q@S; per-head Gated RMSNorm (plain weight) -> out_proj. */
+/* Bound both the host result block and device input/output staging. */
+static int dnproj_batch_rows(int S, int H, int O) {
+ int64_t rows = (32LL << 20) / (((int64_t)H + O) * sizeof(float));
+ if (rows < 1) rows = 1;
+ if (rows > 256) rows = 256;
+ return S < rows ? S : (int)rows;
+}
+
static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_base, float *out) {
(void)pos_base;
Cfg *c = &m->c;
@@ -2220,12 +2555,12 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas
float scale = 1.f / sqrtf((float)kdim);
int H = c->hidden;
- /* qkv and z live in ONE buffer: the fused GPU projection writes
- * [conv_dim ++ value_dim] in a single GEMV, and the CPU fallback fills the
- * same two regions. Either way the code below reads qkv/z unchanged. */
- float *qkvz = falloc((int64_t)conv_dim + value_dim);
- float *qkv = qkvz;
- float *z = qkvz + conv_dim;
+ /* Input projections have no recurrent dependency. Keep qkv ++ z for a
+ * bounded block, then consume rows in order through conv and recurrence. */
+ int proj_dim = conv_dim + value_dim;
+ int B = qt_dnproj_ready(layer) ? dnproj_batch_rows(S, H, proj_dim) : 1;
+ float *qkvz = falloc((int64_t)B * proj_dim);
+ int gpu_block = 0;
float *b = falloc(vh);
float *a = falloc(vh);
float *beta= falloc(vh);
@@ -2245,11 +2580,15 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas
const float *xs = x + (int64_t)s * H;
extern double g_dn_sub[4];
double _d0 = tm_now();
- /* projections (single-token matmuls). One fused GEMV when this layer's
- * dnproj is placed on a GPU, the two CPU matmuls otherwise. */
- if (!qt_dnproj_matmul(layer, qkvz, xs, H, conv_dim + value_dim)) {
- matmul_d(qkv, xs, l->dn_qkv, 1, H, conv_dim);
- matmul_d(z, xs, l->dn_z, 1, H, value_dim);
+ if (s % B == 0) {
+ int rows = S - s < B ? S - s : B;
+ gpu_block = qt_dnproj_matmul_batch(layer, qkvz, xs, rows, H, proj_dim);
+ }
+ float *qkv = qkvz + (int64_t)(s % B) * proj_dim;
+ float *z = qkv + conv_dim;
+ if (!gpu_block) {
+ matmul_d(qkv, xs, &l->dn_qkv, 1, H, conv_dim);
+ matmul_d(z, xs, &l->dn_z, 1, H, value_dim);
}
matmul(b, xs, l->dn_b, 1, H, vh);
matmul(a, xs, l->dn_a, 1, H, vh);
@@ -2347,7 +2686,8 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas
outr[(int64_t)h * vdim + d] = val * zr[d] / (1.f + expf(-zr[d]));
}
}
- matmul_d(out + (int64_t)s * H, outr, l->dn_out, 1, value_dim, H);
+ if (!qtd(l->qth_dnout, out + (int64_t)s * H, outr, value_dim, H))
+ matmul_d(out + (int64_t)s * H, outr, &l->dn_out, 1, value_dim, H);
if (tm_on() && S==1){ g_dn_sub[3]+=tm_now()-_d0; }
if (layer == 0 && s == 0 && getenv("DN_DBG")) {
FILE *dbg = fopen(getenv("DN_DBG"), "wb");
@@ -2371,6 +2711,117 @@ static void deltanet(Model *m, Layer *l, int layer, float *x, int S, int pos_bas
free(conv_out); free(q); free(k); free(outv); free(outr); free(kv); free(delta);
}
+/* The rest of the dense trunk, offered to the placer by name and layer with
+ * the bytes of the dense-i8 copies (docs/qwen36-cuda-tier.md, "Placement"):
+ * dnout -- the DeltaNet out_proj, one matrix per DeltaNet layer
+ * attnproj -- q, k, v, o of every attention layer, offered as one item
+ * shexp -- gate, up, down of the shared expert, every layer
+ * Measured on ds (CPU, 8 threads): of the 37.5 ms a DeltaNet layer stack
+ * costs per decoded token, 23.4 are the input projections (already placeable
+ * as "dnproj"), 8.3 the out_proj and norm, 3.3 the convolution and 2.4 the
+ * recurrence -- the matmuls are the cost, not the recurrence, so this is
+ * where the trunk goes. Offer order after lmhead and dnproj: dnout, attnproj,
+ * shexp, each in layer order, so a partial placement is whole layers. A
+ * component is offered only when every matrix of it has a dense-i8 copy. */
+static void trunk_offer_dense(Model *m){
+ Cfg *c = &m->c;
+ for (int i = 0; i < c->n_layers; i++) {
+ if (c->is_attn[i]) continue;
+ size_t b = qdw_bytes(&m->L[i].dn_out);
+ if (b) qt_trunk_offer("dnout", i, b);
+ }
+ for (int i = 0; i < c->n_layers; i++) {
+ if (!c->is_attn[i]) continue;
+ Layer *l = &m->L[i];
+ size_t bq = qdw_bytes(&l->q), bk = qdw_bytes(&l->k), bv = qdw_bytes(&l->v), bo = qdw_bytes(&l->o);
+ if (bq && bk && bv && bo) qt_trunk_offer("attnproj", i, bq + bk + bv + bo);
+ }
+ for (int i = 0; i < c->n_layers; i++) {
+ Layer *l = &m->L[i];
+ size_t bg = qdw_bytes(&l->sh_g), bu = qdw_bytes(&l->sh_u), bd = qdw_bytes(&l->sh_d);
+ if (bg && bu && bd) qt_trunk_offer("shexp", i, bg + bu + bd);
+ }
+}
+/* After qt_init decided: upload what was placed, keep the handles in the
+ * Layer. Every matrix falls back on its own, so a failed upload costs one
+ * GEMV on the CPU, never the component. Returns the number of matrices
+ * placed; `vram_bytes` gets their size. */
+static int trunk_place_dense(Model *m, double *vram_bytes){
+ Cfg *c = &m->c; int placed = 0; double vram = 0;
+ for (int i = 0; i < c->n_layers; i++) {
+ Layer *l = &m->L[i];
+ if (!c->is_attn[i]) {
+ l->qth_dnout = qdw_place(&l->dn_out, qt_place_of("dnout", i));
+ if (l->qth_dnout) { placed++; vram += (double)qdw_bytes(&l->dn_out); }
+ } else {
+ int dev = qt_place_of("attnproj", i);
+ l->qth_q = qdw_place(&l->q, dev); l->qth_k = qdw_place(&l->k, dev);
+ l->qth_v = qdw_place(&l->v, dev); l->qth_o = qdw_place(&l->o, dev);
+ if (l->qth_q) { placed++; vram += (double)qdw_bytes(&l->q); }
+ if (l->qth_k) { placed++; vram += (double)qdw_bytes(&l->k); }
+ if (l->qth_v) { placed++; vram += (double)qdw_bytes(&l->v); }
+ if (l->qth_o) { placed++; vram += (double)qdw_bytes(&l->o); }
+ }
+ int dev = qt_place_of("shexp", i);
+ l->qth_shg = qdw_place(&l->sh_g, dev); l->qth_shu = qdw_place(&l->sh_u, dev); l->qth_shd = qdw_place(&l->sh_d, dev);
+ if (l->qth_shg) { placed++; vram += (double)qdw_bytes(&l->sh_g); }
+ if (l->qth_shu) { placed++; vram += (double)qdw_bytes(&l->sh_u); }
+ if (l->qth_shd) { placed++; vram += (double)qdw_bytes(&l->sh_d); }
+ }
+ if (vram_bytes) *vram_bytes = vram;
+ return placed;
+}
+
+/* Measured, not assumed. The placer prices a trunk component by the bytes it
+ * saves on the CPU's memory bus, which presumes the GPU answers a GEMV faster
+ * than the CPU does. Four Tesla M10 (sm_50) said otherwise: every placed
+ * component ran slower there, lm_head 68.8 ms against 41.7 on the CPU, and
+ * decode fell from 3.56 to 2.68 tok/s (#1652). So before any trunk upload,
+ * one DeltaNet input projection (the most numerous placed matrix) is timed
+ * both ways on the device that would host it, ten GEMVs each, best of three
+ * rounds after a warm-up, and the trunk goes to VRAM only if the GPU wins.
+ * Only the automatic placement is questioned: a hand-written COLI_PLACE
+ * stands. COLI_TRUNK_PROBE=0 skips the probe and trusts the placer. The
+ * probe's copy stays resident (one projection, ~25 MB on the 35B). */
+static int trunk_probe_gpu_wins(Model *m){
+ const char *e = getenv("COLI_TRUNK_PROBE");
+ if (e && *e == '0') return 1;
+ if (!qt_place_is_auto()) return 1;
+ Cfg *c = &m->c;
+ QW *w = NULL; int dev = QT_PLACE_CPU;
+ for (int i = 0; i < c->n_layers && !w; i++) {
+ if (c->is_attn[i]) continue;
+ if (m->L[i].dn_qkv.q) { w = &m->L[i].dn_qkv; dev = qt_place_of("dnproj", i); }
+ }
+ if (!w) return 1; /* dense-i8 off: nothing will be placed */
+ if (dev == QT_PLACE_CPU) dev = qt_place_of("lmhead", 0);
+ if (dev == QT_PLACE_CPU) return 1; /* nothing placed: nothing to measure */
+ int I = w->I, O = w->O;
+ int h = qt_dense_init(w->q, w->sc, I, O, dev);
+ if (h < 0) return 1; /* cannot measure: the placer's word stands */
+ float *x = malloc((size_t)I * sizeof(float)), *y = malloc((size_t)O * sizeof(float));
+ if (!x || !y) { free(x); free(y); return 1; }
+ for (int i = 0; i < I; i++) x[i] = sinf(0.37f * (float)i);
+ double gpu = 1e30, cpu = 1e30;
+ for (int r = 0; r < 3; r++) {
+ for (int k = 0; k < 3; k++) if (!qt_dense_matmul(h, y, x, I, O)) { free(x); free(y); return 1; }
+ double t0 = now_s();
+ for (int k = 0; k < 10; k++) if (!qt_dense_matmul(h, y, x, I, O)) { free(x); free(y); return 1; }
+ double tg = (now_s() - t0) / 10;
+ for (int k = 0; k < 3; k++) matmul_q(y, x, w->q, w->sc, I, O);
+ t0 = now_s();
+ for (int k = 0; k < 10; k++) matmul_q(y, x, w->q, w->sc, I, O);
+ double tc = (now_s() - t0) / 10;
+ if (tg < gpu) gpu = tg;
+ if (tc < cpu) cpu = tc;
+ }
+ free(x); free(y);
+ int wins = gpu < cpu;
+ fprintf(stderr, "[place] probe: one [%d x %d] int8 GEMV takes %.3f ms on CUDA dev %d, %.3f ms on the CPU -> trunk %s\n",
+ O, I, gpu * 1e3, dev, cpu * 1e3, wins ? "to VRAM" : "stays on the CPU");
+ return wins;
+}
+
static void layers_forward_range(Model *m, float *x, int S, int pos_base,
int layer_begin, int layer_end,
int allow_prefetch, FILE *lf) {
@@ -2438,10 +2889,13 @@ static int g_pin_use_logit = 0; /* 1 quando questa richiesta e ripartita da
typedef struct { float **rec, **conv; int n_layers; } Q36PinState;
/* Stato della lettura del prefill: dichiarato qui perche step() lo consulta e
- * step() viene prima del codice di servizio che lo accende. */
+ * step() viene prima del codice di servizio che lo accende. Solo il servizio
+ * lo accende e solo il servizio definisce serve_echo: senza main non esiste. */
+#ifndef QWEN36_NO_MAIN
static int g_echo_k = 0; /* 0 = spento */
static const char *g_echo_id = NULL;
static void serve_echo(const char *id, int pos, int token, const float *lo, int V, int k);
+#endif
static float *step(Model *m, const int *ids, int S, int pos_base) {
Cfg *c = &m->c; int D = c->hidden;
@@ -2482,6 +2936,7 @@ static float *step(Model *m, const int *ids, int S, int pos_base) {
* token, il cui predittore sta nello stato precedente. Per questo il
* chiamante arretra di uno il riuso del prefisso quando la lettura e
* accesa: cosi il primo token dell'opzione ricade sempre qui dentro. */
+#ifndef QWEN36_NO_MAIN
if (g_echo_k > 0 && g_echo_id && S > 0) {
float *erow = falloc(D), *elog = falloc(c->vocab);
/* Il primo token fresco e predetto dallo stato PRECEDENTE, che dopo un
@@ -2492,17 +2947,18 @@ static float *step(Model *m, const int *ids, int S, int pos_base) {
for (int p = 0; p + 1 < S; p++) {
rmsnorm_row(erow, x + (int64_t)p*D, m->final_norm, D, c->eps);
if (!qt_lmhead_matmul(elog, erow, D, c->vocab))
- matmul_d(elog, erow, m->lm_head, 1, D, c->vocab);
+ matmul_d(elog, erow, &m->lm_head, 1, D, c->vocab);
serve_echo(g_echo_id, pos_base + p + 1, ids[p+1], elog, c->vocab, g_echo_k);
}
free(erow); free(elog);
}
+#endif
float *last = falloc(D);
rmsnorm_row(last, x + (int64_t)(S-1)*D, m->final_norm, D, c->eps);
float *logit = falloc(c->vocab);
double _th = tm_now();
if (!qt_lmhead_matmul(logit, last, D, c->vocab))
- matmul_d(logit, last, m->lm_head, 1, D, c->vocab);
+ matmul_d(logit, last, &m->lm_head, 1, D, c->vocab);
if (tm_on()) { tm_add(S, 5, tm_now()-_th); if (S==1) g_tm_dec_tokens++; else g_tm_pre_tokens += S; }
free(x); free(last);
if (lf) fclose(lf);
@@ -2563,7 +3019,7 @@ static void pilot_prefetch(Model *m, int lnext, const float *x, int S) {
Layer *l = &m->L[lnext];
float *nrm_x = falloc((int64_t)S * D);
for (int s = 0; s < S; s++) rmsnorm_row(nrm_x + (int64_t)s*D, x + (int64_t)s*D, l->post_ln, D, c->eps);
- matmul_d(logits, nrm_x, l->gate, S, D, E); /* int8 copy (f32 may be freed) */
+ matmul_d(logits, nrm_x, &l->gate, S, D, E); /* int8 copy (f32 may be freed) */
free(nrm_x);
for (int s = 0; s < S; s++) {
float *pr = logits + (int64_t)s*E;
@@ -3092,14 +3548,36 @@ static void hits_emit(Model *m){
}
static double tm_sum(int idx){ return (g_tm_dec[idx]+g_tm_pre[idx])/1e3; } /* ms -> s */
+/* The generation budget a request gets. max_tokens is a CEILING, not a
+ * target (#260/#382, the rule GLM and DeepSeek V4 already apply): the prompt
+ * must fit with room for one token (none for a read-only logprobs request,
+ * docs/brio.md), and the budget is then clamped to what the context can hold.
+ * Returns the budget, or -1 when the PROMPT does not fit. Refusing when
+ * prompt + budget exceeded the context (#1641) turned the gateway's default
+ * output budget -- 8192 here, the whole default context -- into a 400 on
+ * every message of `coli chat` and on every request without max_tokens. */
+static int qwen36_serve_budget(int np, int max_tok, int max_ctx, int read_only){
+ if (np < 1) return -1;
+ int room = max_ctx - np;
+ if (read_only) return room < 0 ? -1 : (max_tok < room ? max_tok : room);
+ if (room < 1) return -1;
+ return max_tok > room ? room : max_tok;
+}
+
static void serve_one(Model *m, ServeReq *q){
int *ids=NULL, np=0;
encode_text(q->payload, &ids, &np); /* payload is raw prompt text; qwen36 adds no BOS */
int max_ctx = qwen36_max_ctx();
- if(np<1 || np+q->max_tok>max_ctx){
+ int budget = qwen36_serve_budget(np, q->max_tok, max_ctx, q->logprobs > 0);
+ if(budget < 0){
printf("ERROR %s CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d\n",q->id,np,q->max_tok,max_ctx);
fflush(stdout); free(ids); return;
}
+ if(budget < q->max_tok){
+ fprintf(stderr,"[serve] max_tokens %d clamped to %d (context %d - prompt %d); raise Q36_MAXT for longer answers\n",
+ q->max_tok, budget, max_ctx, np);
+ q->max_tok = budget;
+ }
printf("ACCEPT %s %d\n",q->id,np); fflush(stdout);
m->max_t = np + q->max_tok;
/* Grow the cache BEFORE deciding, so the decision sees the state that will
@@ -3317,6 +3795,15 @@ int main(int argc, char **argv) {
g_pilot = getenv("PILOT") ? atoi(getenv("PILOT")) : 0;
g_wide = getenv("WIDE") ? atoi(getenv("WIDE")) : 1;
if (g_wide < 1) g_wide = 1; if (g_wide > 4) g_wide = 4;
+ g_cache_route = getenv("CACHE_ROUTE") ? atoi(getenv("CACHE_ROUTE")) : 0; /* prefer resident experts (VRAM tier, then RAM cache) inside the top-M window; changes which experts run: docs/CACHE_ROUTE.md */
+ g_route_j = getenv("ROUTE_J") ? atoi(getenv("ROUTE_J")) : 2; /* true top-J always taken, resident or not, under CACHE_ROUTE=1 */
+ g_route_m = getenv("ROUTE_M") ? atoi(getenv("ROUTE_M")) : 12; /* rank window inside which a resident expert may replace an unresident one */
+ g_route_p = getenv("ROUTE_P") ? (float)atof(getenv("ROUTE_P")) : 0.f; /* cumulative-mass window for CACHE_ROUTE (0 = fixed M) */
+ g_route_alpha = getenv("ROUTE_ALPHA") ? (float)atof(getenv("ROUTE_ALPHA")) : 1.f; /* scale substituted experts' gate mass before renorm (1 = off) */
+ g_route_agree = getenv("ROUTE_AGREE") ? atoi(getenv("ROUTE_AGREE")) : g_cache_route; /* overlap% + KL vs the true top-K in the footer; auto-on under CACHE_ROUTE=1 */
+ if (g_cache_route)
+ fprintf(stderr, "[qwen36] CACHE_ROUTE=1 J=%d M=%d P=%.2f alpha=%.2f: VRAM tier > RAM cache > disk inside the top-M window (lossy: A/B it)\n",
+ g_route_j, g_route_m, g_route_p, g_route_alpha);
if (getenv("OPENAI")) g_openai = 1; /* OpenAI-compatible output */
const char *mv = getenv("MODEL"); if (mv && *mv) g_model = mv;
int hot_n = getenv("HOT") ? atoi(getenv("HOT")) : 0;
@@ -3394,36 +3881,10 @@ int main(int argc, char **argv) {
g_expert_gs = m.c.expert_gs;
if (g_expert_gs) fprintf(stderr, "[qwen36] group-scaled experts: gs=%d\n", g_expert_gs);
fprintf(stderr, "resident weights loaded in %.1fs | RSS after load: %.2f GB\n", m.dense_load_s, rss_gb());
- /* quantize the large dense matrices to int8 (COLI_DENSE_I8=0 disables) */
- if (dense_i8_on()) {
- double tq = now_s();
- Cfg *qc = &m.c; int D2 = qc->hidden;
- int q_out = qc->q_heads * qc->q_head_dim, kv_out = qc->kv_heads * qc->k_head_dim;
- for (int i = 0; i < qc->n_layers; i++) {
- Layer *l = &m.L[i];
- qdw_register(l->q, D2, q_out); qdw_register(l->k, D2, kv_out);
- qdw_register(l->v, D2, kv_out); qdw_register(l->o, qc->o_in, D2);
- qdw_register(l->gate, D2, qc->n_experts);
- qdw_register(l->sh_g, D2, qc->shared_inter); qdw_register(l->sh_u, D2, qc->shared_inter);
- qdw_register(l->sh_d, qc->shared_inter, D2);
- qdw_register(l->dn_qkv, D2, qc->dn_conv_dim);
- qdw_register(l->dn_z, D2, qc->dn_vheads * qc->dn_vdim);
- qdw_register(l->dn_out, qc->dn_vheads * qc->dn_vdim, D2);
- }
- qdw_register(m.lm_head, D2, qc->vocab);
- /* Free the f32 originals -- the pointers only serve as lookup keys in
- * matmul_d from here on (never dereferenced again).
- * COLI_KEEP_F32=1 keeps them (debug). */
- double freed = 0;
- if (!getenv("COLI_KEEP_F32")) {
- for (int i = 0; i < g_qdw_n; i++) {
- freed += (double)g_qdw[i].I * g_qdw[i].O * sizeof(float);
- free((void*)g_qdw[i].w);
- }
- }
- fprintf(stderr, "[dense-i8] %d matrices quantized in %.1f s, %.1f GB f32 freed\n",
- g_qdw_n, now_s()-tq, freed/1073741824.0);
- }
+ /* dense matrices are quantized to int8 (+ int4 planar per COLI_DENSE_BITS/
+ * COLI_DENSE_INT4, see dense_int4_wanted) during model_init above
+ * (COLI_DENSE_I8=0 disables it) -- see load_tq/QW; model_init_range
+ * already logged the count and freed bytes. */
/* Optional CUDA VRAM expert tier (COLI_CUDA=1): hot experts live in
* DEVICE_LOCAL memory across the configured GPUs, misses fall back to the
@@ -3434,7 +3895,7 @@ int main(int argc, char **argv) {
* riservare qualunque budget: e' int4 impacchettato che va in VRAM come
* fmt=4, int8 come fmt=1. Sbagliare qui era #1331 -- budget riservato,
* planned=1, e zero promozioni per tutta la vita del processo. */
- int expert_is_int4 = 1;
+ int expert_is_int4 = 1, expert_mixed = 0;
{
char probe[256];
snprintf(probe, sizeof(probe),
@@ -3442,41 +3903,46 @@ int main(int argc, char **argv) {
st_tensor *pt = st_find(&m.S, probe);
int64_t want = 2*(int64_t)m.c.inter*m.c.hidden + (int64_t)m.c.hidden*m.c.inter;
if (pt && pt->nbytes == want) expert_is_int4 = 0; /* int8: un byte per elemento */
+ /* mixed (convert_qwen36.py --down-bits): int4 gate/up + int8 down = 2/3 of int8 */
+ if (pt && pt->nbytes == want * 2 / 3) { expert_is_int4 = 0; expert_mixed = 1; }
}
/* Una riga, sempre: e' l'unico modo di verificare il probe dall'esterno
* (CI sul container tiny int8, #1331) senza una scheda. */
fprintf(stderr, "[qwen36] expert format on disk: %s\n",
- expert_is_int4 ? "int4 packed (tier fmt=4)" : "int8 (tier fmt=1)");
+ expert_mixed ? "int4 gate/up + int8 down (mixed; CPU path, no VRAM tier yet)"
+ : expert_is_int4 ? "int4 packed (tier fmt=4)" : "int8 (tier fmt=1)");
g_expert_is_int4 = expert_is_int4;
+ g_expert_mixed = expert_mixed; g_expert_down_gs = expert_mixed ? m.c.expert_down_gs : 0;
+ if (expert_mixed && m.c.expert_down_bits == 0)
+ fprintf(stderr, "[qwen36] mixed layout on disk but qwen36_meta.json has no expert_down_bits -- down scales assumed per row\n");
/* Offer the dense trunk to the placer before the tier decides its budget:
* sizes only, from the same dense-i8 entries the uploads below will use.
* No entry (dense-i8 off) means nothing to offer, and the CPU path stands. */
{
int O_qkv = m.c.dn_conv_dim, O_z = m.c.dn_vheads * m.c.dn_vdim;
- for (int i = 0; i < g_qdw_n; i++)
- if (g_qdw[i].w == m.lm_head)
- qt_trunk_offer("lmhead", 0, (size_t)g_qdw[i].I * g_qdw[i].O + (size_t)g_qdw[i].O * sizeof(float));
+ if (m.lm_head.q)
+ qt_trunk_offer("lmhead", 0, (size_t)m.lm_head.I * m.lm_head.O + (size_t)m.lm_head.O * sizeof(float));
for (int i = 0; i < m.c.n_layers; i++) {
if (m.c.is_attn[i]) continue;
- int have = 0;
- for (int j = 0; j < g_qdw_n; j++)
- if (g_qdw[j].w == m.L[i].dn_qkv || g_qdw[j].w == m.L[i].dn_z) have++;
- if (have == 2)
+ if (m.L[i].dn_qkv.q && m.L[i].dn_z.q)
qt_trunk_offer("dnproj", i, (size_t)(O_qkv + O_z) * m.c.hidden + (size_t)(O_qkv + O_z) * sizeof(float));
}
+ trunk_offer_dense(&m); /* dnout, attnproj, shexp: the rest of the per-token dense work */
}
- if (qt_init(m.c.n_layers, m.c.n_experts, m.c.hidden, m.c.inter, cap, m.c.topk,
+ if (expert_mixed && getenv("COLI_CUDA") && getenv("COLI_CUDA")[0] == '1')
+ fprintf(stderr, "[qwen36] COLI_CUDA=1 ignored: the VRAM expert tier does not take the mixed layout yet (one format per expert)\n");
+ if (!expert_mixed &&
+ qt_init(m.c.n_layers, m.c.n_experts, m.c.hidden, m.c.inter, cap, m.c.topk,
m.c.expert_gs, expert_is_int4)) {
fprintf(stderr, "[gpu] MoE experts -> CUDA VRAM tier\n");
atexit(qt_shutdown);
+ /* The placer predicted; measure before uploading a byte of trunk. */
+ if (!trunk_probe_gpu_wins(&m)) qt_trunk_withdraw("measured slower than the CPU");
/* R4 role split: park the dense-i8 lm_head on COLI_LMHEAD_GPU. The
- * qdw entry keyed by m.lm_head holds the int8 rows + per-row scales
- * the CPU path uses; the GPU applies the identical semantics. */
- for (int i = 0; i < g_qdw_n; i++)
- if (g_qdw[i].w == m.lm_head) {
- qt_lmhead_init(g_qdw[i].q, g_qdw[i].sc, g_qdw[i].I, g_qdw[i].O);
- break;
- }
+ * QW struct on m.lm_head holds the int8 rows + per-row scales the
+ * CPU path uses; the GPU applies the identical semantics. */
+ if (m.lm_head.q)
+ qt_lmhead_init(m.lm_head.q, m.lm_head.sc, m.lm_head.I, m.lm_head.O);
/* R4 step 2: DeltaNet input projections, per layer, wherever
* COLI_PLACE puts them. qkv and z are both [O_x, hidden] int8 with
* per-row scales, so fusing them is a concatenation along O -- two
@@ -3490,11 +3956,8 @@ int main(int argc, char **argv) {
if (m.c.is_attn[i]) continue;
int dev = qt_place_of("dnproj", i);
if (dev == QT_PLACE_CPU) continue;
- const int8_t *q1 = NULL, *q2 = NULL; const float *s1 = NULL, *s2 = NULL;
- for (int j = 0; j < g_qdw_n; j++) {
- if (g_qdw[j].w == m.L[i].dn_qkv) { q1 = g_qdw[j].q; s1 = g_qdw[j].sc; }
- if (g_qdw[j].w == m.L[i].dn_z) { q2 = g_qdw[j].q; s2 = g_qdw[j].sc; }
- }
+ const int8_t *q1 = m.L[i].dn_qkv.q, *q2 = m.L[i].dn_z.q;
+ const float *s1 = m.L[i].dn_qkv.sc, *s2 = m.L[i].dn_z.sc;
if (!q1 || !q2) continue; /* dense-i8 off: CPU path stands */
int8_t *qf = malloc((size_t)Of * Hd);
float *sf = malloc((size_t)Of * sizeof(float));
@@ -3512,6 +3975,15 @@ int main(int argc, char **argv) {
fprintf(stderr, "[dnp] %d DeltaNet-Projektionen auf GPU (%.2f GB VRAM)\n",
placed, vram / 1073741824.0);
}
+ /* The rest of the trunk, wherever the placer put it: out_proj,
+ * attention projections, shared expert. Handles live in the Layer. */
+ {
+ double vram = 0;
+ int placed = trunk_place_dense(&m, &vram);
+ if (placed)
+ fprintf(stderr, "[dense] %d trunk matrices on GPU (dnout/attnproj/shexp, %.2f GB VRAM)\n",
+ placed, vram / 1073741824.0);
+ }
/* Warmstart: fill the VRAM budget BEFORE the first token (heat order
* when HEAT_FILE exists, natural order otherwise), loading all RAM
* slots along the way. */
@@ -3536,6 +4008,7 @@ int main(int argc, char **argv) {
printf("TF-NLL: %.4f nats/token over %d tokens | ppl = %.2f\n", nll, scored, exp(nll));
printf("Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0,
(unsigned long long)m.hits, (unsigned long long)m.miss);
+ route_footer(stdout, &m);
printf("Speed: %.2f tok/s (%.1fs for %d tokens) | PEAK RSS: %.2f GB\n", scored/dt, dt, scored, rss_gb());
free(buf); free(arena); return 0;
}
@@ -3601,6 +4074,7 @@ int main(int argc, char **argv) {
fprintf(stderr, "\nPEAK RSS: %.2f GB\n", rss_gb());
fprintf(stderr, "Expert cache hit rate: %.1f%% (hit=%llu miss=%llu)\n", tot?100.0*m.hits/tot:0.0,
(unsigned long long)m.hits, (unsigned long long)m.miss);
+ route_footer(stderr, &m);
fprintf(stderr, "Speed: %.2f tok/s (%.1fs for %d tokens)\n", n_new/dt, dt, n_new);
free(buf); free(arena);
/* Oracle mode is a gate, not a report: a mismatch must fail the caller.
@@ -3629,12 +4103,12 @@ typedef struct {
static void qwen36_segment_layer_free(Layer *layer) {
free(layer->in_ln); free(layer->post_ln);
- free(layer->q); free(layer->k); free(layer->v); free(layer->o);
- free(layer->qn); free(layer->kn); free(layer->gate); free(layer->gate_bias);
- free(layer->sh_g); free(layer->sh_u); free(layer->sh_d); free(layer->sh_gate);
- free(layer->dn_qkv); free(layer->dn_z); free(layer->dn_b); free(layer->dn_a);
+ qw_free(&layer->q); qw_free(&layer->k); qw_free(&layer->v); qw_free(&layer->o);
+ free(layer->qn); free(layer->kn); qw_free(&layer->gate); free(layer->gate_bias);
+ qw_free(&layer->sh_g); qw_free(&layer->sh_u); qw_free(&layer->sh_d); free(layer->sh_gate);
+ qw_free(&layer->dn_qkv); qw_free(&layer->dn_z); free(layer->dn_b); free(layer->dn_a);
free(layer->dn_conv); free(layer->dn_dtbias); free(layer->dn_alog);
- free(layer->dn_norm); free(layer->dn_out);
+ free(layer->dn_norm); qw_free(&layer->dn_out);
}
static void qwen36_segment_model_destroy(Qwen36SegmentEngine *engine) {
@@ -4022,7 +4496,7 @@ static void qwen36_edge_engine_destroy(void *engine_impl) {
Qwen36EdgeEngine *engine = (Qwen36EdgeEngine *)engine_impl;
if (!engine) return;
free(engine->model.embed);
- free(engine->model.lm_head);
+ qw_free(&engine->model.lm_head);
free(engine->model.final_norm);
free(engine->model.c.is_attn);
st_destroy(&engine->model.S);
@@ -4061,9 +4535,10 @@ static int qwen36_edge_engine_open(
engine->model.embed = load_t_n(
&engine->model, "model.embed_tokens.weight",
(int64_t)config->vocab * config->hidden);
- engine->model.lm_head = load_t_n(
- &engine->model, "lm_head.weight",
- (int64_t)config->vocab * config->hidden);
+ /* quantize=0: this engine never ran the old post-hoc qdw_register pass
+ * either (only main()'s static Model did), so lm_head stays f32-only here,
+ * exactly as before. */
+ load_tq(&engine->model, "lm_head.weight", config->hidden, config->vocab, 0, "lmhead", &engine->model.lm_head);
engine->model.final_norm = load_t_n(
&engine->model, "model.norm.weight", config->hidden);
engine->model.quant_bits = container_layer_is_int4(&engine->model, 0) ? 4 : 8;
@@ -4211,7 +4686,7 @@ static int qwen36_edge_select(void *engine_impl,
}
rmsnorm_row(normalized, input + (size_t)row * config->hidden,
engine->model.final_norm, config->hidden, config->eps);
- matmul_d(logits, normalized, engine->model.lm_head,
+ matmul_d(logits, normalized, &engine->model.lm_head,
1, config->hidden, config->vocab);
if (coli_edge_argmax(logits, (uint32_t)config->vocab,
&request->token_ids[row],
@@ -4242,7 +4717,7 @@ static int qwen36_edge_logits(void *engine_impl,
rmsnorm_row(normalized, input + (size_t)row * config->hidden,
engine->model.final_norm, config->hidden, config->eps);
matmul_d(request->logits + (size_t)row * config->vocab,
- normalized, engine->model.lm_head,
+ normalized, &engine->model.lm_head,
1, config->hidden, config->vocab);
}
free(normalized);
diff --git a/c/qwen36_tier.c b/c/qwen36_tier.c
index e06c2de6a..742769d9a 100644
--- a/c/qwen36_tier.c
+++ b/c/qwen36_tier.c
@@ -45,7 +45,7 @@ static struct {
QSlot *slot; /* [nl*ne] */
pthread_mutex_t mx;
pthread_t th;
- int th_stop;
+ int th_stop, waiters;
/* upload ring with staging copies */
struct { int layer, eid; uint8_t *w; float *s; int v_layer, v_eid; } q[QT_QCAP];
int qh, qt_, qn;
@@ -68,6 +68,13 @@ static struct {
uint32_t *heat0; /* heat table loaded from HEAT_FILE */
} G;
+/* Count parked callers so shutdown can reclaim their shared storage safely. */
+static void wait_take_locked(void){
+ G.waiters++;
+ pthread_cond_wait(&G.cv_take,&G.mx);
+ if(--G.waiters==0 && G.th_stop) pthread_cond_broadcast(&G.cv_take);
+}
+
static QSlot *qs(int layer, int eid){ return &G.slot[(size_t)layer*G.ne + eid]; }
static int home(int eid){ return eid % G.ndev; }
@@ -156,7 +163,7 @@ static void *uploader(void *arg){
pthread_cond_broadcast(&G.cv_take); /* queue space available */
if(ve>=0){
/* LFRU swap: free the victim only when no group is in flight */
- while(G.issue_open && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx);
+ while(G.issue_open && !G.th_stop) wait_take_locked();
QSlot *v=qs(vl,ve);
if(G.th_stop && G.issue_open){
/* Shutting down with a group still open: qt_take() -- the only
@@ -205,6 +212,11 @@ static void *uploader(void *arg){
&& coli_cuda_tensor_upload(&td, w+2*mb, sc+2*G.Ih, 2, G.Ih, G.D, dv);
}
free(w); free(sc);
+ if(!ok){
+ if(tg) coli_cuda_tensor_free(tg);
+ if(tu) coli_cuda_tensor_free(tu);
+ if(td) coli_cuda_tensor_free(td);
+ }
pthread_mutex_lock(&G.mx);
QSlot *s=qs(layer,eid);
if(ok){ s->tg=tg; s->tu=tu; s->td=td; s->resident=1; G.uploads++; }
@@ -408,6 +420,29 @@ static double auto_displaced_value(int di, size_t room, int k, size_t exp_bytes,
return value;
}
+/* Whether the trunk placement is the automatic one (COLI_PLACE unset or
+ * "auto"): the only decision the engine's startup probe may overturn. A
+ * hand-written list, or "off", is the user's word and stands. */
+int qt_place_is_auto(void){ return G_auto_on && auto_mode(); }
+
+/* Undo the automatic trunk placement: every offer back to the CPU, lm_head and
+ * the DeltaNet projections included, and the bytes it had taken back into each
+ * device's expert budget. The engine calls this BEFORE any trunk upload, when
+ * its startup probe measured the GPU GEMV slower than the CPU's: on four Tesla
+ * M10 every placed component lost, lm_head 68.8 ms against 41.7 on the CPU,
+ * decode 2.68 against 3.56 tok/s (#1652). Nothing to undo when the placement
+ * was not automatic. */
+void qt_trunk_withdraw(const char *why){
+ if(!qt_place_is_auto()) return;
+ size_t back = 0;
+ for(int o = 0; o < G_offer_n; o++) G_offer[o].dev = QT_PLACE_CPU;
+ G_auto_lmh = QT_PLACE_CPU; G_lmh.dev_ok = 0;
+ for(int l = 0; l < QT_DN_MAX_LAYERS; l++) G_auto_dnp[l] = QT_PLACE_CPU;
+ for(int i = 0; i < G.ndev; i++){ back += G_trunk_bytes[i]; G.budget[i] += G_trunk_bytes[i]; G_trunk_bytes[i] = 0; }
+ fprintf(stderr,"[place] trunk stays on the CPU (%s): %.2f GB of VRAM back to the experts\n",
+ why ? why : "withdrawn", back/1073741824.0);
+}
+
static void auto_place(int nl, int ne, int topk, const size_t *capacity, const uint32_t *heat0){
size_t room[QT_MAX_DEV];
for(int i = 0; i < G.ndev; i++){ room[i] = capacity[i]; G_trunk_bytes[i] = 0; }
@@ -468,6 +503,7 @@ static const float *G_fp8_lut;
static int G_upload_sync; /* QT_UPLOAD_SYNC=1: qt_issue waits for in-flight uploads first (tests) */
int qt_init_fp8(int nl, int ne, int D, int Ih, int cap, int topk, const float *e4m3_lut){
+ if(G.on) return 0;
G_fp8_stream = 1; G_fp8_lut = e4m3_lut;
int ok = qt_init(nl, ne, D, Ih, cap, topk, 0, 0);
if(!ok) G_fp8_stream = 0;
@@ -494,6 +530,7 @@ static size_t dev_alloc_footprint(size_t bytes){
int qt_init(int nl, int ne, int D, int Ih, int cap, int topk, int expert_gs,
int expert_is_int4){
+ if(G.on) return 0;
const char *e=getenv("COLI_CUDA");
if(!(e && *e=='1')) return 0;
if(cap != ne && !G_fp8_stream){
@@ -531,7 +568,7 @@ int qt_init(int nl, int ne, int D, int Ih, int cap, int topk, int expert_gs,
* the caller repeat every device in COLI_GPUS as well -- forgetting that
* would silently drop a component back to the CPU mid-A/B. */
{
- static const char *comps[] = {"lmhead","dnproj","dnout","attnproj"};
+ static const char *comps[] = {"lmhead","dnproj","dnout","attnproj","shexp"};
for(size_t ci=0; ci CPU\n",ld);
}
/* every other component's devices, deduplicated */
- static const char *comps[] = {"dnproj","dnout","attnproj"};
+ static const char *comps[] = {"dnproj","dnout","attnproj","shexp"};
for(size_t ci=0; ci COLI_GPUS bleibt\n",ed);
} else if(!G_auto_on && G_place_n && qt_place_named("experts")){
fprintf(stderr,"[place] experts=cpu -> VRAM-Tier aus\n");
- return 0;
+ goto fail_storage;
} else if(nres && !G_auto_on){
int w=0;
for(int i=0;i= G_dense_n || !G_dense[h].on) return 0;
- if(coli_cuda_matmul(&G_dense[h].t, y, x, NULL, NULL, 1, 1, I, O, G_dense[h].dev, 0)) return 1;
+int qt_dense_matmul_batch(int h, float *y, const float *x, int S, int I, int O){
+ if(h < 0 || h >= G_dense_n || !G_dense[h].on || S <= 0) return 0;
+ if(coli_cuda_matmul(&G_dense[h].t, y, x, NULL, NULL, 1, S, I, O, G_dense[h].dev, 0)) return 1;
fprintf(stderr,"[dense] handle %d GPU matmul failed; CPU from here on\n", h);
G_dense[h].on = 0;
return 0;
}
+int qt_dense_matmul(int h, float *y, const float *x, int I, int O){
+ return qt_dense_matmul_batch(h, y, x, 1, I, O);
+}
int qt_dense_count(void){ return G_dense_n; }
-int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O){
- if(layer < 0 || layer >= QT_DN_MAX_LAYERS || !G_dnp[layer].on) return 0;
- if(coli_cuda_matmul(&G_dnp[layer].t,y,x,NULL,NULL,1,1,I,O,G_dnp[layer].dev,0))
+int qt_dnproj_ready(int layer){
+ return layer >= 0 && layer < QT_DN_MAX_LAYERS && G_dnp[layer].on;
+}
+int qt_dnproj_matmul_batch(int layer, float *y, const float *x, int S, int I, int O){
+ if(!qt_dnproj_ready(layer) || S <= 0) return 0;
+ if(coli_cuda_matmul(&G_dnp[layer].t,y,x,NULL,NULL,1,S,I,O,G_dnp[layer].dev,0))
return 1;
fprintf(stderr,"[dnp] layer %d GPU matmul failed; CPU from here on\n", layer);
G_dnp[layer].on = 0;
return 0;
}
+int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O){
+ return qt_dnproj_matmul_batch(layer, y, x, 1, I, O);
+}
+
int qt_lmhead_matmul(float *y, const float *x, int I, int O){
if(!G_lmh.on) return 0;
/* cached-tensor path: upload params are ignored once *t exists */
@@ -899,7 +963,7 @@ void qt_note_block(int layer,int eid,
pthread_mutex_lock(&G.mx);
if(G_fp8_stream) stream_point(s,g4,u4,d4,gs,us,ds);
else if(!s->g4){ s->g4=g4; s->u4=u4; s->d4=d4; s->gs=gs; s->us=us; s->ds=ds; }
- while(G.qn>=QT_QCAP && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx);
+ while(G.qn>=QT_QCAP && !G.th_stop) wait_take_locked();
enqueue_locked(layer,eid,-1,-1,0);
if(G_fp8_stream) stream_forget(s);
pthread_mutex_unlock(&G.mx);
@@ -992,7 +1056,7 @@ void qt_note_planned(int layer,int eid,
}
if(G_fp8_stream) stream_point(s,g4,u4,d4,gs,us,ds);
else if(!s->g4){ s->g4=g4; s->u4=u4; s->d4=d4; s->gs=gs; s->us=us; s->ds=ds; }
- while(G.qn>=QT_QCAP && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx);
+ while(G.qn>=QT_QCAP && !G.th_stop) wait_take_locked();
if(!enqueue_locked(layer,eid,-1,-1,1)){
/* not enqueueable (e.g. already resident): return the reservation */
if(s->planned) G.used[home(eid)]-=G.exp_bytes;
@@ -1011,7 +1075,7 @@ void qt_note_planned(int layer,int eid,
void qt_fill_wait(void){
if(!G.on) return;
pthread_mutex_lock(&G.mx);
- while(G.inflight>0 && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx);
+ while(G.inflight>0 && !G.th_stop) wait_take_locked();
pthread_mutex_unlock(&G.mx);
}
@@ -1058,7 +1122,8 @@ uint32_t qt_issue(int layer,const int *eids,int K,const float *x){
* did not get scheduled once before the run was over (0 uploads, 0 hits,
* six entries still queued). No group is open here, so the wait cannot
* meet a swap parked on issue_open. */
- if(G_upload_sync) while(G.inflight>0 && !G.th_stop) pthread_cond_wait(&G.cv_take,&G.mx);
+ if(G_upload_sync) while(G.inflight>0 && !G.th_stop) wait_take_locked();
+ if(G.th_stop){ pthread_mutex_unlock(&G.mx); return 0; }
if(layer==0) qt_lfru_tick_locked();
G.issue_open=1;
for(int k=0;ktg) coli_cuda_tensor_free(s->tg);
+ if(s->tu) coli_cuda_tensor_free(s->tu);
+ if(s->td) coli_cuda_tensor_free(s->td);
+ }
+ free(G.slot); G.slot=NULL;
+ free(G.is_x); G.is_x=NULL; G.is_x_floats=0;
+ free(G.fill_order); G.fill_order=NULL;
+ free(G.heat0); G.heat0=NULL;
+ pthread_cond_destroy(&G.cv_take);
+ pthread_cond_destroy(&G.cv);
+ pthread_mutex_destroy(&G.mx);
+ G.issue_open=0;
+ memset(G.is_cnt,0,sizeof G.is_cnt);
G.on=0;
G_fp8_stream=0;
coli_cuda_shutdown();
diff --git a/c/qwen36_tier.h b/c/qwen36_tier.h
index 5caef9f6e..41a73bf73 100644
--- a/c/qwen36_tier.h
+++ b/c/qwen36_tier.h
@@ -65,6 +65,12 @@ int qt_place_of(const char *component, int layer);
* bytes come out of that device's expert budget. Sizes only; the tensors
* follow through qt_lmhead_init / qt_dnproj_init as before. */
void qt_trunk_offer(const char *component, int layer, size_t bytes);
+/* The automatic placement is a prediction; the engine measures it at startup
+ * (one GEMV both ways, qwen36.c trunk_probe_gpu_wins) and withdraws the whole
+ * trunk when the GPU loses, giving the bytes back to the expert budget. Only
+ * the automatic placement can be withdrawn; a COLI_PLACE list stands. */
+int qt_place_is_auto(void);
+void qt_trunk_withdraw(const char *why);
/* DeltaNet input projections, qkv ++ z fused into one resident tensor per
* layer: one GEMV instead of two, and the engine's qkv/z buffers are laid out
@@ -72,6 +78,8 @@ void qt_trunk_offer(const char *component, int layer, size_t bytes);
int qt_dnproj_init(int layer, const int8_t *q, const float *sc,
int I, int O, int device);
int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O);
+int qt_dnproj_ready(int layer);
+int qt_dnproj_matmul_batch(int layer, float *y, const float *x, int S, int I, int O);
/* Generic resident dense matrix (int8 per-row, one GEMV per call), addressed
* by a handle: the Qwen3.8 trunk uses this for every matrix it places. Offer
* the size with qt_trunk_offer(name, layer, bytes) before qt_init, ask
@@ -79,6 +87,8 @@ int qt_dnproj_matmul(int layer, float *y, const float *x, int I, int O);
* Returns the handle (>= 0) or -1 (stays on the CPU). */
int qt_dense_init(const int8_t *q, const float *sc, int I, int O, int device);
int qt_dense_matmul(int handle, float *y, const float *x, int I, int O);
+/* Row-major x[S,I] -> y[S,O], using the same resident int8 tensor. */
+int qt_dense_matmul_batch(int handle, float *y, const float *x, int S, int I, int O);
int qt_dense_count(void);
/* fp8 streaming mode (Qwen3.8): experts arrive as e4m3 bytes with 128x128
@@ -86,6 +96,8 @@ int qt_dense_count(void);
* the tier copies what it uploads inside the qt_note call and keeps no
* pointer into the engine's slot. e4m3_lut is quant.h's E4M3_LUT, published
* to the backend so fmt=8 uploads are accepted. */
+/* Init returns 0 without changing an active tier. Shut down before reinit;
+ * callers must serialize init/shutdown with new work. */
int qt_init_fp8(int n_layers, int n_experts, int hidden, int inter,
int cap_experts_per_layer, int topk, const float *e4m3_lut);
int qt_init(int n_layers, int n_experts, int hidden, int inter,
@@ -108,8 +120,10 @@ void qt_note(int layer, int eid,
* the GPU. Compute the misses on the CPU, then call qt_take(). */
uint32_t qt_issue(int layer, const int *eids, int K, const float *x);
-/* Collect the GPU results and accumulate val[k]*y_k into out[hidden]. */
-void qt_take(uint32_t mask, const float *val, int K, float *out);
+/* Collect all GPU results and accumulate val[k]*y_k into out[hidden].
+ * Returns 0 on collection failure, leaving out unchanged. The caller must
+ * stop inference: experts selected by qt_issue were not computed on CPU. */
+int qt_take(uint32_t mask, const float *val, int K, float *out);
/* Warmstart: plan the full fill set (heat order, budget reserved), then any
* number of loader threads may call qt_note_planned per planned expert. */
@@ -135,17 +149,22 @@ static inline int qt_lmhead_matmul(float*a,const float*b,int c,int d){(void)a;(
#define QT_PLACE_CPU (-1)
static inline int qt_place_of(const char*a,int b){(void)a;(void)b;return QT_PLACE_CPU;}
static inline void qt_trunk_offer(const char*a,int b,size_t c){(void)a;(void)b;(void)c;}
+static inline int qt_place_is_auto(void){return 0;}
+static inline void qt_trunk_withdraw(const char*a){(void)a;}
static inline int qt_dnproj_init(int a,const int8_t*b,const float*c,int d,int e,int f){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;return 0;}
static inline int qt_dnproj_matmul(int a,float*b,const float*c,int d,int e){(void)a;(void)b;(void)c;(void)d;(void)e;return 0;}
+static inline int qt_dnproj_ready(int a){(void)a;return 0;}
+static inline int qt_dnproj_matmul_batch(int a,float*b,const float*c,int d,int e,int f){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;return 0;}
static inline int qt_dense_init(const int8_t*a,const float*b,int c,int d,int e){(void)a;(void)b;(void)c;(void)d;(void)e;return -1;}
static inline int qt_dense_matmul(int a,float*b,const float*c,int d,int e){(void)a;(void)b;(void)c;(void)d;(void)e;return 0;}
+static inline int qt_dense_matmul_batch(int a,float*b,const float*c,int d,int e,int f){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;return 0;}
static inline int qt_dense_count(void){return 0;}
static inline int qt_ready(void){return 0;}
static inline int qt_is_resident(int a,int b){(void)a;(void)b;return 0;}
static inline void qt_shutdown(void){}
static inline void qt_note(int a,int b,const uint8_t*c,const uint8_t*d,const uint8_t*e,const float*f,const float*g,const float*h){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;(void)g;(void)h;}
static inline uint32_t qt_issue(int a,const int*b,int c,const float*d){(void)a;(void)b;(void)c;(void)d;return 0;}
-static inline void qt_take(uint32_t a,const float*b,int c,float*d){(void)a;(void)b;(void)c;(void)d;}
+static inline int qt_take(uint32_t a,const float*b,int c,float*d){(void)b;(void)c;(void)d;return a==0;}
static inline int qt_plan_fill(int*a,int*b,int c){(void)a;(void)b;(void)c;return 0;}
static inline void qt_note_planned(int a,int b,const uint8_t*c,const uint8_t*d,const uint8_t*e,const float*f,const float*g,const float*h){(void)a;(void)b;(void)c;(void)d;(void)e;(void)f;(void)g;(void)h;}
static inline int qt_fill_next(int*a,int*b){(void)a;(void)b;return 0;}
diff --git a/c/qwen38.c b/c/qwen38.c
index 9eea6cfb1..494f0b389 100644
--- a/c/qwen38.c
+++ b/c/qwen38.c
@@ -1514,8 +1514,10 @@ static int q38_prefix_cache_save(Model *m,const int *ids,int len,const float *lo
static int q38_prefix_restore(Model *m,const int *ids,int len){
if(!m||!ids||len<1||!g_q38_prefix.valid||g_q38_prefix.owner!=m||
g_q38_prefix.len<1||g_q38_prefix.len>len||
- memcmp(g_q38_prefix.ids,ids,(size_t)g_q38_prefix.len*sizeof(int)))return 0;
- q38_prefix_copy_state(m,0);m->kv_len=g_q38_prefix.len;return g_q38_prefix.len;
+ memcmp(g_q38_prefix.ids,ids,(size_t)g_q38_prefix.len*sizeof(int))||
+ !kv_prefix_holds(&m->kvp,g_q38_prefix.ids,g_q38_prefix.len))return 0;
+ q38_prefix_copy_state(m,0);m->kv_len=g_q38_prefix.len;
+ m->kvp.len=g_q38_prefix.len;return g_q38_prefix.len;
}
static const float *q38_prefix_cached_logits(Model *m){
@@ -1595,6 +1597,22 @@ static void serve_hits(Model *m){
printf("HITS %d %d %s\n",rows,E,hex); fflush(stdout); free(hex); free(bm);
}
+/* The generation budget a request gets. max_tokens is a CEILING, not a
+ * target (#260/#382, the rule GLM and DeepSeek V4 already apply): the prompt
+ * must fit with room for one token (none for a read-only logprobs request,
+ * docs/brio.md), and the budget is then clamped to what the context can hold.
+ * Returns the budget, or -1 when the PROMPT does not fit. Refusing when
+ * prompt + budget exceeded the context (#1641) turned the gateway's default
+ * output budget -- 8192 here, the whole default context -- into a 400 on
+ * every message of `coli chat` and on every request without max_tokens. */
+static int q38_serve_budget(int np, int max_tok, int max_ctx, int read_only){
+ if (np < 1) return -1;
+ int room = max_ctx - np;
+ if (read_only) return room < 0 ? -1 : (max_tok < room ? max_tok : room);
+ if (room < 1) return -1;
+ return max_tok > room ? room : max_tok;
+}
+
static int serve_one(Model *m, ServeReq *q){
int *ids=NULL, np=0;
encode_text_n(q->payload,(size_t)q->plen,&ids,&np); /* byte-counted prompt; qwen38 adds no BOS */
@@ -1619,11 +1637,16 @@ static int serve_one(Model *m, ServeReq *q){
}
int max_ctx=m->kv_cap;
/* max_tokens=0 in modalita jev: leggere il prompt e fermarsi (serve_codec.h) */
- int tok_min = (q->logprobs > 0) ? 0 : 1;
- if(np<1 || np>max_ctx || q->max_tokmax_tok>max_ctx-np){
+ int budget = q38_serve_budget(np, q->max_tok, max_ctx, q->logprobs > 0);
+ if(budget < 0){
printf("ERROR %s CONTEXT_EXCEEDED prompt_tokens=%d requested=%d capacity=%d\n",q->id,np,q->max_tok,max_ctx);
fflush(stdout); free(ids); return 0;
}
+ if(budget < q->max_tok){
+ fprintf(stderr,"[serve] max_tokens %d clamped to %d (context %d - prompt %d); raise Q38_MAXT for longer answers\n",
+ q->max_tok, budget, max_ctx, np);
+ q->max_tok = budget;
+ }
printf("ACCEPT %s %d\n",q->id,np); fflush(stdout);
double request_started=now_s();
uint64_t hits_before=m->hits, misses_before=m->miss;
@@ -1632,13 +1655,20 @@ static int serve_one(Model *m, ServeReq *q){
* il client ha dichiarato, e battono la cache automatica, che insegue solo
* l'ultimo prompt. Se nessuno serve, si ricade su quella. */
int reuse=0; const float *pin_lo=NULL;
+ /* Image embeddings are not described by token identity. */
+ if(m->vis_map && m->vis_rows_n>0){
+ kv_prefix_taint(&m->kvp);q38_prefix_cache_invalidate();
+ }
{
int ps=coli_pin_best(&g_q38_pins,ids,np);
while(ps>=0){
ColiPin *k=&g_q38_pins.slot[ps];
Q38PinState *st=(Q38PinState*)k->state;
- if(st && q38_pin_state_copy(m,&st,0)){
- m->kv_len=k->len; reuse=k->len; pin_lo=k->logit;
+ /* Pins omit K/V/indexer rows. Reject one whose rows were overwritten. */
+ if(st && kv_prefix_holds(&m->kvp,k->ids,k->len) &&
+ q38_pin_state_copy(m,&st,0)){
+ m->kv_len=k->len; m->kvp.len=k->len;
+ reuse=k->len; pin_lo=k->logit;
coli_pin_touch(&g_q38_pins,ps);
break;
}
@@ -1918,7 +1948,7 @@ int main(int argc, char **argv) {
Model m; model_init(&m, snap, cap, bits);
q38_tier_start(&m, cap); /* COLI_CUDA=1: hot experts stream to VRAM (qwen36_tier.c) */
- q38_trunk_cpu_int8(&m); /* Q38_TRUNK_CPU_INT8=1: the trunk's int8 rows on the CPU (reference) */
+ q38_trunk_cpu_int8(&m); /* the trunk's int8 rows on the CPU, BF16 released (Q38_TRUNK_CPU_INT8=0 keeps BF16) */
if(is_ref)ref_logits=read_reference_logits(ref_root,m.c.vocab);
g_capture_last_logit=ref_logits!=NULL||getenv("DUMP")!=NULL;
q38_telemetry_init(snap, &m);
diff --git a/c/qwen38_core.h b/c/qwen38_core.h
index 30dcfdb24..8c658e5ae 100644
--- a/c/qwen38_core.h
+++ b/c/qwen38_core.h
@@ -9,6 +9,7 @@
*/
#ifndef COLI_QWEN38_CORE_H
#define COLI_QWEN38_CORE_H
+#include "kv_prefix.h"
#include /* q38_ehit_mark publishes the lazy HITS table under a lock */
#define Q38_MAX_LAYERS 512
@@ -50,7 +51,7 @@ typedef struct {
Q38WeightKind kind;
unsigned owns_data:1, owns_scales:1;
int gpu; /* 0 = CPU; else 1 + tier handle of an int8 copy resident in VRAM (decode, S == 1) */
- int8_t *q8; float *q8sc; /* Q38_TRUNK_CPU_INT8=1: the same int8 rows kept on the CPU (reference for the GPU path, no GPU needed) */
+ int8_t *q8; float *q8sc; /* the trunk's int8 rows on the CPU (default; Q38_TRUNK_CPU_INT8=0 keeps BF16): the same rows the GPU holds, met by an int8 activation in idot.h */
} Q38Weight;
typedef struct { float *norm; Q38Weight down, up, inject; } GatedResidual;
@@ -110,6 +111,15 @@ typedef struct {
int ready; /* 0 unknown, 1 resident, -1 incompatible */
} Q38ExpertScaleCache;
+typedef enum {
+ Q38_EXPERT_BATCH_FALLBACK_NONE = 0,
+ Q38_EXPERT_BATCH_FALLBACK_DISABLED,
+ Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY,
+ Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK,
+ Q38_EXPERT_BATCH_FALLBACK_DUPLICATE,
+ Q38_EXPERT_BATCH_FALLBACK_LAYOUT,
+} Q38ExpertBatchFallback;
+
typedef struct {
Cfg c;
shards S;
@@ -127,6 +137,7 @@ typedef struct {
float **DN_rec, **DN_conv;
float **K, **V, **IK;
int kv_len, kv_cap, max_t;
+ kv_prefix kvp; /* token identity of the live attention rows, not a snapshot */
st_tensor *ple_parts[Q38_MAX_PLE_PARTS];
char ple_part_names[Q38_MAX_PLE_PARTS][320];
int64_t ple_part_start[Q38_MAX_PLE_PARTS + 1];
@@ -138,8 +149,10 @@ typedef struct {
int ple_history_len;
int range_begin, range_end;
int native_fp8, native_bf16, expert_prefetch, expert_parallel_reads;
+ Q38ExpertBatchFallback expert_batch_fallback;
int prefill_batch;
uint64_t resident_weight_bytes;
+ int trunk_table_built; /* q38_trunk_offer_all ran for this load (the table is process-wide, the model is not) */
double dense_load_s;
/* vision. `vis_map` mappa la posizione ASSOLUTA nella sequenza alla riga di
* `vis_rows`, oppure -1. Assoluta e non relativa al chunk: il prefill arriva
@@ -264,6 +277,75 @@ static void q38_matmul_bf16(float *y,const float *x,const uint16_t *W,
}
}
+/* The routed experts as the checkpoint ships them: e4m3 bytes with one f32
+ * scale per 128x128 block. quant.h's matmul_fp8 decodes every byte through a
+ * 256-entry table, one gather per weight; this kernel decodes eight bytes at a
+ * time in registers (quant.h e4m3_decode8: a shift and one multiply, NaNs
+ * kept) and multiplies them with FMA. The block scale still applies once per
+ * block and the blocks still add in double, so the result differs from the
+ * scalar kernel only by the float summation order inside a block. For a
+ * batch of rows (prefill) the block is decoded once and held while every row
+ * runs through it: the matrix streams past once, not once per row.
+ * Q38_FP8_KERNEL=scalar restores the table kernel (bisecting a difference). */
+static int q38_fp8_vector_on(void) {
+ static int v=-1;
+ if(v<0){ const char *e=getenv("Q38_FP8_KERNEL"); v=!(e&&!strcmp(e,"scalar")); }
+ return v;
+}
+#ifdef __AVX2__
+#define Q38_FP8_ROWS 8
+static void q38_matmul_fp8_vec(float *y,const float *x,const uint8_t *q8,
+ const float *bscale,int S,int I,int O) {
+ int64_t nblkI=fp8_nblk(I);
+ #pragma omp parallel for schedule(static)
+ for(int o=0;ogpu&&weight->rows==O&&weight->cols==I&&
qt_dense_matmul(weight->gpu-1,y,x,I,O))return;
- if(S==1&&weight&&weight->q8&&weight->rows==O&&weight->cols==I){
- /* the int8 rows the GPU would hold, computed here: what the trunk
- * quantization alone does to the output, GPU or not */
- const int8_t *q=weight->q8; const float *sc=weight->q8sc;
- #pragma omp parallel for schedule(static)
- for(int o=0;oq8&&weight->rows==O&&weight->cols==I){
+ /* the trunk's int8 rows (the same the GPU holds) meet an int8
+ * activation in the integer kernel: x quantized once per row with one
+ * scale, then maddubs / vpdpbusd dot products (idot.h). Decode and
+ * prefill take the same path, so GPU or not the trunk quantization is
+ * the only thing that separates the output from the BF16 run. */
+ int8_t *xq=(int8_t*)malloc((size_t)S*I); float *sx=(float*)malloc((size_t)S*sizeof(float));
+ if(!xq||!sx){fprintf(stderr,"OOM activation quantization\n");exit(1);}
+ for(int s=0;sq8,weight->q8sc,S,I,O);
+ free(xq);free(sx);
return;
}
if(!weight||weight->rows!=O||weight->cols!=I||!weight->data){
@@ -293,7 +376,7 @@ static void q38_weight_matmul(float *y,const float *x,const Q38Weight *weight,
else if(weight->kind==Q38_WEIGHT_BF16)
q38_matmul_bf16(y,x,(const uint16_t*)weight->data,S,I,O);
else if(weight->kind==Q38_WEIGHT_FP8&&weight->scales)
- matmul_fp8(y,x,(const uint8_t*)weight->data,weight->scales,S,I,O);
+ q38_matmul_fp8(y,x,(const uint8_t*)weight->data,weight->scales,S,I,O);
else {fprintf(stderr,"unsupported matmul weight kind %d\n",(int)weight->kind);exit(1);}
}
@@ -1310,26 +1393,72 @@ typedef struct {
* residents are protected from victim selection, so no worker can overwrite a
* slot another selected expert will consume. Smaller caches and heterogeneous
* layouts retain the serial LRU path. */
+/* Says once per model why the parallel read path is not taken, so a slow
+ * prefill on a small cache or a converted container is not a mystery. */
+static int q38_expert_batch_fallback(Model *m,Q38ExpertBatchFallback reason,
+ int layer,int first,int second) {
+ if(m->expert_batch_fallback!=Q38_EXPERT_BATCH_FALLBACK_NONE)return 0;
+ m->expert_batch_fallback=reason;
+ fprintf(stderr,"[qwen38 expert I/O] parallel reads unavailable: ");
+ switch(reason){
+ case Q38_EXPERT_BATCH_FALLBACK_DISABLED:
+ fprintf(stderr,"Q38_EXPERT_PARALLEL_READS=0");break;
+ case Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY:
+ fprintf(stderr,"cache holds %d experts/layer but the route needs %d; "
+ "lower --ctx/Q38_MAXT or raise --ram",first,second);break;
+ case Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK:
+ fprintf(stderr,"layer %d has no compatible resident FP8 scale bank",layer);break;
+ case Q38_EXPERT_BATCH_FALLBACK_DUPLICATE:
+ fprintf(stderr,"layer %d route repeats expert %d",layer,first);break;
+ case Q38_EXPERT_BATCH_FALLBACK_LAYOUT:
+ fprintf(stderr,"layer %d expert %d is not native block-FP8",layer,first);break;
+ default:
+ fprintf(stderr,"unknown reason");break;
+ }
+ fprintf(stderr,"; using serial expert reads\n");
+ return 0;
+}
+
static int q38_expert_get_batch(Model *m,int layer,const int *experts,int count,
Slot **selected) {
- if(!m->expert_parallel_reads||!experts||!selected||count<2||
- count>Q38_MAX_TOPK)return 0;
+ if(!experts||!selected||count<2)return 0;
+ if(!m->expert_parallel_reads)
+ return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_DISABLED,layer,0,0);
LCache *cache=&m->cache[layer];
- if(cache->capcache->cap)
+ return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_CACHE_CAPACITY,
+ layer,cache->cap,count);
+ if(!q38_prepare_expert_scale_bank(m,layer))
+ return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_SCALE_BANK,layer,0,0);
+ /* The demand set is no longer bounded by the decode top-k: the MoE prefill
+ * hands over the whole chunk union (up to the cache cap) so its loads run
+ * one OMP wave instead of serial groups of Q38_MAX_TOPK. Load grouping
+ * never touches FP order: routed outputs are written per assignment and
+ * the per-position expert sum follows the router order, so a bigger wave
+ * only changes WHICH slots serve the reads, not the arithmetic. */
+ Q38ExpertLoadJob *jobs=malloc((size_t)count*sizeof(*jobs));
+ if(!jobs)return 0;
for(int index=0;index=m->c.experts)return 0;
+ if(expert<0||expert>=m->c.experts){free(jobs);return 0;}
q38_ehit_mark(m,layer,expert);
for(int previous=0;previousby_expert[expert];
if(slot_index>=0){
- if(slot_index>=cache->n||cache->slots[slot_index].eid!=expert)return 0;
+ if(slot_index>=cache->n||cache->slots[slot_index].eid!=expert){free(jobs);return 0;}
continue;
}
st_tensor *weight[3];
- if(!q38_native_fp8_expert_tensors(m,layer,expert,weight))return 0;
+ if(!q38_native_fp8_expert_tensors(m,layer,expert,weight)){
+ free(jobs);
+ return q38_expert_batch_fallback(m,Q38_EXPERT_BATCH_FALLBACK_LAYOUT,
+ layer,expert,0);
+ }
}
unsigned char *protected_slots=(unsigned char*)calloc((size_t)cache->cap,1);
if(!protected_slots){fprintf(stderr,"OOM expert batch reservations\n");exit(1);}
@@ -1390,6 +1519,7 @@ static int q38_expert_get_batch(Model *m,int layer,const int *experts,int count,
cache->by_expert[jobs[job].expert]=(int)(slot-cache->slots);
}
}
+ free(jobs);
return 1;
}
@@ -1519,17 +1649,59 @@ static void q38_ple(Model *m,const int *ids,int S,const float *hyper,float *out)
q38_tm_add(m,Q38_TM_PLE,phase_started);
}
+/* The chunk ceiling and the workspace budget used to be compile-time only. An
+ * isolated prefill measurement (274 tokens, one forward) showed that the
+ * ceiling -- not the budget -- is what binds: 32 rows cut the prompt into nine
+ * chunks, each chunk touches ~91 distinct experts, so a loaded expert serves
+ * ~3.3 rows. That drags 4.69 MiB of FP8 weights in for three rows of
+ * activations, which is decode-grade arithmetic intensity inside a path that is
+ * supposed to be batched, and it shows: 1.29 TFLOP in 48.8 s is 26.5 GFLOP/s,
+ * a few percent of what the cores can do. Both values are therefore runtime
+ * knobs now. Widening the chunk cannot change any result -- boundaries alter
+ * neither routing nor accumulation order -- so this is a pure A/B. */
+static int q38_env_positive_int(const char *name,int default_value,
+ int max_value) {
+ const char *value=getenv(name);
+ if(!value||!*value)return default_value;
+ char *end=NULL;long parsed=strtol(value,&end,10);
+ if(end==value||*end||parsed<1||parsed>(long)max_value){
+ fprintf(stderr,"%s must be an integer in 1..%d\n",name,max_value);
+ exit(1);
+ }
+ return (int)parsed;
+}
+
+static int q38_prefill_batch_rows(void) {
+ static int cached=0;
+ if(!cached)
+ cached=q38_env_positive_int("Q38_PREFILL_BATCH_ROWS",
+ Q38_PREFILL_BATCH_ROWS,1<<20);
+ return cached;
+}
+
+/* Expressed in MiB because the byte count is the thing a human gets wrong. */
+static uint64_t q38_prefill_workspace_bytes(void) {
+ static uint64_t cached=0;
+ if(!cached)
+ cached=(uint64_t)q38_env_positive_int(
+ "Q38_PREFILL_WORKSPACE_MIB",
+ (int)(Q38_PREFILL_WORKSPACE_BYTES>>20),4096)<<20;
+ return cached;
+}
+
/* Choose a context-independent prefill chunk whose private workspace fits the
* common target. Callers provide exact fixed and per-row byte counts; even a
* hostile-but-valid geometry gets one row rather than an unbounded allocation. */
static int q38_bounded_prefill_rows(int requested,uint64_t fixed,
uint64_t per_row) {
- int rows=requested1;rows--)
if(per_row<=UINT64_MAX/(uint64_t)rows&&
fixed<=UINT64_MAX-per_row*(uint64_t)rows&&
- fixed+per_row*(uint64_t)rows<=Q38_PREFILL_WORKSPACE_BYTES)
+ fixed+per_row*(uint64_t)rows<=budget)
return rows;
return 1;
}
@@ -1655,6 +1827,11 @@ static void q38_attention(Model *m,Layer *l,int layer,const float *x,int S,int p
float *ip=falloc((int64_t)S*(IQ+c->idx_kheads)*ID);
q38_dense_matmul(m,qp,x,&l->q,S,H,QH*2*D);q38_dense_matmul(m,kp,x,&l->k,S,H,KVH*D);q38_dense_matmul(m,vp,x,&l->v,S,H,KVH*D);
q38_dense_matmul(m,ip,x,&l->idx_qk,S,H,(IQ+c->idx_kheads)*ID);
+ /* Cause before parallelism: the K/V/IK writes are disjoint per position
+ * (each s writes only its own row) and must be complete before the
+ * ranking, which reads the whole IK[0..pos] prefix. Guarded on S>1 so
+ * decode keeps the serial path it has today. */
+ #pragma omp parallel for schedule(static) if(S>1)
for(int s=0;sIK[layer]+(int64_t)pos*ID,ip+(int64_t)s*(IQ+1)*ID+(int64_t)IQ*ID,(size_t)ID*sizeof(float));
}
- float *heads=falloc((int64_t)S*QH*D),*qidx=falloc((int64_t)IQ*ID),*pool=falloc(ID);
- int *selected=(int*)malloc((size_t)maxsel*sizeof(int));
- if(!selected){fprintf(stderr,"OOM QSA selection\n");exit(1);}
+ float *heads=falloc((int64_t)S*QH*D);
+ /* Ranking and attention are independent per position: no shared writes
+ * (heads is row-disjoint, the scratch is per-thread) and no FP order
+ * changes inside a position, so the result is bit-identical to the
+ * serial path. The scheduling is dynamic because the ranking cost grows
+ * with the position (the IK prefix to read is O(pos)). */
+ double index_dt=0,attn_dt=0;
+ /* Wall, not aggregate CPU: the reduction below sums per-thread seconds, so
+ * at prefill these two phases would report ~20x what the clock saw while
+ * every other phase reports wall -- on a 3006-token prompt the phase sum
+ * came to 510 s against a 268 s TTFT, the parts outweighing the whole.
+ * Take the clock across the whole team and split it by CPU share. At
+ * decode S==1 the loop is serial, cpu_total equals the wall, and the
+ * rescale below is an exact no-op. */
+ double qsa_wall_started=now_s();
+ #pragma omp parallel for schedule(dynamic,8) reduction(+:index_dt,attn_dt) if(S>1)
for(int s=0;sidx_qn,ID,c->eps);q38_rope(qh,ID,c->rotary_dim,pos,c->theta);}
int take=blocksidx_budget/R?blocks:c->idx_budget/R,nsel=0;
Q38Block *rank=blocks?(Q38Block*)malloc((size_t)blocks*sizeof(Q38Block)):NULL;
@@ -1682,7 +1875,7 @@ static void q38_attention(Model *m,Layer *l,int layer,const float *x,int S,int p
if(blocks)qsort(rank,(size_t)blocks,sizeof(Q38Block),q38_block_desc);
for(int z=0;zqn,D,c->eps);q38_rope(qh,D,c->rotary_dim,pos,c->theta);
@@ -1693,10 +1886,18 @@ static void q38_attention(Model *m,Layer *l,int layer,const float *x,int S,int p
for(int j=0;jV[layer]+((int64_t)khidx*m->kv_cap+selected[j])*D;for(int d=0;d0.0){
+ index_dt=qsa_wall*(index_dt/cpu_total);
+ attn_dt =qsa_wall*(attn_dt /cpu_total);
}
+ m->timers.seconds[Q38_TM_QSA_INDEX]+=index_dt;
+ m->timers.seconds[Q38_TM_QSA_ATTENTION]+=attn_dt;
q38_dense_matmul(m,out,heads,&l->o,S,QH*D,H);
- free(qp);free(kp);free(vp);free(ip);free(heads);free(qidx);free(pool);free(selected);
+ free(qp);free(kp);free(vp);free(ip);free(heads);
}
/* The single-row path is intentionally kept separate from prefill. Decode is
@@ -1725,15 +1926,16 @@ static void q38_tier_note(int layer,int eid,const Slot *ex) {
* stays for prefill and as fallback. Q38_TRUNK_GPU=0 keeps the trunk on the
* CPU (parity runs against the BF16 reference). */
typedef struct { Q38Weight *w; char name[16]; int layer; } Q38TrunkItem;
-static Q38TrunkItem *g_trunk; static int g_trunk_n, g_trunk_cap;
+static Q38TrunkItem *g_trunk; static int g_trunk_n, g_trunk_cap, g_trunk_offer_gpu;
+static long g_trunk_min_kb; static const char *g_trunk_skip; /* read once per load in q38_trunk_offer_all */
static void q38_trunk_add(Q38Weight *w,const char *name,int layer) {
if(!w||!w->data||(w->kind!=Q38_WEIGHT_BF16&&w->kind!=Q38_WEIGHT_F32))return;
size_t bytes=(size_t)w->rows*w->cols+(size_t)w->rows*sizeof(float);
/* Q38_TRUNK_MIN_KB (default 1024): a round trip costs more than a tiny
- * GEMV saves; Q38_TRUNK_SKIP=name,name: leave those components on the
- * CPU (bisecting a numeric difference, or a component that does not pay) */
- static long min_kb=-1; static const char *skip;
- if(min_kb<0){ const char *e=getenv("Q38_TRUNK_MIN_KB"); min_kb=e?atol(e):1024; skip=getenv("Q38_TRUNK_SKIP"); }
+ * GEMV saves, and a tiny matrix in BF16 costs nothing on the CPU either;
+ * Q38_TRUNK_SKIP=name,name: leave those components in BF16 on the CPU
+ * (bisecting a numeric difference, or a component that does not pay) */
+ long min_kb=g_trunk_min_kb; const char *skip=g_trunk_skip;
if(bytes<(size_t)min_kb*1024)return;
if(skip&&*skip){
size_t n=strlen(name); const char *s=skip;
@@ -1747,15 +1949,19 @@ static void q38_trunk_add(Q38Weight *w,const char *name,int layer) {
}
Q38TrunkItem *it=&g_trunk[g_trunk_n++]; it->w=w; it->layer=layer;
snprintf(it->name,sizeof it->name,"%s",name);
- qt_trunk_offer(it->name,layer,bytes);
+ if(g_trunk_offer_gpu)qt_trunk_offer(it->name,layer,bytes);
}
static int q38_trunk_enabled(void) {
const char *e=getenv("Q38_TRUNK_GPU"); return !(e&&e[0]=='0'&&!e[1]);
}
-/* offers, before qt_init: lm_head first (the placer takes it first), then the
- * layers in order so a partial placement is a prefix of the layers */
+/* the trunk table, built once: lm_head first (the placer takes it first),
+ * then the layers in order so a partial placement is a prefix of the layers.
+ * The same table feeds the CPU's int8 rows; the placer is told about the
+ * matrices only when the GPU trunk is enabled (Q38_TRUNK_GPU). */
static void q38_trunk_offer_all(Model *m) {
- if(!q38_trunk_enabled())return;
+ g_trunk_n=0; m->trunk_table_built=1; /* rebuilt per load: a test opens several models in one process */
+ g_trunk_offer_gpu=q38_trunk_enabled();
+ { const char *e=getenv("Q38_TRUNK_MIN_KB"); g_trunk_min_kb=e?atol(e):1024; g_trunk_skip=getenv("Q38_TRUNK_SKIP"); }
Cfg *c=&m->c;
q38_trunk_add(&m->lm_head,"lmhead",0);
for(int l=0;llayers;l++){
@@ -1794,19 +2000,33 @@ static void q38_trunk_quantize(const Q38Weight *w,int8_t **qp,float **scp) {
}
*qp=q; *scp=sc;
}
-/* Q38_TRUNK_CPU_INT8=1: keep the int8 rows on the CPU instead (or as well),
- * so the quantization can be judged without a GPU (PPL, token parity) */
+/* The trunk on the CPU: int8 rows with one scale per row, and the BF16 copy
+ * released. The trunk is read whole on every token (3.6 G weights on the
+ * released checkpoint, more than the ten routed experts), so its bytes are
+ * the decode's floor: int8 halves them and the integer kernel keeps up with
+ * the memory. Default on; Q38_TRUNK_CPU_INT8=0 keeps the BF16 rows and the
+ * f32 kernel, the numeric reference. A matrix the tier already quantized for
+ * the GPU keeps those same rows here (prefill rows run on the CPU). */
+static int q38_trunk_cpu_int8_wanted(void) {
+ const char *e=getenv("Q38_TRUNK_CPU_INT8"); return !(e&&e[0]=='0'&&!e[1]);
+}
static void q38_trunk_cpu_int8(Model *m) {
- const char *e=getenv("Q38_TRUNK_CPU_INT8");
- if(!e||e[0]!='1'||e[1])return;
- if(!g_trunk_n) q38_trunk_offer_all(m);
- double t0=now_s(); size_t bytes=0;
+ if(!q38_trunk_cpu_int8_wanted())return;
+ if(!m->trunk_table_built)q38_trunk_offer_all(m); /* the tier may have built it already */
+ double t0=now_s(); size_t bytes=0,released=0; int n=0;
for(int i=0;iq8)continue;
- q38_trunk_quantize(w,&w->q8,&w->q8sc); bytes+=(size_t)w->rows*w->cols;
+ Q38Weight *w=g_trunk[i].w;
+ if(!w->q8)q38_trunk_quantize(w,&w->q8,&w->q8sc);
+ bytes+=(size_t)w->rows*w->cols+(size_t)w->rows*sizeof(float); n++;
+ if(w->owns_data&&w->data){
+ /* every path that reads this matrix now goes through q8 */
+ uint64_t was=q38_weight_bytes(w);
+ free(w->data); w->data=NULL; w->owns_data=0; released+=was;
+ m->resident_weight_bytes-=was; m->resident_weight_bytes+=(size_t)w->rows*w->cols+(size_t)w->rows*sizeof(float);
+ }
}
- fprintf(stderr,"[qwen38] trunk: %d matrices int8 on the CPU (%.2f GiB) in %.1fs (Q38_TRUNK_CPU_INT8)\n",
- g_trunk_n,bytes/1073741824.0,now_s()-t0);
+ fprintf(stderr,"[qwen38] trunk: %d matrices int8 on the CPU (%.2f GiB, %.2f GiB of BF16 released) in %.1fs; Q38_TRUNK_CPU_INT8=0 keeps BF16\n",
+ n,bytes/1073741824.0,released/1073741824.0,now_s()-t0);
}
/* after qt_init: quantize and upload what the placer accepted */
static void q38_trunk_place_all(Model *m) {
@@ -1817,7 +2037,8 @@ static void q38_trunk_place_all(Model *m) {
int dev=qt_place_of(it->name,it->layer);
if(dev==QT_PLACE_CPU)continue;
int O=w->rows,I=w->cols;
- int8_t *q; float *sc; q38_trunk_quantize(w,&q,&sc);
+ if(!w->q8)q38_trunk_quantize(w,&w->q8,&w->q8sc); /* kept: the CPU answers prefill rows from the same bytes */
+ const int8_t *q=w->q8; const float *sc=w->q8sc;
int h=qt_dense_init(q,sc,I,O,dev);
if(h>=0&&getenv("Q38_TRUNK_SELFTEST")){
/* DIAG: GPU int8 GEMV against the same int8 matrix on the CPU */
@@ -1829,11 +2050,10 @@ static void q38_trunk_place_all(Model *m) {
fprintf(stderr,"[selftest] %-7s L%-2d [O=%d I=%d] ok=%d rel.err %.2e worst row %d gpu %.5g cpu %.5g\n",it->name,it->layer,O,I,ok,den>0?sqrt(num/den):-1.0,worst,yg[worst],yc[worst]);
free(x);free(yg);free(yc);
}
- free(q); free(sc);
if(h>=0){ w->gpu=h+1; placed++; placed_bytes+=(size_t)O*I; }
}
if(g_trunk_n)
- fprintf(stderr,"[qtier] qwen38 trunk: %d of %d offered matrices resident as int8 (%.2f GiB) in %.1fs; the rest stays BF16 on the CPU\n",
+ fprintf(stderr,"[qtier] qwen38 trunk: %d of %d offered matrices resident as int8 (%.2f GiB) in %.1fs; the rest answers from the CPU\n",
placed,g_trunk_n,placed_bytes/1073741824.0,now_s()-t0);
}
@@ -1900,7 +2120,10 @@ static void q38_moe_decode(Model *m,Layer *l,int layer,const float *x,int S,floa
}
/* GPU experts land after the CPU ones: same values, one more group in
* the float sum (that is the only ordering difference to a CPU run). */
- qt_take(qmask,route_gates,K,ys);
+ if(!qt_take(qmask,route_gates,K,ys)){
+ fprintf(stderr,"qwen38: CUDA expert collection failed at layer %d; stopping inference\n",layer);
+ exit(1);
+ }
for(int d=0;dcache[layer].cap;
- if(load_limit>Q38_MAX_TOPK)load_limit=Q38_MAX_TOPK;
if(load_limit<1)load_limit=1;
for(int unique_base=0;unique_basekvp);
Cfg *c=&m->c;
for(int i=0;ilayers;i++)if(!c->is_attn[i]){
memset(m->DN_rec[i],0,(size_t)c->dn_vheads*c->dn_kdim*c->dn_vdim*sizeof(float));
@@ -2146,6 +2369,7 @@ static void ensure_kv(Model *m) {
m->K[i]=falloc((int64_t)c->kv_heads*m->max_t*c->head_dim);m->V[i]=falloc((int64_t)c->kv_heads*m->max_t*c->head_dim);m->IK[i]=falloc((int64_t)m->max_t*c->idx_dim);
}
m->kv_cap=m->max_t;
+ kv_prefix_alloc(&m->kvp,m->kv_cap); /* ensure_kv discards the old rows */
}
/* Run only the requested native layer interval over hyper-residual activations.
@@ -2226,6 +2450,10 @@ static float *step(Model *m,const int *ids,int S,int pos_base) {
q38_gr_read(m,&l->mlp_gr,hyper,S,mixed,inject);q38_moe(m,l,i,mixed,S,block);q38_gr_apply(c,hyper,block,inject,S);
}
q38_gr_read(m,&m->final_gr,hyper,S,mixed,NULL);m->kv_len=pos_base+S;
+ /* Rewinding and writing a shorter branch invalidates its old tail. */
+ if(m->kvp.len>pos_base)m->kvp.len=pos_base;
+ kv_prefix_record(&m->kvp,ids,pos_base,S);
+ if(m->vis_map && m->vis_rows_n>0)kv_prefix_taint(&m->kvp);
float *logit=falloc(c->vocab);double phase_started=now_s();
/* Lettura del prefill: la posizione p predice il token p+1. Il primo token
* fresco lo predice la fotografia del prefisso, quando c'e. Pagata solo da
@@ -2310,6 +2538,7 @@ static void q38_layer_free(Layer *l) {
static void q38_model_free(Model *m) {
if(!m) return;
+ kv_prefix_free(&m->kvp);
for(int i=0;ic.layers;i++) {
if(m->L)q38_layer_free(&m->L[i]);
if(m->cache) {
diff --git a/c/resource_plan.py b/c/resource_plan.py
index 18087a585..955f29a77 100644
--- a/c/resource_plan.py
+++ b/c/resource_plan.py
@@ -434,6 +434,8 @@ def read_ssd_probe(model_dir):
def discover_gpus():
+ if sys.platform == "darwin":
+ return _discover_metal_gpus()
# NVIDIA first; if there are none (or no nvidia-smi), fall back to ROCm/HIP so
# a working AMD engine isn't planned CPU-only and --gpu N stops failing (#662).
devices = _discover_nvidia_gpus()
@@ -442,6 +444,33 @@ def discover_gpus():
return _discover_amd_gpus()
+def _discover_metal_gpus():
+ """Return Apple Metal devices without pretending unified RAM is VRAM."""
+ try:
+ result = subprocess.run(
+ ["system_profiler", "SPDisplaysDataType", "-json"],
+ text=True, capture_output=True, check=True, timeout=10)
+ displays = json.loads(result.stdout).get("SPDisplaysDataType", [])
+ except (OSError, subprocess.SubprocessError, ValueError, TypeError):
+ return []
+ devices = []
+ for index, display in enumerate(displays):
+ if not isinstance(display, dict):
+ continue
+ metal = display.get("spdisplays_mtlgpufamilysupport")
+ if not metal:
+ continue
+ name = display.get("sppci_model") or display.get("_name")
+ if not isinstance(name, str) or not name:
+ continue
+ # Apple Silicon has one unified pool. Its free capacity cannot be used
+ # as an independent VRAM budget, so retain device identity only.
+ devices.append({"index": index, "name": name,
+ "total_bytes": 0, "free_bytes": None,
+ "unified_memory": True, "backend": "metal"})
+ return devices
+
+
def _discover_nvidia_gpus():
command = ["nvidia-smi", "--query-gpu=index,name,memory.total,memory.free",
"--format=csv,noheader,nounits"]
@@ -934,6 +963,26 @@ def _next_actions(bottleneck_class, projected_hit, probe_state, probe_gbs,
}
+def _family_expert_cache_knob(family_id, cache_bytes):
+ """Env var that actually sizes the expert LRU, when it is not RAM_GB.
+
+ Kimi K3 reads K3_EXPERT_GB (default 8) and treats RAM_GB as a ceiling;
+ main() never takes argv as a cap. GLM-5.3 reads GLM53_EXPERT_GB and never
+ RAM_GB. Exporting only RAM_GB therefore left both engines on their own
+ default cache under --auto-tier.
+ """
+ if (not isinstance(cache_bytes, int) or isinstance(cache_bytes, bool)
+ or cache_bytes <= 0):
+ return None
+ if family_id == "kimi":
+ return ("K3_EXPERT_GB",
+ "Kimi K3 sizes the expert LRU from K3_EXPERT_GB; RAM_GB is only a ceiling")
+ if family_id == "glm53":
+ return ("GLM53_EXPERT_GB",
+ "GLM-5.3 sizes the expert LRU from GLM53_EXPERT_GB, not RAM_GB")
+ return None
+
+
def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0,
available_memory=None, available_disk=None, gpus=None,
policy="quality", physical_cpus=None, cpu_sockets=None,
@@ -1100,6 +1149,13 @@ def build_plan(model, ram_gb=0, context=4096, gpu_indices=None, vram_gb=0,
tune = _auto_tune(bottleneck_class, projected_hit, planning_gpus, cpu_sockets,
plan_has_metal=False,
engine_group=resolved.descriptor.engine_group)
+ # DRAFT/PIPE/PIN/NUMA stay colibri.c-only (#1585). These two are the
+ # opposite leftover: the engines that size the expert LRU from their own
+ # *EXPERT_GB variable, which RAM_GB does not set.
+ knob = _family_expert_cache_knob(resolved.descriptor.id, cache_bytes)
+ if knob:
+ name, reason = knob
+ tune[name] = {"value": f"{cache_bytes / GB:.3f}", "reason": reason}
probe_state, probe_gbs = ssd_probe_state(info["path"])
actions = _next_actions(bottleneck_class, projected_hit, probe_state,
probe_gbs, planning_gpus)
@@ -1184,6 +1240,10 @@ def environment_for_plan(plan, env=None, cuda_enabled=True):
result.setdefault("REPIN", "64")
ram = plan["tiers"]["ram"]
result.setdefault("RAM_GB", f"{ram['budget_bytes'] / GB:.3f}")
+ knob = _family_expert_cache_knob(plan.get("model", {}).get("family_id"),
+ ram.get("expert_cache_bytes"))
+ if knob:
+ result.setdefault(knob[0], f"{ram['expert_cache_bytes'] / GB:.3f}")
planned_cap = ram.get("cache_slots_per_layer")
if (plan.get("model", {}).get("family_id") == "qwen38" and
(not isinstance(planned_cap, int) or
@@ -1241,14 +1301,22 @@ def format_plan(plan):
f"cap {tiers['ram']['cache_slots_per_layer']}/layer"]
vram = tiers["vram"]
if vram["devices"]:
- names = ", ".join(
- f"{gpu['index']}:{gpu['name']}"
- + ("" if plans_placement(gpu) else " (identity only)")
- for gpu in vram["devices"])
- trunk = vram.get("trunk_bytes", 0)
- lines.append("VRAM " + (f"{format_bytes(trunk)} int8 trunk + " if trunk else "") +
- f"{format_bytes(vram['budget_bytes'])} hot tier · "
- f"~{vram['expert_capacity']} experts · {names}")
+ metal_identity = [gpu for gpu in vram["devices"]
+ if gpu.get("backend") == "metal"
+ and gpu.get("free_bytes") is None]
+ if len(metal_identity) == len(vram["devices"]):
+ names = ", ".join(f"{gpu['index']}:{gpu['name']}"
+ for gpu in metal_identity)
+ lines.append(f"Metal {names} · unified memory · no independent VRAM budget")
+ else:
+ names = ", ".join(
+ f"{gpu['index']}:{gpu['name']}"
+ + ("" if plans_placement(gpu) else " (identity only)")
+ for gpu in vram["devices"])
+ trunk = vram.get("trunk_bytes", 0)
+ lines.append("VRAM " + (f"{format_bytes(trunk)} int8 trunk + " if trunk else "") +
+ f"{format_bytes(vram['budget_bytes'])} hot tier · "
+ f"~{vram['expert_capacity']} experts · {names}")
else:
# Backend-neutral, matching the accelerator wording #903 settled on:
# an AMD or Intel host that finds nothing is not "no NVIDIA device".
diff --git a/c/serve_budget.h b/c/serve_budget.h
new file mode 100644
index 000000000..36c8ae507
--- /dev/null
+++ b/c/serve_budget.h
@@ -0,0 +1,21 @@
+#ifndef COLI_SERVE_BUDGET_H
+#define COLI_SERVE_BUDGET_H
+
+/* max_tokens is a ceiling, not a target (#260/#382/#1641).
+ * Returns the tokens the request may generate, or -1 when the PROMPT does
+ * not fit. Generation needs one free position; a read-only logprobs request
+ * may fill the context exactly. Refusing when prompt + budget exceeded the
+ * context turned coli chat's interactive default (16384) into a 400 on every
+ * Kimi/Inkling message against the default 8192-token window. */
+
+static inline int coli_serve_budget(int prompt, int requested, int context,
+ int read_only)
+{
+ if (prompt < 1) return -1;
+ int room = context - prompt;
+ if (read_only) return room < 0 ? -1 : (requested < room ? requested : room);
+ if (room < 1) return -1;
+ return requested > room ? room : requested;
+}
+
+#endif
diff --git a/c/sse41_kernels.h b/c/sse41_kernels.h
new file mode 100644
index 000000000..4f93eee40
--- /dev/null
+++ b/c/sse41_kernels.h
@@ -0,0 +1,162 @@
+#ifndef COLIBRI_SSE41_KERNELS_H
+#define COLIBRI_SSE41_KERNELS_H
+/*
+ * sse41_kernels.h — shared SSE 4.1 primitives for Colibri engines.
+ *
+ * The engines (c/deepseek_v4.c, c/colibri.c, c/kimi_k3.c, c/olmoe.c, c/inkling.c)
+ * have an `#if defined(__AVX2__)` dispatch for fast paths and fall through to scalar
+ * on pre-Haswell hardware (Sandy Bridge: AVX 1.0, no FMA, no AVX-2). This header
+ * provides the missing middle tier: 128-bit SIMD primitives, FMA-free.
+ *
+ * Why a separate header (not just inline in each .c):
+ * - 109 AVX2 sites total across 5 engines need patching. Copy-pasting the
+ * 128-bit intrinsics 109 times is a typo factory. A single macro definition
+ * is the difference between correct and wrong-on-100-sites.
+ * - The single most critical shared piece is the FMA-emulation macro: on
+ * Sandy Bridge, _mm_mul_ps + _mm_add_ps has double rounding vs. hardware
+ * FMA's single rounding, so the output is NOT bit-identical to AVX2 (1-2 ULP
+ * difference). A typo in a single copy is a silent correctness bug.
+ *
+ * This is the minimum needed for the SSE 4.1 fallback. More primitives can be
+ * added as additional engines are patched.
+ */
+#if defined(__SSE2__)
+
+#include
+#include
+
+/*
+ * COLIBRI_FMA: emulate FMA on non-FMA hardware.
+ *
+ * On FMA hardware: maps to _mm_fmadd_ps (single rounding, 1 instruction).
+ * On Sandy Bridge (no FMA): separate mul+add, double rounding, 2 instructions.
+ *
+ * On Sandy Bridge this is NOT bit-identical to AVX2/FMA output — typically
+ * within 1-2 ULP. Test tolerance must accommodate this.
+ */
+#if defined(__FMA__)
+# define COLIBRI_FMA(a, b, c) _mm_fmadd_ps((a), (b), (c))
+#else
+# define COLIBRI_FMA(a, b, c) _mm_add_ps(_mm_mul_ps((a), (b)), (c))
+#endif
+
+/*
+ * SSE 4.1 (or lower) load/store helpers. Sandy Bridge has these natively.
+ * _mm_load_ps is aligned; _mm_loadu_ps is unaligned. For 128-bit (16-byte)
+ * data, aligned loads are faster but UB on misaligned pointers. Default to
+ * unaligned: buffers from malloc / numa_slab_bind have no 16-byte guarantee.
+ * Aligned loads can be added as a profiled follow-up.
+ */
+static inline __m128 colibri_sse41_loadu_ps(const float *p) { return _mm_loadu_ps(p); }
+static inline void colibri_sse41_storeu_ps(float *p, __m128 v) { _mm_storeu_ps(p, v); }
+
+/*
+ * Min/max (SSE 4.1 native). Identical to AVX2, just narrower width.
+ */
+static inline __m128 colibri_sse41_min_ps(__m128 a, __m128 b) { return _mm_min_ps(a, b); }
+static inline __m128 colibri_sse41_max_ps(__m128 a, __m128 b) { return _mm_max_ps(a, b); }
+
+/*
+ * Prefetch (SSE 1+, always available). Identical to AVX2/FMA path.
+ */
+static inline void colibri_sse41_prefetch(const void *p) { _mm_prefetch(p, _MM_HINT_T0); }
+
+/* Grouped-int4 kernels for the SSE4.1 tier. Callers select even group sizes
+ * and pass a multiple of four output rows; scalar dispatch handles the rest.
+ * Keep the scalar pair-sum, scale-multiply and accumulator-add order intact. */
+#if defined(__SSE4_1__) && !defined(__AVX2__)
+/* Load one packed byte from each of four output rows, then unpack their low and
+ * high offset nibbles into four f32 lanes. Keeping independent output rows in
+ * the lanes preserves the scalar operation order within every row. */
+static inline void colibri_sse41_i4_rows4(const uint8_t *q4,int rb,int o,int byte,
+ __m128 *lo,__m128 *hi){
+ const __m128i m4=_mm_set1_epi8(0x0F), b8=_mm_set1_epi8(8);
+ /* Read exactly one byte per row, including when a row is one byte wide. */
+ __m128i by=_mm_cvtsi32_si128(q4[(int64_t)(o+0)*rb+byte]);
+ by=_mm_insert_epi8(by,q4[(int64_t)(o+1)*rb+byte],1);
+ by=_mm_insert_epi8(by,q4[(int64_t)(o+2)*rb+byte],2);
+ by=_mm_insert_epi8(by,q4[(int64_t)(o+3)*rb+byte],3);
+ __m128i qlo=_mm_sub_epi8(_mm_and_si128(by,m4),b8);
+ __m128i qhi=_mm_sub_epi8(_mm_and_si128(_mm_srli_epi16(by,4),m4),b8);
+ *lo=_mm_cvtepi32_ps(_mm_cvtepi8_epi32(qlo));
+ *hi=_mm_cvtepi32_ps(_mm_cvtepi8_epi32(qhi));
+}
+
+static inline __m128 colibri_sse41_f32_rows4(const float *p,int stride,int o,int i){
+ return _mm_set_ps(p[(int64_t)(o+3)*stride+i],p[(int64_t)(o+2)*stride+i],
+ p[(int64_t)(o+1)*stride+i],p[(int64_t)(o+0)*stride+i]);
+}
+
+/* Process four output rows at once without a horizontal reduction. Each lane
+ * uses the scalar kernel's pair sum, scale multiply, and accumulator add in
+ * the same order, so the result can remain byte-identical on pre-FMA CPUs. */
+static void matmul_i4_grouped_sse41_rows4(float *y,const float *x,
+ const uint8_t *q4,const float *scale,
+ int S,int I,int O,int gs,int rb,int ng,
+ int o4){
+ #pragma omp parallel for schedule(static)
+ for(int o=0;oI) end=I;
+ __m128 sc=colibri_sse41_f32_rows4(scale,ng,o,g); int i=base;
+ for(;i+1>1,&lo,&hi);
+ __m128 pair=_mm_add_ps(_mm_mul_ps(_mm_set1_ps(xs[i]),lo),
+ _mm_mul_ps(_mm_set1_ps(xs[i+1]),hi));
+ a=_mm_add_ps(a,_mm_mul_ps(pair,sc));
+ }
+ if(i>1,&lo,&hi); (void)hi;
+ a=_mm_add_ps(a,_mm_mul_ps(_mm_mul_ps(_mm_set1_ps(xs[i]),lo),sc));
+ }
+ }
+ colibri_sse41_storeu_ps(y+(int64_t)s*O+o,a);
+ }
+ }
+}
+
+/* Fused gate/up shares activation loads while keeping independent accumulators. */
+static void matmul_i4_grouped_pair_sse41_rows4(float *yg,float *yu,const float *x,
+ const uint8_t *qg,const float *sg,
+ const uint8_t *qu,const float *su,
+ int S,int I,int O,int gs,int rb,
+ int ng,int o4){
+ #pragma omp parallel for schedule(static)
+ for(int o=0;oI) end=I;
+ __m128 scg=colibri_sse41_f32_rows4(sg,ng,o,g);
+ __m128 scu=colibri_sse41_f32_rows4(su,ng,o,g); int i=base;
+ for(;i+1>1,&gl,&gh);
+ colibri_sse41_i4_rows4(qu,rb,o,i>>1,&ul,&uh);
+ __m128 x0=_mm_set1_ps(xs[i]),x1=_mm_set1_ps(xs[i+1]);
+ __m128 gp=_mm_add_ps(_mm_mul_ps(x0,gl),_mm_mul_ps(x1,gh));
+ __m128 up=_mm_add_ps(_mm_mul_ps(x0,ul),_mm_mul_ps(x1,uh));
+ ag=_mm_add_ps(ag,_mm_mul_ps(gp,scg));
+ au=_mm_add_ps(au,_mm_mul_ps(up,scu));
+ }
+ if(i>1,&gl,&gh);
+ colibri_sse41_i4_rows4(qu,rb,o,i>>1,&ul,&uh); (void)gh; (void)uh;
+ __m128 xi=_mm_set1_ps(xs[i]);
+ ag=_mm_add_ps(ag,_mm_mul_ps(_mm_mul_ps(xi,gl),scg));
+ au=_mm_add_ps(au,_mm_mul_ps(_mm_mul_ps(xi,ul),scu));
+ }
+ }
+ colibri_sse41_storeu_ps(yg+(int64_t)s*O+o,ag);
+ colibri_sse41_storeu_ps(yu+(int64_t)s*O+o,au);
+ }
+ }
+}
+#endif /* __SSE4_1__ && !__AVX2__ */
+
+#endif /* __SSE2__ */
+#endif /* COLIBRI_SSE41_KERNELS_H */
diff --git a/c/st.h b/c/st.h
index 8ad5ffd3b..286cfd754 100644
--- a/c/st.h
+++ b/c/st.h
@@ -138,6 +138,121 @@ static inline float f16_to_f32(uint16_t h) {
float f; memcpy(&f, &u, 4); return f;
}
+/* ---- bulk BF16/F16 -> F32, AVX2/SSE4.1/scalar tiers --------------------
+ * st_read_f32/st_read_slice_f32 convert whole tensors (up to the embed/
+ * lm_head matrix, vocab*hidden elements) through bf16_to_f32/f16_to_f32 one
+ * halfword at a time; that loop is pure overhead once the read syscall is
+ * off the critical path. The tiers below batch it, same idea and layering
+ * as gsgemv.h's AVX2/SSE4.1 kernels (immintrin.h only under the ISA guard,
+ * sse41_kernels.h not needed here -- no FMA, no shared load helper to reuse).
+ *
+ * BF16 -> F32 is an exact zero-pad widening for every bit pattern (BF16
+ * shares F32's 8-bit exponent field, so there is no special-casing --
+ * zero, normal, subnormal, inf, NaN all take the same `<<16`), so the
+ * vectorized tiers cannot disagree with the scalar reference.
+ *
+ * F16 -> F32's zero/normal/inf-NaN classes are each the same closed-form
+ * bit algebra as the scalar reference above, just run on several lanes at
+ * once -- no reassociation, nothing to round, so still bit-exact. True
+ * subnormals (exp==0, man!=0) need the scalar reference's shift-to-normalize
+ * loop, which does not vectorize; those (rare in real model weights) fall
+ * back to f16_to_f32 per element. Exhaustive verification (all 65536
+ * patterns per format) lives in tests/test_st_f16_bf16_simd.c. */
+#if defined(__AVX2__) || defined(__SSE4_1__)
+#include
+#endif
+
+#if defined(__AVX2__)
+static void bf16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) {
+ int64_t i = 0;
+ for (; i + 8 <= n; i += 8) {
+ __m128i h = _mm_loadu_si128((const __m128i*)(src + i));
+ __m256i w = _mm256_slli_epi32(_mm256_cvtepu16_epi32(h), 16);
+ _mm256_storeu_ps(dst + i, _mm256_castsi256_ps(w));
+ }
+ for (; i < n; i++) dst[i] = bf16_to_f32(src[i]);
+}
+static void f16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) {
+ int64_t i = 0;
+ const __m256i vsign_mask = _mm256_set1_epi32(0x8000);
+ const __m256i vexp_mask = _mm256_set1_epi32(0x1F);
+ const __m256i vman_mask = _mm256_set1_epi32(0x3FF);
+ const __m256i v112 = _mm256_set1_epi32(112);
+ const __m256i vinfnan_e = _mm256_set1_epi32(0x7F800000);
+ const __m256i vzero = _mm256_setzero_si256();
+ const __m256i v31 = _mm256_set1_epi32(31);
+ for (; i + 8 <= n; i += 8) {
+ __m128i h16 = _mm_loadu_si128((const __m128i*)(src + i));
+ __m256i h = _mm256_cvtepu16_epi32(h16);
+ __m256i sign = _mm256_slli_epi32(_mm256_and_si256(h, vsign_mask), 16);
+ __m256i exp = _mm256_and_si256(_mm256_srli_epi32(h, 10), vexp_mask);
+ __m256i man = _mm256_and_si256(h, vman_mask);
+ __m256i normal_u = _mm256_or_si256(sign, _mm256_or_si256(
+ _mm256_slli_epi32(_mm256_add_epi32(exp, v112), 23), _mm256_slli_epi32(man, 13)));
+ __m256i infnan_u = _mm256_or_si256(sign, _mm256_or_si256(vinfnan_e, _mm256_slli_epi32(man, 13)));
+ __m256i exp_is_zero = _mm256_cmpeq_epi32(exp, vzero);
+ __m256i man_is_zero = _mm256_cmpeq_epi32(man, vzero);
+ __m256i is_zero = _mm256_and_si256(exp_is_zero, man_is_zero);
+ __m256i is_subnorm = _mm256_andnot_si256(man_is_zero, exp_is_zero);
+ __m256i is_infnan = _mm256_cmpeq_epi32(exp, v31);
+ __m256i result = _mm256_blendv_epi8(normal_u, sign, is_zero);
+ result = _mm256_blendv_epi8(result, infnan_u, is_infnan);
+ _mm256_storeu_ps(dst + i, _mm256_castsi256_ps(result));
+ int m = _mm256_movemask_ps(_mm256_castsi256_ps(is_subnorm));
+ if (m) for (int k = 0; k < 8; k++) if ((m >> k) & 1) dst[i+k] = f16_to_f32(src[i+k]);
+ }
+ for (; i < n; i++) dst[i] = f16_to_f32(src[i]);
+}
+#elif defined(__SSE4_1__)
+static void bf16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) {
+ int64_t i = 0;
+ for (; i + 4 <= n; i += 4) {
+ __m128i h = _mm_loadl_epi64((const __m128i*)(src + i));
+ __m128i w = _mm_slli_epi32(_mm_cvtepu16_epi32(h), 16);
+ _mm_storeu_ps(dst + i, _mm_castsi128_ps(w));
+ }
+ for (; i < n; i++) dst[i] = bf16_to_f32(src[i]);
+}
+static void f16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) {
+ int64_t i = 0;
+ const __m128i vsign_mask = _mm_set1_epi32(0x8000);
+ const __m128i vexp_mask = _mm_set1_epi32(0x1F);
+ const __m128i vman_mask = _mm_set1_epi32(0x3FF);
+ const __m128i v112 = _mm_set1_epi32(112);
+ const __m128i vinfnan_e = _mm_set1_epi32(0x7F800000);
+ const __m128i vzero = _mm_setzero_si128();
+ const __m128i v31 = _mm_set1_epi32(31);
+ for (; i + 4 <= n; i += 4) {
+ __m128i h16 = _mm_loadl_epi64((const __m128i*)(src + i));
+ __m128i h = _mm_cvtepu16_epi32(h16);
+ __m128i sign = _mm_slli_epi32(_mm_and_si128(h, vsign_mask), 16);
+ __m128i exp = _mm_and_si128(_mm_srli_epi32(h, 10), vexp_mask);
+ __m128i man = _mm_and_si128(h, vman_mask);
+ __m128i normal_u = _mm_or_si128(sign, _mm_or_si128(
+ _mm_slli_epi32(_mm_add_epi32(exp, v112), 23), _mm_slli_epi32(man, 13)));
+ __m128i infnan_u = _mm_or_si128(sign, _mm_or_si128(vinfnan_e, _mm_slli_epi32(man, 13)));
+ __m128i exp_is_zero = _mm_cmpeq_epi32(exp, vzero);
+ __m128i man_is_zero = _mm_cmpeq_epi32(man, vzero);
+ __m128i is_zero = _mm_and_si128(exp_is_zero, man_is_zero);
+ __m128i is_subnorm = _mm_andnot_si128(man_is_zero, exp_is_zero);
+ __m128i is_infnan = _mm_cmpeq_epi32(exp, v31);
+ __m128i result = _mm_blendv_epi8(normal_u, sign, is_zero);
+ result = _mm_blendv_epi8(result, infnan_u, is_infnan);
+ _mm_storeu_ps(dst + i, _mm_castsi128_ps(result));
+ int m = _mm_movemask_ps(_mm_castsi128_ps(is_subnorm));
+ if (m) for (int k = 0; k < 4; k++) if ((m >> k) & 1) dst[i+k] = f16_to_f32(src[i+k]);
+ }
+ for (; i < n; i++) dst[i] = f16_to_f32(src[i]);
+}
+#else
+static void bf16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) {
+ for (int64_t i = 0; i < n; i++) dst[i] = bf16_to_f32(src[i]);
+}
+static void f16_to_f32_bulk(const uint16_t *src, float *dst, int64_t n) {
+ for (int64_t i = 0; i < n; i++) dst[i] = f16_to_f32(src[i]);
+}
+#endif
+
static int st_open_fd(shards *S, const char *path) {
for (int i = 0; i < S->nfd; i++) if (!strcmp(S->paths[i], path)) return S->fds[i];
int fd = open(path, COMPAT_O_RDONLY);
@@ -449,7 +564,9 @@ static const char *st_basename(const char *p) {
static void st_index_load(st_index *ix, const char *dir) {
if (ix->tried) return;
ix->tried = 1;
- char path[1200]; snprintf(path, sizeof(path), "%s/model.safetensors.index.json", dir);
+ char path[1200];
+ int written = snprintf(path, sizeof(path), "%s/model.safetensors.index.json", dir);
+ if (written < 0 || (size_t)written >= sizeof(path)) return;
FILE *f = fopen(path, "rb");
if (!f) return;
fseek(f, 0, SEEK_END); long size = ftell(f); fseek(f, 0, SEEK_SET);
@@ -954,9 +1071,9 @@ static int64_t st_read_f32(shards *S, const char *name, float *out, int drop) {
if (t->dtype == 2) {
memcpy(out, raw, t->nbytes);
} else if (t->dtype == 0) {
- uint16_t *p = (uint16_t *)raw; for (int64_t i = 0; i < t->numel; i++) out[i] = bf16_to_f32(p[i]);
+ bf16_to_f32_bulk((uint16_t *)raw, out, t->numel);
} else {
- uint16_t *p = (uint16_t *)raw; for (int64_t i = 0; i < t->numel; i++) out[i] = f16_to_f32(p[i]);
+ f16_to_f32_bulk((uint16_t *)raw, out, t->numel);
}
free(raw);
if (drop) posix_fadvise(t->fd, t->off, t->nbytes, POSIX_FADV_DONTNEED);
@@ -1342,8 +1459,8 @@ static void st_read_slice_f32(shards *S, const char *name, int64_t elem_off, int
if (nb) st_pread_full(t->fd, raw, nb, boff, "pread slice"); /* dev #331: chunked + EINTR + honest short-read */
if (nb) {
if (t->dtype == 2) memcpy(out, raw, (size_t)nb);
- else if (t->dtype == 0) { uint16_t *p = raw; for (int64_t i = 0; i < n_elems; i++) out[i] = bf16_to_f32(p[i]); }
- else { uint16_t *p = raw; for (int64_t i = 0; i < n_elems; i++) out[i] = f16_to_f32(p[i]); }
+ else if (t->dtype == 0) bf16_to_f32_bulk((uint16_t *)raw, out, n_elems);
+ else f16_to_f32_bulk((uint16_t *)raw, out, n_elems);
}
free(raw);
if (drop && nb) posix_fadvise(t->fd, boff, nb, POSIX_FADV_DONTNEED);
diff --git a/c/tests/bench_cuda_resident_batch.cu b/c/tests/bench_cuda_resident_batch.cu
new file mode 100644
index 000000000..2f911fbca
--- /dev/null
+++ b/c/tests/bench_cuda_resident_batch.cu
@@ -0,0 +1,80 @@
+/* Bounded resident-int8 projection microbenchmark; no model files required.
+ * Times synchronous host API calls (copies + compute), excluding upload.
+ * Compare S one-row calls with one S-row call on identical resident weights.
+ * This is not full-model prefill throughput or a CPU/GPU comparison. */
+#include "../backend_cuda.h"
+#include
+#include
+#include
+#include
+#include
+
+static double median(std::vector values) {
+ std::sort(values.begin(),values.end());
+ return values[values.size()/2];
+}
+
+static bool measure(int I,int O,int S,int device) {
+ std::vector q((size_t)I*O);
+ std::vector scale(O), x((size_t)S*I), serial((size_t)S*O), batch(serial.size());
+ for(size_t i=0;i times[2], ratios;
+ for(int rep=0;rep<9 && ok;rep++) {
+ double pair[2]={};
+ for(int arm=0;arm<2 && ok;arm++) {
+ int mode=(rep+arm)%2;
+ auto start=std::chrono::steady_clock::now();
+ ok=run(mode!=0);
+ pair[mode]=std::chrono::duration(std::chrono::steady_clock::now()-start).count();
+ times[mode].push_back(pair[mode]);
+ }
+ if(ok) ratios.push_back(pair[0]/pair[1]);
+ for(size_t i=0;i1e-5*(1+std::fabs(reference))) ok=false;
+ }
+ coli_cuda_tensor_free(tensor);
+ if(!ok){std::fprintf(stderr,"FAIL: resident batch I=%d O=%d S=%d\n",I,O,S);return false;}
+ std::printf("{\"input\":%d,\"output\":%d,\"rows\":%d,\"pairs\":9,"
+ "\"serial_median_ms\":%.6f,\"batch_median_ms\":%.6f,"
+ "\"paired_speedup_median\":%.6f,\"cpu_sample_max_abs_error\":%.9g,"
+ "\"serial_ms\":[",I,O,S,median(times[0]),median(times[1]),median(ratios),max_error);
+ for(size_t i=0;i
#include
-#include
+#if !defined(__HIPCC__)
+#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */
+#endif
#include
#include
diff --git a/c/tests/bench_dsv4_mxfp8.cu b/c/tests/bench_dsv4_mxfp8.cu
index 209433ee4..c072719a2 100644
--- a/c/tests/bench_dsv4_mxfp8.cu
+++ b/c/tests/bench_dsv4_mxfp8.cu
@@ -1,4 +1,6 @@
-#include
+#if !defined(__HIPCC__)
+#include /* under HIP, backend_gpu_compat.h (via backend_cuda.cu) provides these */
+#endif
#include
#include
#include
diff --git a/c/tests/bench_i4p_gidot.c b/c/tests/bench_i4p_gidot.c
new file mode 100644
index 000000000..4843a0703
--- /dev/null
+++ b/c/tests/bench_i4p_gidot.c
@@ -0,0 +1,135 @@
+/* Microbenchmark: per-row vs multi-row K1b grouped planar IDOT (quant.h).
+ * NOT a unit test -- test_int_kernel_exact.c proves correctness/bit-equality.
+ *
+ * This measures the multi-row claim behind the fmt=4 tile work: the per-row
+ * kernel re-loads and re-masks every weight block once per activation row, so
+ * batched calls (prefill batch-union rows, the serve mux's decode batch) pay
+ * S times the weight traffic. The 1x4 tile (and AMX on Sapphire Rapids) pays
+ * it once per tile. The OLD kernel below is a verbatim copy of the pre-tile
+ * body; the NEW one is the real dispatcher, so on an AMX host this also
+ * benches the tile-unit path (AMX_S_MIN gates it; force with AMX_S_MIN=2).
+ *
+ * Run: make tests/bench_i4p_gidot ARCH=native && ./tests/bench_i4p_gidot
+ * (not in TEST_BINS -- not a gate) */
+#define main coli_glm_main_unused
+#include "../colibri.c"
+#undef main
+#include
+#include
+
+static uint32_t rs=0x2545F491u;
+static uint32_t xr(void){ rs^=rs<<13; rs^=rs>>17; rs^=rs<<5; return rs; }
+
+/* ---- OLD kernel: verbatim copy of the pre-tile per-row body ---- */
+static void gidot_old(float *y, const int8_t *xq, const float *sx,
+ const int32_t *xsg, const uint8_t *q4,
+ const float *scale, int S, int I, int O, int gs){
+ int rb=(I+1)/2, ng=(I+gs-1)/gs, bpg=gs/64;
+ #pragma omp parallel for schedule(static)
+ for(int o=0;o>1);
+ const int8_t *xb=xr2+base;
+#if defined(coli_dpbusd256)
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ __m256i bb=_mm256_loadu_si256((const __m256i*)blk);
+ __m256i acc=_mm256_setzero_si256();
+ acc=coli_dpbusd256(acc,_mm256_and_si256(bb,m4),
+ _mm256_loadu_si256((const __m256i*)xb));
+ acc=coli_dpbusd256(acc,_mm256_and_si256(_mm256_srli_epi16(bb,4),m4),
+ _mm256_loadu_si256((const __m256i*)(xb+32)));
+ d+=hsum256_i32(acc);
+#elif defined(__AVX2__)
+ const __m256i m4=_mm256_set1_epi8(0x0F);
+ const __m256i ones=_mm256_set1_epi16(1);
+ __m256i bb=_mm256_loadu_si256((const __m256i*)blk);
+ __m256i p0=_mm256_maddubs_epi16(_mm256_and_si256(bb,m4),
+ _mm256_loadu_si256((const __m256i*)xb));
+ __m256i p1=_mm256_maddubs_epi16(_mm256_and_si256(_mm256_srli_epi16(bb,4),m4),
+ _mm256_loadu_si256((const __m256i*)(xb+32)));
+ __m256i acc=_mm256_add_epi32(_mm256_madd_epi16(p0,ones),
+ _mm256_madd_epi16(p1,ones));
+ d+=hsum256_i32(acc);
+#else
+ for(int k=0;k<32;k++){
+ d+=(int32_t)(blk[k]&0xF)*xb[k];
+ d+=(int32_t)(blk[k]>>4)*xb[k+32];
+ }
+#endif
+ }
+ a=fmaf((float)(d-8*xg[g]),scl[g],a);
+ }
+ if(g*gs>1];
+ d+=(int32_t)((i&1)?(byte>>4):(byte&0xF))*xr2[i];
+ }
+ a=fmaf((float)(d-8*xg[g]),scl[g],a);
+ }
+ y[(int64_t)s*O+o]=a*sx[s];
+ }
+ }
+}
+
+#define O_DIM 2048
+#define I_DIM 2048
+#define GS 64
+#define REPS 400
+static int cmp_d(const void*a,const void*b){ double x=*(const double*)a,y=*(const double*)b; return xy?1:0; }
+
+int main(void){
+ int rb=(I_DIM+1)/2, ng=(I_DIM+GS-1)/GS;
+ int Ss[]={1,4,8,16};
+ uint8_t *q4=malloc((size_t)O_DIM*rb);
+ float *sc=malloc((size_t)O_DIM*ng*sizeof(float));
+ for(size_t i=0;i<(size_t)O_DIM*rb;i++) q4[i]=(uint8_t)(xr()&0xFF);
+ planarize_i4(q4,O_DIM,I_DIM);
+ for(size_t i=0;i<(size_t)O_DIM*ng;i++) sc[i]=0.0005f+(xr()%911)*1e-6f;
+
+ int Smax=16;
+ int8_t *xq=malloc((size_t)Smax*I_DIM);
+ float *sx=malloc(Smax*sizeof(float));
+ int32_t *xsg=malloc((size_t)Smax*ng*sizeof(int32_t));
+ for(size_t i=0;i<(size_t)Smax*I_DIM;i++){ int v=(int)(xr()%255)-127; xq[i]=(int8_t)v; }
+ for(int s=0;s
+#include
+#include
+#include
+#include
+
+enum { HIDDEN = 4096, OUT = 4096, N_MATRICES = 48 };
+
+static long vmhwm_kb(void) {
+ FILE *f = fopen("/proc/self/status", "r");
+ if (!f) return -1;
+ char line[256]; long kb = -1;
+ while (fgets(line, sizeof line, f)) {
+ if (!strncmp(line, "VmHWM:", 6)) { sscanf(line + 6, "%ld", &kb); break; }
+ }
+ fclose(f);
+ return kb;
+}
+
+static void fill(float *w, int64_t n, int salt) {
+ for (int64_t i = 0; i < n; i++) w[i] = (float)(((i * 2654435761u + salt) % 2003) - 1000) * 0.001f;
+}
+
+static void quantize_row_major(const float *w, int I, int O, int8_t *q, float *sc) {
+ for (int o = 0; o < O; o++) {
+ const float *r = w + (int64_t)o * I; float am = 0.f;
+ for (int i = 0; i < I; i++) { float a = fabsf(r[i]); if (a > am) am = a; }
+ float s = am > 1e-12f ? am / 127.f : 1.f; sc[o] = s; float inv = 1.f / s;
+ int8_t *d = q + (int64_t)o * I;
+ for (int i = 0; i < I; i++) { int v = (int)lrintf(r[i] * inv); if (v > 127) v = 127; if (v < -127) v = -127; d[i] = (int8_t)v; }
+ }
+}
+
+static void run_old(void) {
+ int64_t n = (int64_t)HIDDEN * OUT;
+ float **f32s = malloc(N_MATRICES * sizeof(float*));
+ /* pass 1: "load" every matrix's f32 copy (as model_init_range used to,
+ * for every dense matrix in the whole model, before any quantization) */
+ for (int m = 0; m < N_MATRICES; m++) {
+ f32s[m] = malloc((size_t)n * sizeof(float));
+ fill(f32s[m], n, m);
+ }
+ printf("after loading all %d f32 matrices: VmHWM=%ld MiB\n", N_MATRICES, vmhwm_kb() / 1024);
+ /* pass 2: quantize every matrix (as the old post-hoc qdw_register loop
+ * did), freeing each f32 copy only after ALL quantization is done */
+ int8_t **qs = malloc(N_MATRICES * sizeof(int8_t*));
+ float **scs = malloc(N_MATRICES * sizeof(float*));
+ for (int m = 0; m < N_MATRICES; m++) {
+ qs[m] = malloc((size_t)n);
+ scs[m] = malloc((size_t)OUT * sizeof(float));
+ quantize_row_major(f32s[m], HIDDEN, OUT, qs[m], scs[m]);
+ }
+ printf("after quantizing all (pre-free): VmHWM=%ld MiB\n", vmhwm_kb() / 1024);
+ for (int m = 0; m < N_MATRICES; m++) free(f32s[m]);
+ printf("[old pattern] peak VmHWM=%ld MiB\n", vmhwm_kb() / 1024);
+}
+
+static void run_new(void) {
+ int64_t n = (int64_t)HIDDEN * OUT;
+ int8_t **qs = malloc(N_MATRICES * sizeof(int8_t*));
+ float **scs = malloc(N_MATRICES * sizeof(float*));
+ /* load_tq: read one matrix's f32 copy, quantize it, free it, next matrix */
+ for (int m = 0; m < N_MATRICES; m++) {
+ float *w = malloc((size_t)n * sizeof(float));
+ fill(w, n, m);
+ qs[m] = malloc((size_t)n);
+ scs[m] = malloc((size_t)OUT * sizeof(float));
+ quantize_row_major(w, HIDDEN, OUT, qs[m], scs[m]);
+ free(w);
+ }
+ printf("[new pattern] peak VmHWM=%ld MiB\n", vmhwm_kb() / 1024);
+}
+
+int main(int argc, char **argv) {
+ if (argc != 2 || (strcmp(argv[1], "old") && strcmp(argv[1], "new"))) {
+ fprintf(stderr, "usage: %s old|new\n", argv[0]); return 2;
+ }
+ printf("N_MATRICES=%d each %dx%d f32 (%.1f MiB) -- %s pattern\n",
+ N_MATRICES, HIDDEN, OUT, (double)HIDDEN * OUT * 4 / 1048576.0, argv[1]);
+ if (!strcmp(argv[1], "old")) run_old(); else run_new();
+ return 0;
+}
diff --git a/c/tests/bench_qwen36_dense_batch.c b/c/tests/bench_qwen36_dense_batch.c
index 63f0fc224..36e19e55a 100644
--- a/c/tests/bench_qwen36_dense_batch.c
+++ b/c/tests/bench_qwen36_dense_batch.c
@@ -39,13 +39,13 @@ static double shared_run(Model *m,Layer *l,const float *x,const float *seed,
static int shared_benchmark(void) {
Model m;memset(&m,0,sizeof(m));m.c.hidden=I;m.c.shared_inter=O;
Layer l;memset(&l,0,sizeof(l));
- l.sh_g=falloc((int64_t)O*I);l.sh_u=falloc((int64_t)O*I);
- l.sh_d=falloc((int64_t)I*O);l.sh_gate=falloc(I);
- for(int64_t i=0;i<(int64_t)O*I;i++){l.sh_g[i]=value(i,2);l.sh_u[i]=value(i,3);}
- for(int64_t i=0;i<(int64_t)I*O;i++)l.sh_d[i]=value(i,4);
+ l.sh_g.w=falloc((int64_t)O*I);l.sh_u.w=falloc((int64_t)O*I);
+ l.sh_d.w=falloc((int64_t)I*O);l.sh_gate=falloc(I);
+ for(int64_t i=0;i<(int64_t)O*I;i++){((float*)l.sh_g.w)[i]=value(i,2);((float*)l.sh_u.w)[i]=value(i,3);}
+ for(int64_t i=0;i<(int64_t)I*O;i++)((float*)l.sh_d.w)[i]=value(i,4);
for(int i=0;i