From ea492c5f9095df747c2f0b571a71c01930c6f424 Mon Sep 17 00:00:00 2001 From: Circle-Cheng Date: Thu, 3 Sep 2026 10:56:51 +0800 Subject: [PATCH] fix(gemma4): support dense Gemma-4 GGUF checkpoints MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit parse_gguf_config unconditionally required gemma4.expert_count and hardcoded moe_enabled=True, so a dense Gemma-4 GGUF (gemma-4-12B-it, gemma-4-31B-it) could not load at all — the GGUF path structurally could not represent a non-MoE checkpoint, while docs/models.md lists dense family members and GGUF is the documented non-safetensors path. Mirror the HF-side parse_config: default the expert fields to 0 when absent from the metadata, derive moe_enabled = num_experts > 0, and gate expert_quant/moe_weight_format on it. MoE GGUFs parse exactly as before. Regression tests build a minimal GGUF header (magic + version + zero counts) so the parser runs against metadata dicts without any model download: MoE keys present -> moe path unchanged; keys absent or zero -> dense path with expert_quant none. Fixes #357 --- python/freetoken/models/gemma4/gguf.py | 21 ++++-- tests/models/test_gemma4_gguf_config.py | 89 +++++++++++++++++++++++++ 2 files changed, 104 insertions(+), 6 deletions(-) create mode 100644 tests/models/test_gemma4_gguf_config.py diff --git a/python/freetoken/models/gemma4/gguf.py b/python/freetoken/models/gemma4/gguf.py index 437822b51..d7696283c 100644 --- a/python/freetoken/models/gemma4/gguf.py +++ b/python/freetoken/models/gemma4/gguf.py @@ -74,6 +74,14 @@ def g(key: str): full_kv = int(kv_per_layer[full_layer_ids[0]]) if full_layer_ids else int(kv_per_layer[0]) max_pos = int(g("context_length")) + # Dense Gemma-4 checkpoints (gemma-4-12B-it, gemma-4-31B-it) carry no expert fields + # in their GGUF metadata; only MoE members (e.g. Gemma-4-26B-A4B) do. Mirror the HF + # side (gemma4.config.parse_config): default the expert fields to 0 and derive + # moe_enabled from them instead of assuming every GGUF is MoE. + num_experts = int(m.get("gemma4.expert_count", 0) or 0) + num_experts_per_tok = int(m.get("gemma4.expert_used_count", 0) or 0) + moe_intermediate_size = int(m.get("gemma4.expert_feed_forward_length", 0) or 0) + moe_enabled = num_experts > 0 full_rotary = RotaryConfig( head_dim=full_head_dim, rotary_dim=_full_rotary_dim(shim, full_head_dim), @@ -105,15 +113,16 @@ def g(key: str): rms_norm_eps=float(g("attention.layer_norm_rms_epsilon")), tie_word_embeddings=bool(shim.tie_word_embeddings), rotary_config=full_rotary, - num_experts=int(g("expert_count")), - num_experts_per_tok=int(g("expert_used_count")), - moe_intermediate_size=int(g("expert_feed_forward_length")), + num_experts=num_experts, + num_experts_per_tok=num_experts_per_tok, + moe_intermediate_size=moe_intermediate_size, norm_topk_prob=True, model_type="gemma4", architectures=list(shim.architectures), - moe_enabled=True, - expert_quant="q4_0", - moe_weight_format="q4_0", + moe_enabled=moe_enabled, + # Native-Q4_0 offload-cache path for the routed experts (MoE checkpoints only). + expert_quant="q4_0" if moe_enabled else "none", + moe_weight_format="q4_0" if moe_enabled else "none", use_qk_norm=True, attn_sm_scale=1.0, final_logit_softcapping=float(g("final_logit_softcapping")), diff --git a/tests/models/test_gemma4_gguf_config.py b/tests/models/test_gemma4_gguf_config.py new file mode 100644 index 000000000..e5594c27f --- /dev/null +++ b/tests/models/test_gemma4_gguf_config.py @@ -0,0 +1,89 @@ +"""parse_gguf_config must handle dense Gemma-4 GGUFs (issue #357). + +Dense checkpoints (gemma-4-12B-it, gemma-4-31B-it) carry no gemma4.expert_* keys +in their GGUF metadata. The old parser unconditionally required expert_count and +hardcoded moe_enabled=True, so dense GGUFs could not load at all. +""" + +import struct + +import pytest + +from freetoken.models.gguf.config import GgufConfigShim +from freetoken.models.gemma4.gguf import parse_gguf_config + + +def make_shim(metadata_overrides: dict, tmp_path) -> GgufConfigShim: + """A metadata-only shim over an empty file; _full_rotary_dim falls back to head_dim//4.""" + empty = tmp_path / "meta_only.gguf" + # Minimal GGUF: magic + version 3 + kv_count=0 + tensor_count=0. GGUFReader opens + # it fine with no fields/tensors, so _full_rotary_dim takes its metadata-only + # fallback (head_dim//4) without needing a real checkpoint. + empty.write_bytes(b"GGUF" + struct.pack("