Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions common/arg.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -2784,6 +2784,26 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.lazy_mode = LLAMA_LAZY_MODE_ON;
}
).set_env("LLAMA_ARG_NGRAM_ON_DISK"));
add_opt(common_arg(
{"--expert-cache-slots"}, "N",
"DeepSeek V4.1 routed experts resident per layer; requires --expert-cache-mib",
[](common_params & params, int value) {
if (value <= 0) {
throw std::invalid_argument("invalid value");
}
params.expert_cache_slots = value;
}
).set_env("LLAMA_ARG_EXPERT_CACHE_SLOTS"));
add_opt(common_arg(
{"--expert-cache-mib"}, "MiB",
"aggregate DeepSeek V4.1 fixed expert slot-tensor capacity; requires --expert-cache-slots",
[](common_params & params, int value) {
if (value <= 0) {
throw std::invalid_argument("invalid value");
}
params.expert_cache_mib = value;
}
).set_env("LLAMA_ARG_EXPERT_CACHE_MIB"));
add_opt(common_arg(
{"-cmoe", "--cpu-moe"},
"keep all Mixture of Experts (MoE) weights in the CPU",
Expand Down
2 changes: 2 additions & 0 deletions common/common.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1693,6 +1693,8 @@ struct llama_model_params common_model_params_to_llama(common_params & params) {
mparams.check_tensors = params.check_tensors;
mparams.use_extra_bufts = !params.no_extra_bufts;
mparams.no_host = params.no_host;
mparams.expert_cache_slots = params.expert_cache_slots;
mparams.expert_cache_bytes = params.expert_cache_mib > 0 ? (size_t) params.expert_cache_mib << 20 : 0;
if (params.kv_overrides.empty()) {
mparams.kv_overrides = NULL;
} else {
Expand Down
2 changes: 2 additions & 0 deletions common/common.h
Original file line number Diff line number Diff line change
Expand Up @@ -625,6 +625,8 @@ struct common_params {
bool no_op_offload = false; // globally disable offload host tensor operations to device
bool no_extra_bufts = false; // disable extra buffer types (used for weight repacking)
bool no_host = false; // bypass host buffer allowing extra buffers to be used
int32_t expert_cache_slots = 0; // DeepSeek V4.1 routed experts resident per layer
int32_t expert_cache_mib = 0; // aggregate fixed slot-tensor capacity
bool single_turn = false; // single turn chat conversation

ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
Expand Down
51 changes: 51 additions & 0 deletions gguf-py/gguf/constants.py
Original file line number Diff line number Diff line change
Expand Up @@ -559,6 +559,7 @@ class MODEL_ARCH(IntEnum):
DEEPSEEK2OCR = auto()
DEEPSEEK32 = auto()
DEEPSEEK4 = auto()
DEEPSEEK41 = auto()
CHATGLM = auto()
GLM4 = auto()
GLM4_MOE = auto()
Expand Down Expand Up @@ -823,6 +824,10 @@ class MODEL_TENSOR(IntEnum):
PLE_NORM_QUERY = auto() # qwen4exp
PLE_NORM_CONV = auto() # qwen4exp
PLE_CONV1D = auto() # qwen4exp
ENGRAM_EMBD = auto()
ENGRAM_Q_NORM = auto()
ENGRAM_K_NORM = auto()
ENGRAM_KV = auto()
ATTN_COMPRESSOR_WKV = auto()
ATTN_COMPRESSOR_WGATE = auto()
ATTN_COMPRESSOR_APE = auto()
Expand Down Expand Up @@ -1317,6 +1322,7 @@ class MODEL_TENSOR(IntEnum):
MODEL_ARCH.DEEPSEEK2OCR: "deepseek2-ocr",
MODEL_ARCH.DEEPSEEK32: "deepseek32",
MODEL_ARCH.DEEPSEEK4: "deepseek4",
MODEL_ARCH.DEEPSEEK41: "deepseek41",
MODEL_ARCH.CHATGLM: "chatglm",
MODEL_ARCH.GLM4: "glm4",
MODEL_ARCH.GLM4_MOE: "glm4moe",
Expand Down Expand Up @@ -1580,6 +1586,10 @@ class MODEL_TENSOR(IntEnum):
MODEL_TENSOR.PLE_NORM_QUERY: "blk.{bid}.ple_norm_query", # qwen4exp
MODEL_TENSOR.PLE_NORM_CONV: "blk.{bid}.ple_norm_conv", # qwen4exp
MODEL_TENSOR.PLE_CONV1D: "blk.{bid}.ple_conv1d", # qwen4exp
MODEL_TENSOR.ENGRAM_EMBD: "blk.{bid}.engram_embd",
MODEL_TENSOR.ENGRAM_Q_NORM: "blk.{bid}.engram_q_norm",
MODEL_TENSOR.ENGRAM_K_NORM: "blk.{bid}.engram_k_norm",
MODEL_TENSOR.ENGRAM_KV: "blk.{bid}.engram_kv",
MODEL_TENSOR.ATTN_COMPRESSOR_WKV: "blk.{bid}.attn_compressor_kv",
MODEL_TENSOR.ATTN_COMPRESSOR_WGATE: "blk.{bid}.attn_compressor_gate",
MODEL_TENSOR.ATTN_COMPRESSOR_APE: "blk.{bid}.attn_compressor_ape",
Expand Down Expand Up @@ -3940,6 +3950,47 @@ class MODEL_TENSOR(IntEnum):
MODEL_TENSOR.NEXTN_SHARED_HEAD_HEAD,
MODEL_TENSOR.NEXTN_SHARED_HEAD_NORM,
],
MODEL_ARCH.DEEPSEEK41: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.OUTPUT,
MODEL_TENSOR.ATTN_NORM,
MODEL_TENSOR.ATTN_SINKS,
MODEL_TENSOR.ATTN_Q_A,
MODEL_TENSOR.ATTN_Q_B,
MODEL_TENSOR.ATTN_Q_A_NORM,
MODEL_TENSOR.ATTN_KV,
MODEL_TENSOR.ATTN_KV_A_NORM,
MODEL_TENSOR.ATTN_OUT_A,
MODEL_TENSOR.ATTN_OUT_B,
MODEL_TENSOR.HC_ATTN_FN,
MODEL_TENSOR.HC_ATTN_BASE,
MODEL_TENSOR.HC_ATTN_SCALE,
MODEL_TENSOR.HC_FFN_FN,
MODEL_TENSOR.HC_FFN_BASE,
MODEL_TENSOR.HC_FFN_SCALE,
MODEL_TENSOR.ATTN_COMPRESSOR_WKV,
MODEL_TENSOR.ATTN_COMPRESSOR_WGATE,
MODEL_TENSOR.ATTN_COMPRESSOR_NORM,
MODEL_TENSOR.INDEXER_K_NORM,
MODEL_TENSOR.INDEXER_PROJ,
MODEL_TENSOR.INDEXER_ATTN_K,
MODEL_TENSOR.INDEXER_ATTN_Q_B,
MODEL_TENSOR.FFN_GATE_INP,
MODEL_TENSOR.FFN_EXP_PROBS_B,
MODEL_TENSOR.FFN_EXP_PROBS_B_VL,
MODEL_TENSOR.FFN_NORM,
MODEL_TENSOR.FFN_GATE_EXP,
MODEL_TENSOR.FFN_DOWN_EXP,
MODEL_TENSOR.FFN_UP_EXP,
MODEL_TENSOR.FFN_GATE_SHEXP,
MODEL_TENSOR.FFN_DOWN_SHEXP,
MODEL_TENSOR.FFN_UP_SHEXP,
MODEL_TENSOR.ENGRAM_EMBD,
MODEL_TENSOR.ENGRAM_Q_NORM,
MODEL_TENSOR.ENGRAM_K_NORM,
MODEL_TENSOR.ENGRAM_KV,
],
MODEL_ARCH.ERNIE4_5_MOE: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT_NORM,
Expand Down
49 changes: 49 additions & 0 deletions gguf-py/tests/test_deepseek41_schema.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,49 @@
#!/usr/bin/env python3

import os
import sys
import unittest
from pathlib import Path

if "NO_LOCAL_GGUF" not in os.environ and (Path(__file__).parent.parent.parent / "gguf-py").exists():
sys.path.insert(0, str(Path(__file__).parent.parent))

from gguf.constants import MODEL_ARCH, MODEL_ARCH_NAMES, MODEL_TENSOR, MODEL_TENSORS, TENSOR_NAMES


class TestDeepSeek41Schema(unittest.TestCase):

def test_architecture_name(self):
self.assertEqual(MODEL_ARCH_NAMES[MODEL_ARCH.DEEPSEEK41], "deepseek41")

def test_engram_tensor_names(self):
expected = {
MODEL_TENSOR.ENGRAM_EMBD: "blk.14.engram_embd",
MODEL_TENSOR.ENGRAM_Q_NORM: "blk.14.engram_q_norm",
MODEL_TENSOR.ENGRAM_K_NORM: "blk.14.engram_k_norm",
MODEL_TENSOR.ENGRAM_KV: "blk.14.engram_kv",
}

for tensor, name in expected.items():
self.assertIn(tensor, MODEL_TENSORS[MODEL_ARCH.DEEPSEEK41])
self.assertEqual(TENSOR_NAMES[tensor].format(bid=14), name)

def test_output_head_does_not_require_deepseek4_hc_tensors(self):
tensors = MODEL_TENSORS[MODEL_ARCH.DEEPSEEK41]

self.assertIn(MODEL_TENSOR.OUTPUT_NORM, tensors)
self.assertIn(MODEL_TENSOR.OUTPUT, tensors)
self.assertNotIn(MODEL_TENSOR.HC_HEAD_FN, tensors)
self.assertNotIn(MODEL_TENSOR.HC_HEAD_BASE, tensors)
self.assertNotIn(MODEL_TENSOR.HC_HEAD_SCALE, tensors)

def test_kv_a_norm_does_not_use_deepseek4_tensor_kind(self):
tensors = MODEL_TENSORS[MODEL_ARCH.DEEPSEEK41]

self.assertIn(MODEL_TENSOR.ATTN_KV_A_NORM, tensors)
self.assertNotIn(MODEL_TENSOR.ATTN_KV_NORM, tensors)
self.assertEqual(TENSOR_NAMES[MODEL_TENSOR.ATTN_KV_A_NORM].format(bid=14), "blk.14.attn_kv_a_norm")


if __name__ == "__main__":
unittest.main()
4 changes: 4 additions & 0 deletions include/llama.h
Original file line number Diff line number Diff line change
Expand Up @@ -345,6 +345,10 @@ extern "C" {
// the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
int32_t main_gpu;

// DeepSeek V4.1 routed-expert cache. Both values must be non-zero.
size_t expert_cache_bytes;
int32_t expert_cache_slots;

// proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()
const float * tensor_split;

Expand Down
7 changes: 7 additions & 0 deletions src/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,10 @@ set(LLAMA_CORE_SOURCES
llama-chat.cpp
llama-context.cpp
llama-cparams.cpp
llama-dsv41.cpp
llama-dsv41-engram.cpp
llama-dsv41-expert.cpp
llama-expert-store.cpp
llama-grammar.cpp
llama-graph.cpp
llama-hparams.cpp
Expand All @@ -28,11 +32,14 @@ set(LLAMA_CORE_SOURCES
llama-kv-cache-msa.cpp
llama-kv-cache-dsv4.cpp
llama-memory.cpp
llama-memory-dsv41.cpp
llama-memory-hybrid.cpp
llama-memory-hybrid-iswa.cpp
llama-memory-hybrid-idx.cpp
llama-memory-recurrent.cpp
llama-mmap.cpp
llama-bounded-file.cpp
llama-engram.cpp
llama-ple-disk.cpp
llama-model-loader.cpp
llama-model-saver.cpp
Expand Down
61 changes: 61 additions & 0 deletions src/llama-arch.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -81,6 +81,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
{ LLM_ARCH_DEEPSEEK2OCR, "deepseek2-ocr" },
{ LLM_ARCH_DEEPSEEK32, "deepseek32" },
{ LLM_ARCH_DEEPSEEK4, "deepseek4" },
{ LLM_ARCH_DEEPSEEK41, "deepseek41" },
{ LLM_ARCH_CHATGLM, "chatglm" },
{ LLM_ARCH_GLM4, "glm4" },
{ LLM_ARCH_GLM4_MOE, "glm4moe" },
Expand Down Expand Up @@ -310,6 +311,57 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_PLE_EOS_TOKEN_ID, "%s.ple.eos_token_id" },
{ LLM_KV_PLE_IMAGE_TOKEN_ID, "%s.ple.image_token_id" },

{ LLM_KV_DSV41_CONFIG, "%s.config" },
{ LLM_KV_DSV41_VOCAB_SIZE, "%s.vocab_size" },
{ LLM_KV_DSV41_HIDDEN_SIZE, "%s.hidden_size" },
{ LLM_KV_DSV41_MOE_INTERMEDIATE_SIZE, "%s.moe_intermediate_size" },
{ LLM_KV_DSV41_NUM_HIDDEN_LAYERS, "%s.num_hidden_layers" },
{ LLM_KV_DSV41_NUM_ATTENTION_HEADS, "%s.num_attention_heads" },
{ LLM_KV_DSV41_NUM_KEY_VALUE_HEADS, "%s.num_key_value_heads" },
{ LLM_KV_DSV41_HEAD_DIM, "%s.head_dim" },
{ LLM_KV_DSV41_QK_ROPE_HEAD_DIM, "%s.qk_rope_head_dim" },
{ LLM_KV_DSV41_Q_LORA_RANK, "%s.q_lora_rank" },
{ LLM_KV_DSV41_O_LORA_RANK, "%s.o_lora_rank" },
{ LLM_KV_DSV41_O_GROUPS, "%s.o_groups" },
{ LLM_KV_DSV41_N_ROUTED_EXPERTS, "%s.n_routed_experts" },
{ LLM_KV_DSV41_N_SHARED_EXPERTS, "%s.n_shared_experts" },
{ LLM_KV_DSV41_NUM_EXPERTS_PER_TOK, "%s.num_experts_per_tok" },
{ LLM_KV_DSV41_MAX_POSITION_EMBEDDINGS, "%s.max_position_embeddings" },
{ LLM_KV_DSV41_SLIDING_WINDOW, "%s.sliding_window" },
{ LLM_KV_DSV41_INDEX_N_HEADS, "%s.index_n_heads" },
{ LLM_KV_DSV41_INDEX_HEAD_DIM, "%s.index_head_dim" },
{ LLM_KV_DSV41_INDEX_TOPK, "%s.index_topk" },
{ LLM_KV_DSV41_CANDIDATE_SOURCE_LAYER_ID, "%s.candidate_source_layer_id" },
{ LLM_KV_DSV41_CANDIDATE_TOPK_BLOCKS, "%s.candidate_topk_blocks" },
{ LLM_KV_DSV41_CANDIDATE_BLOCK_SIZE, "%s.candidate_block_size" },
{ LLM_KV_DSV41_HC_MULT, "%s.hc_mult" },
{ LLM_KV_DSV41_HC_SINKHORN_ITERS, "%s.hc_sinkhorn_iters" },
{ LLM_KV_DSV41_ROPE_THETA, "%s.rope_theta" },
{ LLM_KV_DSV41_COMPRESS_ROPE_THETA, "%s.compress_rope_theta" },
{ LLM_KV_DSV41_RMS_NORM_EPS, "%s.rms_norm_eps" },
{ LLM_KV_DSV41_HC_EPS, "%s.hc_eps" },
{ LLM_KV_DSV41_SWIGLU_LIMIT, "%s.swiglu_limit" },
{ LLM_KV_DSV41_ROUTED_SCALING_FACTOR, "%s.routed_scaling_factor" },
{ LLM_KV_DSV41_SCORING_FUNC, "%s.scoring_func" },
{ LLM_KV_DSV41_HIDDEN_ACT, "%s.hidden_act" },
{ LLM_KV_DSV41_TOPK_METHOD, "%s.topk_method" },
{ LLM_KV_DSV41_NORM_TOPK_PROB, "%s.norm_topk_prob" },
{ LLM_KV_DSV41_COMPRESS_RATIOS, "%s.compress_ratios" },
{ LLM_KV_DSV41_KV_SOURCE_LAYER_IDS, "%s.kv_source_layer_ids" },
{ LLM_KV_DSV41_INDEX_SOURCE_LAYER_IDS, "%s.index_source_layer_ids" },
{ LLM_KV_DSV41_ROPE_SCALING_FACTOR, "%s.rope_scaling.factor" },
{ LLM_KV_DSV41_ROPE_SCALING_BETA_FAST, "%s.rope_scaling.beta_fast" },
{ LLM_KV_DSV41_ROPE_SCALING_BETA_SLOW, "%s.rope_scaling.beta_slow" },
{ LLM_KV_DSV41_ROPE_SCALING_ORIG_CTX_LEN, "%s.rope_scaling.original_max_position_embeddings" },
{ LLM_KV_DSV41_ENGRAM_ENCODING, "%s.engram.encoding" },
{ LLM_KV_DSV41_ENGRAM_LAYER_IDS, "%s.engram.layer_ids" },
{ LLM_KV_DSV41_ENGRAM_ROWS, "%s.engram.rows" },
{ LLM_KV_DSV41_ENGRAM_COMPRESSED_VOCAB_SIZE, "%s.engram.compressed_vocab_size" },
{ LLM_KV_DSV41_ENGRAM_PAD_ID, "%s.engram.pad_id" },
{ LLM_KV_DSV41_ENGRAM_TOKEN_MAP, "%s.engram.token_map" },
{ LLM_KV_DSV41_ENGRAM_PRIMES, "%s.engram.primes" },
{ LLM_KV_DSV41_ENGRAM_MULTIPLIERS, "%s.engram.multipliers" },

{ LLM_KV_HASH_LAYER_COUNT, "%s.hash_layer_count" },

{ LLM_KV_ROPE_DIMENSION_COUNT, "%s.rope.dimension_count" },
Expand Down Expand Up @@ -546,6 +598,10 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
{ LLM_TENSOR_PLE_NORM_QUERY, "blk.%d.ple_norm_query" },
{ LLM_TENSOR_PLE_NORM_CONV, "blk.%d.ple_norm_conv" },
{ LLM_TENSOR_PLE_CONV1D, "blk.%d.ple_conv1d" },
{ LLM_TENSOR_ENGRAM_EMBD, "blk.%d.engram_embd" },
{ LLM_TENSOR_ENGRAM_Q_NORM, "blk.%d.engram_q_norm" },
{ LLM_TENSOR_ENGRAM_K_NORM, "blk.%d.engram_k_norm" },
{ LLM_TENSOR_ENGRAM_KV, "blk.%d.engram_kv" },
{ LLM_TENSOR_ATTN_COMPRESSOR_WKV, "blk.%d.attn_compressor_kv" },
{ LLM_TENSOR_ATTN_COMPRESSOR_WGATE, "blk.%d.attn_compressor_gate" },
{ LLM_TENSOR_ATTN_COMPRESSOR_APE, "blk.%d.attn_compressor_ape" },
Expand Down Expand Up @@ -777,6 +833,10 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
{LLM_TENSOR_PLE_NORM_QUERY, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_PLE_NORM_CONV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_PLE_CONV1D, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
{LLM_TENSOR_ENGRAM_EMBD, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_GET_ROWS}},
{LLM_TENSOR_ENGRAM_Q_NORM, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_ENGRAM_K_NORM, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_ENGRAM_KV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_ATTN_COMPRESSOR_WKV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_ATTN_COMPRESSOR_WGATE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_ATTN_COMPRESSOR_APE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_GET_ROWS}},
Expand Down Expand Up @@ -1089,6 +1149,7 @@ bool llm_arch_is_hybrid(const llm_arch & arch) {
case LLM_ARCH_QWEN35MOE:
case LLM_ARCH_QWEN4EXP:
case LLM_ARCH_DEEPSEEK4:
case LLM_ARCH_DEEPSEEK41:
case LLM_ARCH_MINIMAX_01:
return true;
default:
Expand Down
56 changes: 56 additions & 0 deletions src/llama-arch.h
Original file line number Diff line number Diff line change
Expand Up @@ -86,6 +86,7 @@ enum llm_arch {
LLM_ARCH_DEEPSEEK2OCR,
LLM_ARCH_DEEPSEEK32,
LLM_ARCH_DEEPSEEK4,
LLM_ARCH_DEEPSEEK41,
LLM_ARCH_CHATGLM,
LLM_ARCH_GLM4,
LLM_ARCH_GLM4_MOE,
Expand Down Expand Up @@ -315,6 +316,57 @@ enum llm_kv {
LLM_KV_PLE_EOS_TOKEN_ID,
LLM_KV_PLE_IMAGE_TOKEN_ID,

LLM_KV_DSV41_CONFIG,
LLM_KV_DSV41_VOCAB_SIZE,
LLM_KV_DSV41_HIDDEN_SIZE,
LLM_KV_DSV41_MOE_INTERMEDIATE_SIZE,
LLM_KV_DSV41_NUM_HIDDEN_LAYERS,
LLM_KV_DSV41_NUM_ATTENTION_HEADS,
LLM_KV_DSV41_NUM_KEY_VALUE_HEADS,
LLM_KV_DSV41_HEAD_DIM,
LLM_KV_DSV41_QK_ROPE_HEAD_DIM,
LLM_KV_DSV41_Q_LORA_RANK,
LLM_KV_DSV41_O_LORA_RANK,
LLM_KV_DSV41_O_GROUPS,
LLM_KV_DSV41_N_ROUTED_EXPERTS,
LLM_KV_DSV41_N_SHARED_EXPERTS,
LLM_KV_DSV41_NUM_EXPERTS_PER_TOK,
LLM_KV_DSV41_MAX_POSITION_EMBEDDINGS,
LLM_KV_DSV41_SLIDING_WINDOW,
LLM_KV_DSV41_INDEX_N_HEADS,
LLM_KV_DSV41_INDEX_HEAD_DIM,
LLM_KV_DSV41_INDEX_TOPK,
LLM_KV_DSV41_CANDIDATE_SOURCE_LAYER_ID,
LLM_KV_DSV41_CANDIDATE_TOPK_BLOCKS,
LLM_KV_DSV41_CANDIDATE_BLOCK_SIZE,
LLM_KV_DSV41_HC_MULT,
LLM_KV_DSV41_HC_SINKHORN_ITERS,
LLM_KV_DSV41_ROPE_THETA,
LLM_KV_DSV41_COMPRESS_ROPE_THETA,
LLM_KV_DSV41_RMS_NORM_EPS,
LLM_KV_DSV41_HC_EPS,
LLM_KV_DSV41_SWIGLU_LIMIT,
LLM_KV_DSV41_ROUTED_SCALING_FACTOR,
LLM_KV_DSV41_SCORING_FUNC,
LLM_KV_DSV41_HIDDEN_ACT,
LLM_KV_DSV41_TOPK_METHOD,
LLM_KV_DSV41_NORM_TOPK_PROB,
LLM_KV_DSV41_COMPRESS_RATIOS,
LLM_KV_DSV41_KV_SOURCE_LAYER_IDS,
LLM_KV_DSV41_INDEX_SOURCE_LAYER_IDS,
LLM_KV_DSV41_ROPE_SCALING_FACTOR,
LLM_KV_DSV41_ROPE_SCALING_BETA_FAST,
LLM_KV_DSV41_ROPE_SCALING_BETA_SLOW,
LLM_KV_DSV41_ROPE_SCALING_ORIG_CTX_LEN,
LLM_KV_DSV41_ENGRAM_ENCODING,
LLM_KV_DSV41_ENGRAM_LAYER_IDS,
LLM_KV_DSV41_ENGRAM_ROWS,
LLM_KV_DSV41_ENGRAM_COMPRESSED_VOCAB_SIZE,
LLM_KV_DSV41_ENGRAM_PAD_ID,
LLM_KV_DSV41_ENGRAM_TOKEN_MAP,
LLM_KV_DSV41_ENGRAM_PRIMES,
LLM_KV_DSV41_ENGRAM_MULTIPLIERS,

LLM_KV_HASH_LAYER_COUNT,

LLM_KV_ROPE_DIMENSION_COUNT,
Expand Down Expand Up @@ -610,6 +662,10 @@ enum llm_tensor {
LLM_TENSOR_PLE_NORM_QUERY, // qwen4exp
LLM_TENSOR_PLE_NORM_CONV, // qwen4exp
LLM_TENSOR_PLE_CONV1D, // qwen4exp
LLM_TENSOR_ENGRAM_EMBD,
LLM_TENSOR_ENGRAM_Q_NORM,
LLM_TENSOR_ENGRAM_K_NORM,
LLM_TENSOR_ENGRAM_KV,
LLM_TENSOR_ATTN_COMPRESSOR_WKV,
LLM_TENSOR_ATTN_COMPRESSOR_WGATE,
LLM_TENSOR_ATTN_COMPRESSOR_APE,
Expand Down
Loading
Loading