From 70985986933c92cfdeb4fbb0d66ec0e755d1128f Mon Sep 17 00:00:00 2001 From: Jerome Coste Date: Sat, 12 Sep 2026 08:38:51 -0700 Subject: [PATCH 1/3] deepseek41 : add compressed attention runtime Assisted-by: GPT-5.6 Sol Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- src/CMakeLists.txt | 1 + src/llama-dsv41.cpp | 545 ++++++++++++++++++++++++++++++ src/llama-dsv41.h | 234 +++++++++++++ src/llama-hparams.cpp | 23 ++ src/llama-hparams.h | 13 + src/llama-model.cpp | 38 ++- src/models/deepseek41.cpp | 224 ++++++++++++ src/models/models.h | 8 + tests/CMakeLists.txt | 1 + tests/test-deepseek41-runtime.cpp | 298 ++++++++++++++++ tests/test-deepseek41-schema.cpp | 1 + tests/test-llama-archs.cpp | 2 +- 12 files changed, 1385 insertions(+), 3 deletions(-) create mode 100644 src/llama-dsv41.cpp create mode 100644 src/llama-dsv41.h create mode 100644 src/models/deepseek41.cpp create mode 100644 tests/test-deepseek41-runtime.cpp diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 255e8fae1efb..6094ca17339a 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -17,6 +17,7 @@ add_library(llama llama-chat.cpp llama-context.cpp llama-cparams.cpp + llama-dsv41.cpp llama-grammar.cpp llama-graph.cpp llama-hparams.cpp diff --git a/src/llama-dsv41.cpp b/src/llama-dsv41.cpp new file mode 100644 index 000000000000..8b54af6a2c11 --- /dev/null +++ b/src/llama-dsv41.cpp @@ -0,0 +1,545 @@ +#include "llama-dsv41.h" + +#include "ggml.h" + +#include +#include +#include +#include +#include +#include + +static constexpr int32_t DSV41_KV_SOURCES[] = { 2, 8, 14, 20 }; +static constexpr int32_t DSV41_INDEX_SOURCES[] = { 2, 8, 14, 20, 24, 28, 32, 36 }; +static constexpr uint32_t DSV41_ENGRAM_LAYERS[] = { 1, 14 }; +static constexpr uint32_t DSV41_ENGRAM_ROWS[] = { 30000000, 5000000 }; + +static void dsv41_require(bool condition, const char * message) { + if (!condition) { + throw std::runtime_error(std::string("DeepSeek V4.1 metadata: ") + message); + } +} + +void llama_dsv41_validate_config(const llama_dsv41_config & config) { + dsv41_require(config.n_ctx_train == LLAMA_DSV41_N_CTX, "max_position_embeddings must be 1048576"); + dsv41_require(config.n_embd == LLAMA_DSV41_N_EMBD, "hidden_size must be 5120"); + dsv41_require(config.n_layer == LLAMA_DSV41_N_LAYER, "num_hidden_layers must be 40"); + dsv41_require(config.n_vocab == LLAMA_DSV41_N_VOCAB, "vocab_size must be 129280"); + dsv41_require(config.n_head == LLAMA_DSV41_N_HEAD, "num_attention_heads must be 64"); + dsv41_require(config.n_head_kv == LLAMA_DSV41_N_HEAD_KV, "num_key_value_heads must be 1"); + dsv41_require(config.n_head_dim == LLAMA_DSV41_N_HEAD_DIM, "head_dim must be 512"); + dsv41_require(config.n_rot == LLAMA_DSV41_N_ROT, "qk_rope_head_dim must be 64"); + dsv41_require(config.n_lora_q == LLAMA_DSV41_N_LORA_Q, "q_lora_rank must be 1280"); + dsv41_require(config.n_lora_o == LLAMA_DSV41_N_LORA_O, "o_lora_rank must be 1024"); + dsv41_require(config.n_o_group == LLAMA_DSV41_N_O_GROUP, "o_group_num must be 8"); + dsv41_require(config.n_ff_dense == LLAMA_DSV41_N_FF_DENSE, "intermediate_size must be 18432"); + dsv41_require(config.n_ff_expert == LLAMA_DSV41_N_FF_EXP, "moe_intermediate_size must be 2304"); + dsv41_require(config.n_expert == LLAMA_DSV41_N_EXPERT, "n_routed_experts must be 384"); + dsv41_require(config.n_expert_used == LLAMA_DSV41_N_EXPERT_USED, "num_experts_per_tok must be 6"); + dsv41_require(config.n_expert_shared == LLAMA_DSV41_N_EXPERT_SHARED, "n_shared_experts must be 1"); + dsv41_require(config.indexer_n_head == LLAMA_DSV41_N_INDEX_HEAD, "index_n_heads must be 32"); + dsv41_require(config.indexer_head_size == LLAMA_DSV41_N_INDEX_HEAD_DIM, "index_head_dim must be 128"); + dsv41_require(config.indexer_top_k == LLAMA_DSV41_N_INDEX_TOP_K, "index_topk must be 512"); + dsv41_require(config.hc_count == LLAMA_DSV41_HC_MULT, "hc_num_streams must be 4"); + dsv41_require(config.hc_sinkhorn_iters == LLAMA_DSV41_HC_SINKHORN_ITERS, "hc_sinkhorn_iters must be 20"); + dsv41_require(config.raw_window == LLAMA_DSV41_N_SWA, "sliding_window must be 128"); + dsv41_require(config.candidate_source_layer == LLAMA_DSV41_CANDIDATE_SOURCE_LAYER, "candidate_source_layer must be 20"); + dsv41_require(config.candidate_topk_blocks == LLAMA_DSV41_CANDIDATE_TOPK_BLOCKS, "candidate_topk_blocks must be 2048"); + dsv41_require(config.candidate_block_size == LLAMA_DSV41_CANDIDATE_BLOCK_SIZE, "candidate_block_size must be 8"); + dsv41_require(config.f_norm_rms_eps == 1.0e-20f, "rms_norm_eps must be 1e-20"); + dsv41_require(config.hc_eps == 1.0e-6f, "hc_eps must be 1e-6"); + dsv41_require(config.swiglu_clamp == 10.0f, "swiglu_clamp_limit must be 10"); + dsv41_require(config.routed_scale == 1.5f, "routed_scaling_factor must be 1.5"); + dsv41_require(config.rope_theta == 10000.0f, "rope_theta must be 10000"); + dsv41_require(config.compress_rope_theta == 160000.0f, "compress_rope_theta must be 160000"); + dsv41_require(config.yarn_factor == 16.0f, "rope_scaling.factor must be 16"); + dsv41_require(config.yarn_beta_fast == 32.0f, "rope_scaling.beta_fast must be 32"); + dsv41_require(config.yarn_beta_slow == 1.0f, "rope_scaling.beta_slow must be 1"); + dsv41_require(config.yarn_original_context == 65536, "rope_scaling.original_max_position_embeddings must be 65536"); + dsv41_require(config.expert_weights_norm, "norm_topk_prob must be true"); + dsv41_require(config.hidden_act == "silu", "hidden_act must be silu"); + dsv41_require(config.scoring_func == "sqrtsoftplus", "scoring_func must be sqrtsoftplus"); + dsv41_require(config.topk_method == "noaux_tc", "topk_method must be noaux_tc"); + + dsv41_require(config.compress_ratios.size() == LLAMA_DSV41_N_LAYER, "compress_ratios must have 40 entries"); + for (uint32_t il = 0; il < LLAMA_DSV41_N_LAYER; ++il) { + dsv41_require(config.compress_ratios[il] == llama_dsv41_compress_ratio(il), "compress_ratios has an invalid main-layer value"); + } + dsv41_require(config.kv_sources == std::vector(std::begin(DSV41_KV_SOURCES), std::end(DSV41_KV_SOURCES)), "kv_source_layers must be [2,8,14,20]"); + dsv41_require(config.index_sources == std::vector(std::begin(DSV41_INDEX_SOURCES), std::end(DSV41_INDEX_SOURCES)), "index_source_layers must be [2,8,14,20,24,28,32,36]"); + dsv41_require(config.engram_layers == std::vector(std::begin(DSV41_ENGRAM_LAYERS), std::end(DSV41_ENGRAM_LAYERS)), "engram.layer_ids must be [1,14]"); + dsv41_require(config.engram_rows == std::vector(std::begin(DSV41_ENGRAM_ROWS), std::end(DSV41_ENGRAM_ROWS)), "engram.rows must be [30000000,5000000]"); + dsv41_require(config.engram_encoding == LLAMA_DSV41_ENGRAM_ENCODING, "engram.encoding must be e4m3_e8m0_32_row264"); + dsv41_require(config.engram_compressed_vocab_size == LLAMA_DSV41_ENGRAM_COMPRESSED_VOCAB, "engram.compressed_vocab_size must be 99092"); + dsv41_require(config.engram_pad_id == LLAMA_DSV41_ENGRAM_PAD_ID, "engram.pad_id must be 2"); + dsv41_require(config.engram_token_map_size == LLAMA_DSV41_N_VOCAB, "engram.token_map must contain 129280 entries"); + dsv41_require(config.engram_primes_size == LLAMA_DSV41_ENGRAM_PRIMES_COUNT, "engram.primes must contain 48 entries"); + dsv41_require(config.engram_multipliers_size == LLAMA_DSV41_ENGRAM_MULTIPLIERS_COUNT, "engram.multipliers must contain 8 entries"); +} + +const char * llama_dsv41_runtime_dependency_error() { + return "DeepSeek V4.1 execution requires disk-backed Engram and routed-expert streaming support"; +} + +static int32_t dsv41_source_layer(const int32_t * sources, size_t n, uint32_t il) { + int32_t result = -1; + for (size_t i = 0; i < n; ++i) { + if ((uint32_t) sources[i] > il) { + break; + } + result = sources[i]; + } + return result; +} + +llama_dsv41_cache_state::llama_dsv41_cache_state(uint32_t compressed_cache_size) : + compressed_cache_size(compressed_cache_size), + raw(LLAMA_DSV41_N_SWA, -1) { + if (compressed_cache_size == 0) { + throw std::runtime_error("DeepSeek V4.1 compressed cache is empty"); + } + for (int32_t source : DSV41_KV_SOURCES) { + const uint32_t ratio = llama_dsv41_compress_ratio(source); + sources.emplace(source, source_state { + ratio, + std::vector(compressed_cache_size, -1), + std::vector(ratio, -1), + }); + } +} + +void llama_dsv41_cache_state::clear() { + pos = -1; + std::fill(raw.begin(), raw.end(), -1); + for (auto & item : sources) { + std::fill(item.second.compressed.begin(), item.second.compressed.end(), -1); + std::fill(item.second.pending.begin(), item.second.pending.end(), -1); + } + candidates.clear(); +} + +void llama_dsv41_cache_state::append(llama_pos next_pos) { + if (next_pos != pos + 1) { + throw std::runtime_error("DeepSeek V4.1 cache requires contiguous single-sequence positions"); + } + + for (const auto & item : sources) { + const source_state & state = item.second; + if ((next_pos + 1)%state.ratio == 0 && (uint64_t) (next_pos/state.ratio) >= compressed_cache_size) { + throw std::runtime_error("DeepSeek V4.1 compressed cache overflow"); + } + } + + raw[next_pos%raw.size()] = next_pos; + for (auto & item : sources) { + source_state & state = item.second; + state.pending[next_pos%state.ratio] = next_pos; + if ((next_pos + 1)%state.ratio != 0) { + continue; + } + + const uint64_t dst = next_pos/state.ratio; + state.compressed[dst] = next_pos + 1 - state.ratio; + } + pos = next_pos; +} + +void llama_dsv41_cache_state::set_candidate_blocks(const std::vector & blocks) { + candidates = blocks; +} + +llama_pos llama_dsv41_cache_state::position() const { + return pos; +} + +const std::vector & llama_dsv41_cache_state::raw_slots() const { + return raw; +} + +const std::vector & llama_dsv41_cache_state::compressed_slots(uint32_t source_layer) const { + const auto it = sources.find(source_layer); + if (it == sources.end()) { + throw std::runtime_error("DeepSeek V4.1 compressed cache source layer is invalid"); + } + return it->second.compressed; +} + +const std::vector & llama_dsv41_cache_state::pending_slots(uint32_t source_layer) const { + const auto it = sources.find(source_layer); + if (it == sources.end()) { + throw std::runtime_error("DeepSeek V4.1 compressor state source layer is invalid"); + } + return it->second.pending; +} + +const std::vector & llama_dsv41_cache_state::candidate_blocks() const { + return candidates; +} + +uint64_t llama_dsv41_memory_accounting::total() const { + return raw_kv + compressed_kv + index_keys + compressor_carry + + candidate_scores + candidate_ids + position_state + graph_workspace; +} + +llama_dsv41_memory_accounting llama_dsv41_account_memory( + uint32_t n_ctx, + uint32_t n_seq, + uint32_t n_tokens, + uint32_t kv_element_size, + uint32_t index_element_size, + uint64_t graph_workspace) { + if (n_ctx == 0 || n_seq == 0 || n_tokens == 0 || kv_element_size == 0 || index_element_size == 0) { + throw std::runtime_error("DeepSeek V4.1 memory accounting dimensions must be non-zero"); + } + + llama_dsv41_memory_accounting result; + const uint64_t raw_rows = (uint64_t) LLAMA_DSV41_N_LAYER*LLAMA_DSV41_N_SWA*n_seq; + const uint64_t ratio_2_rows = ((uint64_t) n_ctx + 1)/2; + const uint64_t ratio_1_rows = n_ctx; + const uint64_t compressed_rows = (3*ratio_2_rows + ratio_1_rows)*n_seq; + const uint64_t candidate_blocks = ((uint64_t) n_ctx + LLAMA_DSV41_CANDIDATE_BLOCK_SIZE - 1)/ + LLAMA_DSV41_CANDIDATE_BLOCK_SIZE; + uint64_t pending_rows = 0; + uint64_t gated_pending_rows = 0; + for (int32_t source : DSV41_KV_SOURCES) { + const uint32_t ratio = llama_dsv41_compress_ratio(source); + pending_rows += ratio; + if (ratio == 2) { + gated_pending_rows += ratio; + } + } + + result.raw_kv = raw_rows*LLAMA_DSV41_N_HEAD_DIM*kv_element_size; + result.compressed_kv = compressed_rows*LLAMA_DSV41_N_HEAD_DIM*kv_element_size; + result.index_keys = compressed_rows*LLAMA_DSV41_N_INDEX_HEAD_DIM*index_element_size; + result.compressor_carry = + (pending_rows + gated_pending_rows)*LLAMA_DSV41_N_HEAD_DIM*kv_element_size*n_seq; + result.candidate_scores = candidate_blocks*sizeof(float)*n_tokens; + result.candidate_ids = std::min(candidate_blocks, LLAMA_DSV41_CANDIDATE_TOPK_BLOCKS)* + sizeof(int32_t)*n_tokens; + const uint64_t position_rows = + ((uint64_t) LLAMA_DSV41_N_SWA + compressed_rows/n_seq + pending_rows)*n_seq; + result.position_state = position_rows*sizeof(llama_pos); + result.graph_workspace = graph_workspace; + return result; +} + +int32_t llama_dsv41_kv_source_layer(uint32_t il) { + return dsv41_source_layer(DSV41_KV_SOURCES, sizeof(DSV41_KV_SOURCES)/sizeof(DSV41_KV_SOURCES[0]), il); +} + +int32_t llama_dsv41_index_source_layer(uint32_t il) { + return dsv41_source_layer(DSV41_INDEX_SOURCES, sizeof(DSV41_INDEX_SOURCES)/sizeof(DSV41_INDEX_SOURCES[0]), il); +} + +uint32_t llama_dsv41_compress_ratio(uint32_t il) { + if (il < 2) { + return 0; + } + if (il < 20) { + return 2; + } + if (il < LLAMA_DSV41_N_LAYER) { + return 1; + } + throw std::runtime_error("DeepSeek V4.1 layer index is out of range"); +} + +llama_dsv41_layer_plan llama_dsv41_build_layer_plan( + uint32_t il, + const std::vector & positions, + uint32_t compressed_cache_size) { + if (positions.empty()) { + throw std::runtime_error("DeepSeek V4.1 graph plan requires at least one token"); + } + if (positions.front() < 0) { + throw std::runtime_error("DeepSeek V4.1 graph plan requires non-negative positions"); + } + for (size_t i = 1; i < positions.size(); ++i) { + if (positions[i] != positions[i - 1] + 1) { + throw std::runtime_error("DeepSeek V4.1 graph plan requires one contiguous sequence"); + } + } + + llama_dsv41_layer_plan plan = {}; + plan.layer = il; + plan.ratio = llama_dsv41_compress_ratio(il); + plan.kv_source_layer = llama_dsv41_kv_source_layer(il); + plan.index_source_layer = llama_dsv41_index_source_layer(il); + plan.owns_kv_source = plan.kv_source_layer == (int32_t) il; + plan.owns_index_source = plan.index_source_layer == (int32_t) il; + plan.builds_candidates = il == LLAMA_DSV41_CANDIDATE_SOURCE_LAYER; + plan.uses_candidates = plan.owns_index_source && il > LLAMA_DSV41_CANDIDATE_SOURCE_LAYER; + plan.reuses_index_selection = !plan.owns_index_source && plan.index_source_layer >= 0; + plan.collapses_output = il + 1 == LLAMA_DSV41_N_LAYER; + plan.raw_ring_order = llama_dsv41_raw_ring_order(positions.back(), LLAMA_DSV41_N_SWA); + if (plan.owns_kv_source && plan.ratio != 0) { + plan.compression = llama_dsv41_build_compression_plan(positions, plan.ratio, compressed_cache_size); + } + return plan; +} + +llama_dsv41_compression_plan llama_dsv41_build_compression_plan( + const std::vector & positions, + uint32_t ratio, + uint32_t cache_size) { + if (ratio != 1 && ratio != 2) { + throw std::runtime_error("DeepSeek V4.1 compression ratio must be 1 or 2"); + } + if (cache_size == 0) { + throw std::runtime_error("DeepSeek V4.1 compressed cache is empty"); + } + + llama_dsv41_compression_plan plan; + plan.n_visible.resize(positions.size(), 0); + + std::vector latest_state_src(ratio, -1); + std::vector latest_state_pos(ratio, -1); + + const int32_t scratch_offset = ratio; + for (size_t i = 0; i < positions.size(); ++i) { + const llama_pos pos = positions[i]; + if (pos < 0) { + continue; + } + + plan.n_visible[i] = (int32_t) ((pos + 1)/ratio); + plan.n_kv = std::max(plan.n_kv, plan.n_visible[i]); + + if (ratio == 1) { + if ((uint64_t) pos >= cache_size) { + throw std::runtime_error("DeepSeek V4.1 compressed cache overflow"); + } + plan.write_idxs.push_back(pos); + plan.write_pos.push_back((int32_t) pos); + continue; + } + + const int32_t row = (int32_t) (pos%ratio); + plan.state_pos.push_back(row); + if (latest_state_src[row] < 0 || pos > latest_state_pos[row]) { + latest_state_src[row] = (int32_t) i; + latest_state_pos[row] = pos; + } + + if ((pos + 1)%ratio != 0) { + continue; + } + + const llama_pos source_start = pos + 1 - ratio; + const int64_t write_idx = pos/ratio; + if ((uint64_t) write_idx >= cache_size) { + throw std::runtime_error("DeepSeek V4.1 compressed cache overflow"); + } + + for (uint32_t j = 0; j < ratio; ++j) { + const llama_pos source_pos = source_start + j; + int32_t source_idx = (int32_t) (source_pos%ratio); + for (size_t k = 0; k < positions.size(); ++k) { + if (positions[k] == source_pos) { + source_idx = scratch_offset + (int32_t) k; + break; + } + } + plan.state_read_idxs.push_back(source_idx); + } + + plan.write_idxs.push_back(write_idx); + plan.write_pos.push_back((int32_t) source_start); + } + + if (ratio == 2) { + for (uint32_t row = 0; row < ratio; ++row) { + if (latest_state_src[row] >= 0) { + plan.state_persist_src_idxs.push_back(latest_state_src[row]); + plan.state_persist_dst_idxs.push_back((int32_t) row); + } + } + } + + plan.n_kv = std::min(cache_size, std::max(1, plan.n_kv)); + return plan; +} + +std::vector llama_dsv41_select_candidate_blocks( + const std::vector & scores, + uint32_t n_visible, + uint32_t block_size, + uint32_t top_k_blocks) { + if (block_size == 0) { + throw std::runtime_error("DeepSeek V4.1 candidate block size must be non-zero"); + } + + n_visible = std::min(n_visible, scores.size()); + if (n_visible == 0 || top_k_blocks == 0) { + return {}; + } + + const uint32_t n_blocks = (n_visible + block_size - 1)/block_size; + std::vector block_scores(n_blocks, -std::numeric_limits::infinity()); + for (uint32_t block = 0; block < n_blocks; ++block) { + const uint32_t i0 = block*block_size; + const uint32_t i1 = std::min(n_visible, i0 + block_size); + for (uint32_t i = i0; i < i1; ++i) { + if (std::isnan(scores[i])) { + throw std::runtime_error("DeepSeek V4.1 candidate score is NaN"); + } + block_scores[block] = std::max(block_scores[block], scores[i]); + } + } + + if (n_visible%block_size != 0) { + block_scores.back() = std::numeric_limits::infinity(); + } + + std::vector blocks(n_blocks); + std::iota(blocks.begin(), blocks.end(), 0); + std::stable_sort(blocks.begin(), blocks.end(), [&](int32_t a, int32_t b) { + if (block_scores[a] != block_scores[b]) { + return block_scores[a] > block_scores[b]; + } + return a < b; + }); + + blocks.resize(std::min(top_k_blocks, n_blocks)); + return blocks; +} + +std::vector llama_dsv41_candidate_rows( + const std::vector & blocks, + uint32_t n_visible, + uint32_t block_size) { + if (block_size == 0) { + throw std::runtime_error("DeepSeek V4.1 candidate block size must be non-zero"); + } + + std::vector rows; + for (int32_t block : blocks) { + if (block < 0) { + throw std::runtime_error("DeepSeek V4.1 candidate block index must be non-negative"); + } + const uint64_t i0 = (uint64_t) block*block_size; + const uint64_t i1 = std::min(n_visible, i0 + block_size); + for (uint64_t i = i0; i < i1; ++i) { + rows.push_back((int32_t) i); + } + } + return rows; +} + +std::vector llama_dsv41_raw_ring_order(llama_pos pos, uint32_t window) { + if (window == 0 || pos < 0) { + return {}; + } + + const uint32_t n_raw = std::min((uint64_t) pos + 1, window); + const uint32_t start = (uint32_t) ((pos + 1 - n_raw)%window); + + std::vector result(n_raw); + for (uint32_t i = 0; i < n_raw; ++i) { + result[i] = (start + i)%window; + } + return result; +} + +std::vector llama_dsv41_output_collapse( + const std::vector & residual, + const std::vector & pre, + uint32_t n_embd, + uint32_t hc_mult) { + if (pre.size() != hc_mult || residual.size() != (size_t) n_embd*hc_mult) { + throw std::runtime_error("DeepSeek V4.1 output collapse shape mismatch"); + } + + std::vector result(n_embd, 0.0f); + for (uint32_t h = 0; h < hc_mult; ++h) { + for (uint32_t d = 0; d < n_embd; ++d) { + result[d] += residual[(size_t) h*n_embd + d]*pre[h]; + } + } + return result; +} + +ggml_tensor * llama_dsv41_build_ratio_pool( + ggml_context * ctx, + ggml_tensor * kv, + ggml_tensor * gate, + uint32_t ratio) { + if (ratio != 1 && ratio != 2) { + throw std::runtime_error("DeepSeek V4.1 graph compression ratio must be 1 or 2"); + } + if (kv->ne[1] != ratio) { + throw std::runtime_error("DeepSeek V4.1 graph compressor input shape mismatch"); + } + if (ratio == 1) { + return ggml_reshape_2d(ctx, kv, kv->ne[0], kv->ne[2]); + } + if (gate == nullptr || !ggml_are_same_shape(kv, gate)) { + throw std::runtime_error("DeepSeek V4.1 ratio-2 graph requires matching KV and gate tensors"); + } + + ggml_tensor * kv_t = ggml_cont(ctx, ggml_permute(ctx, kv, 1, 0, 2, 3)); + ggml_tensor * gate_t = ggml_cont(ctx, ggml_permute(ctx, gate, 1, 0, 2, 3)); + ggml_tensor * weights = ggml_soft_max(ctx, gate_t); + ggml_tensor * pooled = ggml_sum_rows(ctx, ggml_mul(ctx, kv_t, weights)); + return ggml_reshape_2d(ctx, pooled, kv->ne[0], kv->ne[2]); +} + +ggml_tensor * llama_dsv41_build_shared_softmax( + ggml_context * ctx, + ggml_tensor * raw_scores, + ggml_tensor * compressed_scores) { + if (raw_scores == nullptr && compressed_scores == nullptr) { + throw std::runtime_error("DeepSeek V4.1 attention requires raw or compressed scores"); + } + ggml_tensor * scores = raw_scores; + if (scores == nullptr) { + scores = compressed_scores; + } else if (compressed_scores != nullptr) { + if (raw_scores->ne[1] != compressed_scores->ne[1] || + raw_scores->ne[2] != compressed_scores->ne[2] || + raw_scores->ne[3] != compressed_scores->ne[3]) { + throw std::runtime_error("DeepSeek V4.1 raw and compressed score shapes are incompatible"); + } + scores = ggml_concat(ctx, raw_scores, compressed_scores, 0); + } + return ggml_soft_max(ctx, scores); +} + +ggml_tensor * llama_dsv41_build_output_collapse( + ggml_context * ctx, + ggml_tensor * residual, + ggml_tensor * pre, + uint32_t n_embd, + uint32_t hc_mult, + uint32_t n_tokens) { + if (residual->ne[0] != n_embd || residual->ne[1] != hc_mult || residual->ne[2] != n_tokens || + pre->ne[0] != hc_mult || pre->ne[1] != n_tokens) { + throw std::runtime_error("DeepSeek V4.1 graph output collapse shape mismatch"); + } + + ggml_tensor * residual_t = ggml_cont(ctx, ggml_permute(ctx, residual, 1, 0, 2, 3)); + ggml_tensor * pre_t = ggml_reshape_3d(ctx, pre, hc_mult, 1, n_tokens); + ggml_tensor * collapsed = ggml_sum_rows(ctx, ggml_mul(ctx, residual_t, pre_t)); + collapsed = ggml_reshape_2d(ctx, collapsed, n_embd, n_tokens); + return ggml_cast(ctx, collapsed, GGML_TYPE_BF16); +} + +ggml_tensor * llama_dsv41_build_output( + ggml_context * ctx, + ggml_tensor * residual, + ggml_tensor * pre, + ggml_tensor * output_norm, + ggml_tensor * output, + float rms_eps, + uint32_t hc_mult) { + if (output_norm->ne[0] != residual->ne[0] || output->ne[0] != residual->ne[0]) { + throw std::runtime_error("DeepSeek V4.1 output tensor shape mismatch"); + } + + ggml_tensor * collapsed = llama_dsv41_build_output_collapse( + ctx, residual, pre, residual->ne[0], hc_mult, residual->ne[2]); + ggml_tensor * normalized = ggml_rms_norm(ctx, collapsed, rms_eps); + normalized = ggml_mul(ctx, normalized, output_norm); + return ggml_mul_mat(ctx, output, normalized); +} diff --git a/src/llama-dsv41.h b/src/llama-dsv41.h new file mode 100644 index 000000000000..ebbd0a577a67 --- /dev/null +++ b/src/llama-dsv41.h @@ -0,0 +1,234 @@ +#pragma once + +#include "llama.h" + +#include +#include +#include +#include + +struct ggml_context; +struct ggml_tensor; + +static constexpr uint32_t LLAMA_DSV41_N_LAYER = 40; +static constexpr uint32_t LLAMA_DSV41_N_EMBD = 5120; +static constexpr uint32_t LLAMA_DSV41_N_VOCAB = 129280; +static constexpr uint32_t LLAMA_DSV41_N_HEAD = 64; +static constexpr uint32_t LLAMA_DSV41_N_HEAD_KV = 1; +static constexpr uint32_t LLAMA_DSV41_N_HEAD_DIM = 512; +static constexpr uint32_t LLAMA_DSV41_N_ROT = 64; +static constexpr uint32_t LLAMA_DSV41_N_LORA_Q = 1280; +static constexpr uint32_t LLAMA_DSV41_N_LORA_O = 1024; +static constexpr uint32_t LLAMA_DSV41_N_O_GROUP = 8; +static constexpr uint32_t LLAMA_DSV41_N_FF_DENSE = 18432; +static constexpr uint32_t LLAMA_DSV41_N_EXPERT = 384; +static constexpr uint32_t LLAMA_DSV41_N_EXPERT_USED = 6; +static constexpr uint32_t LLAMA_DSV41_N_EXPERT_SHARED = 1; +static constexpr uint32_t LLAMA_DSV41_N_FF_EXP = 2304; +static constexpr uint32_t LLAMA_DSV41_N_INDEX_HEAD = 32; +static constexpr uint32_t LLAMA_DSV41_N_INDEX_HEAD_DIM = 128; +static constexpr uint32_t LLAMA_DSV41_N_INDEX_TOP_K = 512; +static constexpr uint32_t LLAMA_DSV41_N_SWA = 128; +static constexpr uint32_t LLAMA_DSV41_N_CTX = 1048576; +static constexpr uint32_t LLAMA_DSV41_HC_MULT = 4; +static constexpr uint32_t LLAMA_DSV41_HC_SINKHORN_ITERS = 20; +static constexpr uint32_t LLAMA_DSV41_CANDIDATE_SOURCE_LAYER = 20; +static constexpr uint32_t LLAMA_DSV41_CANDIDATE_TOPK_BLOCKS = 2048; +static constexpr uint32_t LLAMA_DSV41_CANDIDATE_BLOCK_SIZE = 8; +static constexpr uint32_t LLAMA_DSV41_ENGRAM_COMPRESSED_VOCAB = 99092; +static constexpr uint32_t LLAMA_DSV41_ENGRAM_PAD_ID = 2; +static constexpr uint32_t LLAMA_DSV41_ENGRAM_PRIMES_COUNT = 48; +static constexpr uint32_t LLAMA_DSV41_ENGRAM_MULTIPLIERS_COUNT = 8; +static constexpr const char * LLAMA_DSV41_ENGRAM_ENCODING = "e4m3_e8m0_32_row264"; + +struct llama_dsv41_config { + uint32_t n_ctx_train; + uint32_t n_embd; + uint32_t n_layer; + uint32_t n_vocab; + uint32_t n_head; + uint32_t n_head_kv; + uint32_t n_head_dim; + uint32_t n_rot; + uint32_t n_lora_q; + uint32_t n_lora_o; + uint32_t n_o_group; + uint32_t n_ff_dense; + uint32_t n_ff_expert; + uint32_t n_expert; + uint32_t n_expert_used; + uint32_t n_expert_shared; + uint32_t indexer_n_head; + uint32_t indexer_head_size; + uint32_t indexer_top_k; + uint32_t hc_count; + uint32_t hc_sinkhorn_iters; + uint32_t raw_window; + uint32_t candidate_source_layer; + uint32_t candidate_topk_blocks; + uint32_t candidate_block_size; + float f_norm_rms_eps; + float hc_eps; + float swiglu_clamp; + float routed_scale; + float rope_theta; + float compress_rope_theta; + float yarn_factor; + float yarn_beta_fast; + float yarn_beta_slow; + uint32_t yarn_original_context; + bool expert_weights_norm; + std::string hidden_act; + std::string scoring_func; + std::string topk_method; + std::vector compress_ratios; + std::vector kv_sources; + std::vector index_sources; + std::vector engram_layers; + std::vector engram_rows; + std::string engram_encoding; + uint32_t engram_compressed_vocab_size; + uint32_t engram_pad_id; + uint32_t engram_token_map_size; + uint32_t engram_primes_size; + uint32_t engram_multipliers_size; +}; + +void llama_dsv41_validate_config(const llama_dsv41_config & config); +const char * llama_dsv41_runtime_dependency_error(); + +struct llama_dsv41_compression_plan { + std::vector state_pos; + std::vector state_persist_src_idxs; + std::vector state_persist_dst_idxs; + std::vector state_read_idxs; + std::vector write_idxs; + std::vector write_pos; + std::vector n_visible; + int64_t n_kv = 0; +}; + +struct llama_dsv41_layer_plan { + uint32_t layer; + uint32_t ratio; + int32_t kv_source_layer; + int32_t index_source_layer; + bool owns_kv_source; + bool owns_index_source; + bool builds_candidates; + bool uses_candidates; + bool reuses_index_selection; + bool collapses_output; + std::vector raw_ring_order; + llama_dsv41_compression_plan compression; +}; + +class llama_dsv41_cache_state { +public: + explicit llama_dsv41_cache_state(uint32_t compressed_cache_size); + + void clear(); + void append(llama_pos pos); + void set_candidate_blocks(const std::vector & blocks); + + llama_pos position() const; + const std::vector & raw_slots() const; + const std::vector & compressed_slots(uint32_t source_layer) const; + const std::vector & pending_slots(uint32_t source_layer) const; + const std::vector & candidate_blocks() const; + +private: + struct source_state { + uint32_t ratio; + std::vector compressed; + std::vector pending; + }; + + uint32_t compressed_cache_size; + llama_pos pos = -1; + std::vector raw; + std::map sources; + std::vector candidates; +}; + +struct llama_dsv41_memory_accounting { + uint64_t raw_kv = 0; + uint64_t compressed_kv = 0; + uint64_t index_keys = 0; + uint64_t compressor_carry = 0; + uint64_t candidate_scores = 0; + uint64_t candidate_ids = 0; + uint64_t position_state = 0; + uint64_t graph_workspace = 0; + + uint64_t total() const; +}; + +llama_dsv41_memory_accounting llama_dsv41_account_memory( + uint32_t n_ctx, + uint32_t n_seq, + uint32_t n_tokens, + uint32_t kv_element_size, + uint32_t index_element_size, + uint64_t graph_workspace); + +int32_t llama_dsv41_kv_source_layer(uint32_t il); +int32_t llama_dsv41_index_source_layer(uint32_t il); +uint32_t llama_dsv41_compress_ratio(uint32_t il); + +llama_dsv41_layer_plan llama_dsv41_build_layer_plan( + uint32_t il, + const std::vector & positions, + uint32_t compressed_cache_size); + +llama_dsv41_compression_plan llama_dsv41_build_compression_plan( + const std::vector & positions, + uint32_t ratio, + uint32_t cache_size); + +std::vector llama_dsv41_select_candidate_blocks( + const std::vector & scores, + uint32_t n_visible, + uint32_t block_size, + uint32_t top_k_blocks); + +std::vector llama_dsv41_candidate_rows( + const std::vector & blocks, + uint32_t n_visible, + uint32_t block_size); + +std::vector llama_dsv41_raw_ring_order(llama_pos pos, uint32_t window); + +std::vector llama_dsv41_output_collapse( + const std::vector & residual, + const std::vector & pre, + uint32_t n_embd, + uint32_t hc_mult); + +ggml_tensor * llama_dsv41_build_ratio_pool( + ggml_context * ctx, + ggml_tensor * kv, + ggml_tensor * gate, + uint32_t ratio); + +ggml_tensor * llama_dsv41_build_shared_softmax( + ggml_context * ctx, + ggml_tensor * raw_scores, + ggml_tensor * compressed_scores); + +ggml_tensor * llama_dsv41_build_output_collapse( + ggml_context * ctx, + ggml_tensor * residual, + ggml_tensor * pre, + uint32_t n_embd, + uint32_t hc_mult, + uint32_t n_tokens); + +ggml_tensor * llama_dsv41_build_output( + ggml_context * ctx, + ggml_tensor * residual, + ggml_tensor * pre, + ggml_tensor * output_norm, + ggml_tensor * output, + float rms_eps, + uint32_t hc_mult); diff --git a/src/llama-hparams.cpp b/src/llama-hparams.cpp index 34b3c688019a..b3d49ac482f7 100644 --- a/src/llama-hparams.cpp +++ b/src/llama-hparams.cpp @@ -1,5 +1,6 @@ #include "llama-hparams.h" +#include "llama-dsv41.h" #include "ggml.h" #include @@ -47,6 +48,28 @@ bool llama_hparams::is_swa_any() const { return false; } +int32_t llama_hparams::dsv41_kv_source(uint32_t il) const { + if (il >= n_layer()) { + GGML_ABORT("fatal error"); + } + return dsv41_kv_source_layer[il]; +} + +int32_t llama_hparams::dsv41_index_source(uint32_t il) const { + if (il >= n_layer()) { + GGML_ABORT("fatal error"); + } + return dsv41_index_source_layer[il]; +} + +bool llama_hparams::dsv41_is_kv_source(uint32_t il) const { + return dsv41_kv_source(il) == (int32_t) il; +} + +bool llama_hparams::dsv41_is_index_source(uint32_t il) const { + return dsv41_index_source(il) == (int32_t) il; +} + uint32_t llama_hparams::n_head(uint32_t il) const { if (il < n_layer_all) { return n_head_arr[il]; diff --git a/src/llama-hparams.h b/src/llama-hparams.h index 3afa49ebe861..268e28809494 100644 --- a/src/llama-hparams.h +++ b/src/llama-hparams.h @@ -294,6 +294,19 @@ struct llama_hparams { float dsv4_hc_eps = 0.0f; std::array dsv4_compress_ratios; + // DeepSeek-V4.1 + uint32_t dsv41_candidate_source_layer = 0; + uint32_t dsv41_candidate_topk_blocks = 0; + uint32_t dsv41_candidate_block_size = 0; + std::array dsv41_kv_source_layer; + std::array dsv41_index_source_layer; + std::bitset dsv41_engram_layers; + + int32_t dsv41_kv_source(uint32_t il) const; + int32_t dsv41_index_source(uint32_t il) const; + bool dsv41_is_kv_source(uint32_t il) const; + bool dsv41_is_index_source(uint32_t il) const; + // 0 = full rank (DeepSeek-V4) uint32_t hc_low_rank = 0; diff --git a/src/llama-model.cpp b/src/llama-model.cpp index 734bd3edce98..0d4ee5ba136f 100644 --- a/src/llama-model.cpp +++ b/src/llama-model.cpp @@ -6,6 +6,7 @@ #include "llama-impl.h" #include "llama-mmap.h" #include "llama-cparams.h" +#include "llama-dsv41.h" #include "llama-model-loader.h" #include "llama-kv-cache.h" @@ -200,6 +201,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params return new llama_model_dots3note(params); case LLM_ARCH_DEEPSEEK4: return new llama_model_deepseek4(params); + case LLM_ARCH_DEEPSEEK41: + return new llama_model_deepseek41(params); case LLM_ARCH_GLM_DSA: return new llama_model_glm_dsa(params); case LLM_ARCH_MISTRAL4: @@ -371,6 +374,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str const llama_hparams & hparams = ud->model->hparams; const std::string tensor_name = tensor->name; const bool is_dsv4 = ud->model->arch == LLM_ARCH_DEEPSEEK4 || + ud->model->arch == LLM_ARCH_DEEPSEEK41 || (ud->model->arch == LLM_ARCH_DFLASH && hparams.dsv4_hc_mult > 0); static const std::regex pattern_q_weight ("blk\\.\\d*\\.attn_q.weight"); @@ -382,7 +386,7 @@ struct ggml_backend_meta_split_state llama_meta_device_get_split_state(const str static const std::regex pattern_qk_norm ("blk\\.\\d*\\.attn_(q|k)_norm\\.weight"); static const std::regex pattern_kv_cache ("cache_(k|v)_l\\d*"); static const std::regex pattern_idx_cache ("cache_idx_(k|v)_l\\d*"); - static const std::regex pattern_dsv4_state ("dsv4_(csa|hca|lid)_state_(kv|score)_l\\d*"); + static const std::regex pattern_dsv4_state ("dsv4(1)?_(csa|hca|lid|comp|index)_state_(kv|score)_l\\d*"); static const std::regex pattern_attn_sinks ("blk\\.\\d*\\.attn_sinks.weight"); static const std::regex pattern_attn_out_weight ("blk\\.\\d*\\.attn_output.weight"); static const std::regex pattern_attn_out_bias ("blk\\.\\d*\\.attn_output.bias"); @@ -1219,6 +1223,33 @@ void llama_model_base::load_hparams(llama_model_loader & ml) { return; } + if (arch == LLM_ARCH_DEEPSEEK41) { + std::fill(hparams.n_head_arr.begin(), hparams.n_head_arr.end(), 0); + std::fill(hparams.n_head_kv_arr.begin(), hparams.n_head_kv_arr.end(), 0); + std::fill(hparams.n_ff_arr.begin(), hparams.n_ff_arr.end(), 0); + std::fill(hparams.n_ff_exp_arr.begin(), hparams.n_ff_exp_arr.end(), 0); + std::fill(hparams.n_expert_used_arr.begin(), hparams.n_expert_used_arr.end(), 0); + std::fill(hparams.rope_sections.begin(), hparams.rope_sections.end(), 0); + std::fill(hparams.rope_pattern.begin(), hparams.rope_pattern.end(), 1); + std::fill(hparams.is_swa_impl.begin(), hparams.is_swa_impl.end(), 0); + std::fill(hparams.is_recr_impl.begin(), hparams.is_recr_impl.end(), 0); + std::fill(hparams.is_indexer_full_impl.begin(), hparams.is_indexer_full_impl.end(), 0); + std::fill(hparams.dsv41_kv_source_layer.begin(), hparams.dsv41_kv_source_layer.end(), -1); + std::fill(hparams.dsv41_index_source_layer.begin(), hparams.dsv41_index_source_layer.end(), -1); + std::fill(hparams.dsv4_compress_ratios.begin(), hparams.dsv4_compress_ratios.end(), 0); + std::fill(hparams.swiglu_clamp_exp.begin(), hparams.swiglu_clamp_exp.end(), 0.0f); + std::fill(hparams.swiglu_clamp_shexp.begin(), hparams.swiglu_clamp_shexp.end(), 0.0f); + hparams.dsv41_engram_layers.reset(); + + load_arch_hparams(ml); + + pimpl->n_bytes = ml.n_bytes; + pimpl->desc_str = arch_name() + " " + type_name() + " " + ml.ftype_name(); + pimpl->ftype = ml.ftype; + hparams.rope_type = llama_model_rope_type(this); + return; + } + ml.get_key(LLM_KV_CONTEXT_LENGTH, hparams.n_ctx_train); ml.get_key(LLM_KV_EMBEDDING_LENGTH, hparams.n_embd); ml.get_key(LLM_KV_EMBEDDING_LENGTH_OUT, hparams.n_embd_out_impl, false); @@ -2463,6 +2494,8 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params, nullptr); } } break; + case LLM_ARCH_DEEPSEEK41: + throw std::runtime_error(llama_dsv41_runtime_dependency_error()); case LLM_ARCH_DFLASH: { // DSV4 DSpark stages store a single MLA-style K per position (window = the draft ring) @@ -2821,7 +2854,7 @@ int32_t llama_model_n_head_kv(const llama_model * model) { int32_t llama_model_n_swa(const llama_model * model) { // dsv4 kv-cache has SWA but it cannot be used as a rollback because of // other compression ratios, so we return 0 here - if (model->arch == LLM_ARCH_DEEPSEEK4) { + if (model->arch == LLM_ARCH_DEEPSEEK4 || model->arch == LLM_ARCH_DEEPSEEK41) { return 0; } return model->hparams.n_swa; @@ -2907,6 +2940,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) { case LLM_ARCH_DEEPSEEK2OCR: case LLM_ARCH_DEEPSEEK32: case LLM_ARCH_DEEPSEEK4: + case LLM_ARCH_DEEPSEEK41: case LLM_ARCH_MUSE_GLIMMER: case LLM_ARCH_PLM: case LLM_ARCH_CHATGLM: diff --git a/src/models/deepseek41.cpp b/src/models/deepseek41.cpp new file mode 100644 index 000000000000..aba6126e0ca9 --- /dev/null +++ b/src/models/deepseek41.cpp @@ -0,0 +1,224 @@ +#include "llama-dsv41.h" +#include "llama-hparams.h" +#include "models.h" + +#include +#include +#include +#include +#include + +static float dsv41_rope_attn_factor(float freq_scale) { + return 1.0f/(1.0f + 0.1f*logf(1.0f/freq_scale)); +} + +void llama_model_deepseek41::load_arch_hparams(llama_model_loader & ml) { + llama_dsv41_config config = {}; + std::string raw_config; + ml.get_key(LLM_KV_DSV41_CONFIG, raw_config); + ml.get_key(LLM_KV_DSV41_MAX_POSITION_EMBEDDINGS, config.n_ctx_train); + ml.get_key(LLM_KV_DSV41_HIDDEN_SIZE, config.n_embd); + ml.get_key(LLM_KV_DSV41_NUM_HIDDEN_LAYERS, config.n_layer); + ml.get_key(LLM_KV_DSV41_VOCAB_SIZE, config.n_vocab); + ml.get_key(LLM_KV_DSV41_NUM_ATTENTION_HEADS, config.n_head); + ml.get_key(LLM_KV_DSV41_NUM_KEY_VALUE_HEADS, config.n_head_kv); + ml.get_key(LLM_KV_DSV41_HEAD_DIM, config.n_head_dim); + ml.get_key(LLM_KV_DSV41_QK_ROPE_HEAD_DIM, config.n_rot); + ml.get_key(LLM_KV_DSV41_Q_LORA_RANK, config.n_lora_q); + ml.get_key(LLM_KV_DSV41_O_LORA_RANK, config.n_lora_o); + ml.get_key(LLM_KV_DSV41_O_GROUPS, config.n_o_group); + ml.get_key(LLM_KV_DSV41_MOE_INTERMEDIATE_SIZE, config.n_ff_expert); + ml.get_key(LLM_KV_DSV41_N_ROUTED_EXPERTS, config.n_expert); + ml.get_key(LLM_KV_DSV41_NUM_EXPERTS_PER_TOK, config.n_expert_used); + ml.get_key(LLM_KV_DSV41_N_SHARED_EXPERTS, config.n_expert_shared); + ml.get_key(LLM_KV_DSV41_INDEX_N_HEADS, config.indexer_n_head); + ml.get_key(LLM_KV_DSV41_INDEX_HEAD_DIM, config.indexer_head_size); + ml.get_key(LLM_KV_DSV41_INDEX_TOPK, config.indexer_top_k); + ml.get_key(LLM_KV_DSV41_HC_MULT, config.hc_count); + ml.get_key(LLM_KV_DSV41_HC_SINKHORN_ITERS, config.hc_sinkhorn_iters); + ml.get_key(LLM_KV_DSV41_SLIDING_WINDOW, config.raw_window); + ml.get_key(LLM_KV_DSV41_CANDIDATE_SOURCE_LAYER_ID, config.candidate_source_layer); + ml.get_key(LLM_KV_DSV41_CANDIDATE_TOPK_BLOCKS, config.candidate_topk_blocks); + ml.get_key(LLM_KV_DSV41_CANDIDATE_BLOCK_SIZE, config.candidate_block_size); + ml.get_key(LLM_KV_DSV41_RMS_NORM_EPS, config.f_norm_rms_eps); + ml.get_key(LLM_KV_DSV41_HC_EPS, config.hc_eps); + ml.get_key(LLM_KV_DSV41_SWIGLU_LIMIT, config.swiglu_clamp); + ml.get_key(LLM_KV_DSV41_ROUTED_SCALING_FACTOR, config.routed_scale); + ml.get_key(LLM_KV_DSV41_ROPE_THETA, config.rope_theta); + ml.get_key(LLM_KV_DSV41_COMPRESS_ROPE_THETA, config.compress_rope_theta); + ml.get_key(LLM_KV_DSV41_ROPE_SCALING_FACTOR, config.yarn_factor); + ml.get_key(LLM_KV_DSV41_ROPE_SCALING_BETA_FAST, config.yarn_beta_fast); + ml.get_key(LLM_KV_DSV41_ROPE_SCALING_BETA_SLOW, config.yarn_beta_slow); + ml.get_key(LLM_KV_DSV41_ROPE_SCALING_ORIG_CTX_LEN, config.yarn_original_context); + ml.get_key(LLM_KV_DSV41_NORM_TOPK_PROB, config.expert_weights_norm); + ml.get_key(LLM_KV_DSV41_HIDDEN_ACT, config.hidden_act); + ml.get_key(LLM_KV_DSV41_SCORING_FUNC, config.scoring_func); + ml.get_key(LLM_KV_DSV41_TOPK_METHOD, config.topk_method); + ml.get_arr(LLM_KV_DSV41_COMPRESS_RATIOS, config.compress_ratios); + ml.get_arr(LLM_KV_DSV41_KV_SOURCE_LAYER_IDS, config.kv_sources); + ml.get_arr(LLM_KV_DSV41_INDEX_SOURCE_LAYER_IDS, config.index_sources); + ml.get_key(LLM_KV_DSV41_ENGRAM_ENCODING, config.engram_encoding); + ml.get_arr(LLM_KV_DSV41_ENGRAM_LAYER_IDS, config.engram_layers); + ml.get_arr(LLM_KV_DSV41_ENGRAM_ROWS, config.engram_rows); + ml.get_key(LLM_KV_DSV41_ENGRAM_COMPRESSED_VOCAB_SIZE, config.engram_compressed_vocab_size); + ml.get_key(LLM_KV_DSV41_ENGRAM_PAD_ID, config.engram_pad_id); + ml.get_arr_n(LLM_KV_DSV41_ENGRAM_TOKEN_MAP, config.engram_token_map_size); + ml.get_arr_n(LLM_KV_DSV41_ENGRAM_PRIMES, config.engram_primes_size); + ml.get_arr_n(LLM_KV_DSV41_ENGRAM_MULTIPLIERS, config.engram_multipliers_size); + + config.n_ff_dense = LLAMA_DSV41_N_FF_DENSE; + llama_dsv41_validate_config(config); + + if (raw_config.empty()) { + throw std::runtime_error("DeepSeek V4.1 metadata: config must not be empty"); + } + hparams.n_ctx_train = config.n_ctx_train; + hparams.n_embd = config.n_embd; + hparams.n_embd_out_impl = config.n_embd; + hparams.n_layer_all = config.n_layer; + hparams.n_layer_nextn = 0; + hparams.n_expert = config.n_expert; + hparams.n_expert_shared = config.n_expert_shared; + hparams.n_lora_q = config.n_lora_q; + hparams.n_ff_shexp = config.n_ff_expert; + hparams.n_embd_head_k_full = config.n_head_dim; + hparams.n_embd_head_v_full = config.n_head_dim; + hparams.n_embd_head_k_swa = config.n_head_dim; + hparams.n_embd_head_v_swa = config.n_head_dim; + hparams.n_rot_full = config.n_rot; + hparams.n_rot_swa = config.n_rot; + hparams.n_swa = config.raw_window; + hparams.indexer_n_head = config.indexer_n_head; + hparams.indexer_head_size = config.indexer_head_size; + hparams.indexer_top_k = config.indexer_top_k; + hparams.dsv4_o_group_count = config.n_o_group; + hparams.dsv4_o_lora_rank = config.n_lora_o; + hparams.dsv4_hc_mult = config.hc_count; + hparams.dsv4_hc_sinkhorn_iters = config.hc_sinkhorn_iters; + hparams.dsv4_compress_rope_base = config.compress_rope_theta; + hparams.dsv4_hc_eps = config.hc_eps; + hparams.dsv41_candidate_source_layer = config.candidate_source_layer; + hparams.dsv41_candidate_topk_blocks = config.candidate_topk_blocks; + hparams.dsv41_candidate_block_size = config.candidate_block_size; + hparams.f_norm_rms_eps = config.f_norm_rms_eps; + hparams.expert_weights_scale = config.routed_scale; + hparams.expert_weights_norm = config.expert_weights_norm; + hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SQRT_SOFTPLUS; + hparams.rope_freq_base_train = config.rope_theta; + hparams.rope_freq_base_train_swa = config.rope_theta; + hparams.rope_freq_scale_train = 1.0f/config.yarn_factor; + hparams.rope_freq_scale_train_swa = hparams.rope_freq_scale_train; + hparams.n_ctx_orig_yarn = config.yarn_original_context; + hparams.yarn_beta_fast = config.yarn_beta_fast; + hparams.yarn_beta_slow = config.yarn_beta_slow; + hparams.yarn_ext_factor = 1.0f; + hparams.rope_attn_factor = dsv41_rope_attn_factor(hparams.rope_freq_scale_train); + hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; + hparams.causal_attn = true; + + for (uint32_t il = 0; il < config.n_layer; ++il) { + hparams.n_head_arr[il] = config.n_head; + hparams.n_head_kv_arr[il] = config.n_head_kv; + hparams.n_ff_arr[il] = config.n_ff_dense; + hparams.n_ff_exp_arr[il] = config.n_ff_expert; + hparams.n_expert_used_arr[il] = config.n_expert_used; + hparams.swiglu_clamp_exp[il] = config.swiglu_clamp; + hparams.swiglu_clamp_shexp[il] = config.swiglu_clamp; + hparams.dsv4_compress_ratios[il] = config.compress_ratios[il]; + hparams.dsv41_kv_source_layer[il] = llama_dsv41_kv_source_layer(il); + hparams.dsv41_index_source_layer[il] = llama_dsv41_index_source_layer(il); + hparams.is_swa_impl[il] = 1; + } + for (uint32_t il : config.engram_layers) { + hparams.dsv41_engram_layers.set(il); + } + + type = LLM_TYPE_UNKNOWN; +} + +[[noreturn]] void llama_model_deepseek41::load_arch_tensors(llama_model_loader & ml) { + LLAMA_LOAD_LOCALS; + + const int64_t q_lora_rank = hparams.n_lora_q; + const int64_t n_ff_exp = hparams.n_ff_exp(); + const int64_t n_expert_shared = hparams.n_expert_shared; + const int64_t n_embd_head = hparams.n_embd_head_k(); + const int64_t o_groups = hparams.dsv4_o_group_count; + const int64_t o_lora_rank = hparams.dsv4_o_lora_rank; + const int64_t hc_mult = hparams.dsv4_hc_mult; + const int64_t hc_dim = hc_mult*n_embd; + const int64_t hc_mix_dim = (2 + hc_mult)*hc_mult; + + tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), { n_embd, n_vocab }, 0); + output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), { n_embd }, 0); + output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), { n_embd, n_vocab }, 0); + + for (int32_t il = 0; il < n_layer; ++il) { + auto & layer = layers[il]; + + layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", il), { n_embd }, 0); + layer.attn_sinks = create_tensor(tn(LLM_TENSOR_ATTN_SINKS, "weight", il), { n_head }, 0); + layer.wq_a = create_tensor(tn(LLM_TENSOR_ATTN_Q_A, "weight", il), { n_embd, q_lora_rank }, 0); + layer.attn_q_a_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_A_NORM, "weight", il), { q_lora_rank }, 0); + layer.wq_b = create_tensor(tn(LLM_TENSOR_ATTN_Q_B, "weight", il), { q_lora_rank, n_head*n_embd_head }, 0); + layer.wkv = create_tensor(tn(LLM_TENSOR_ATTN_KV, "weight", il), { n_embd, n_embd_head }, 0); + layer.attn_kv_a_norm = create_tensor(tn(LLM_TENSOR_ATTN_KV_A_NORM, "weight", il), { n_embd_head }, 0); + layer.wo_a = create_tensor(tn(LLM_TENSOR_ATTN_OUT_A, "weight", il), { n_head*n_embd_head/o_groups, o_lora_rank, o_groups }, TENSOR_ALLOW_RESHAPE); + layer.wo_b = create_tensor(tn(LLM_TENSOR_ATTN_OUT_B, "weight", il), { o_groups*o_lora_rank, n_embd }, 0); + + layer.hc_attn_fn = create_tensor(tn(LLM_TENSOR_HC_ATTN_FN, "weight", il), { hc_dim, hc_mix_dim }, 0); + layer.hc_attn_base = create_tensor(tn(LLM_TENSOR_HC_ATTN_BASE, "weight", il), { hc_mix_dim }, 0); + layer.hc_attn_scale = create_tensor(tn(LLM_TENSOR_HC_ATTN_SCALE, "weight", il), { 3 }, 0); + layer.hc_ffn_fn = create_tensor(tn(LLM_TENSOR_HC_FFN_FN, "weight", il), { hc_dim, hc_mix_dim }, 0); + layer.hc_ffn_base = create_tensor(tn(LLM_TENSOR_HC_FFN_BASE, "weight", il), { hc_mix_dim }, 0); + layer.hc_ffn_scale = create_tensor(tn(LLM_TENSOR_HC_FFN_SCALE, "weight", il), { 3 }, 0); + + if (hparams.dsv41_is_kv_source(il)) { + layer.attn_comp_wkv = create_tensor(tn(LLM_TENSOR_ATTN_COMPRESSOR_WKV, "weight", il), { n_embd, n_embd_head }, 0); + layer.attn_comp_norm = create_tensor(tn(LLM_TENSOR_ATTN_COMPRESSOR_NORM, "weight", il), { n_embd_head }, 0); + if (hparams.dsv4_compress_ratios[il] == 2) { + layer.attn_comp_wgate = create_tensor(tn(LLM_TENSOR_ATTN_COMPRESSOR_WGATE, "weight", il), { n_embd, n_embd_head }, 0); + } + layer.indexer_attn_k = create_tensor(tn(LLM_TENSOR_INDEXER_ATTN_K, "weight", il), { n_embd_head, hparams.indexer_head_size }, 0); + layer.indexer_k_norm = create_tensor(tn(LLM_TENSOR_INDEXER_K_NORM, "weight", il), { hparams.indexer_head_size }, 0); + } + if (hparams.dsv41_is_index_source(il)) { + layer.indexer_proj = create_tensor(tn(LLM_TENSOR_INDEXER_PROJ, "weight", il), { n_embd, hparams.indexer_n_head }, 0); + layer.indexer_attn_q_b = create_tensor(tn(LLM_TENSOR_INDEXER_ATTN_Q_B, "weight", il), { q_lora_rank, hparams.indexer_n_head*hparams.indexer_head_size }, 0); + } + + layer.ffn_gate_inp = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP, "weight", il), { n_embd, n_expert }, 0); + layer.ffn_exp_probs_b = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias", il), { n_expert }, 0); + layer.ffn_exp_probs_b_vl = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B_VL, "bias", il), { n_expert }, TENSOR_NOT_REQUIRED); + layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", il), { n_embd }, 0); + layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", il), { n_embd, n_ff_exp, n_expert }, 0); + layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", il), { n_ff_exp, n_embd, n_expert }, 0); + layer.ffn_up_exps = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS, "weight", il), { n_embd, n_ff_exp, n_expert }, 0); + layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", il), { n_embd, n_ff_exp*n_expert_shared }, 0); + layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", il), { n_ff_exp*n_expert_shared, n_embd }, 0); + layer.ffn_up_shexp = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP, "weight", il), { n_embd, n_ff_exp*n_expert_shared }, 0); + + if (hparams.dsv41_engram_layers.test(il)) { + const llm_tensor engram_tensors[] = { + LLM_TENSOR_ENGRAM_EMBD, + LLM_TENSOR_ENGRAM_Q_NORM, + LLM_TENSOR_ENGRAM_K_NORM, + LLM_TENSOR_ENGRAM_KV, + }; + for (llm_tensor tensor : engram_tensors) { + const std::string name = tn(tensor, "weight", il).str(); + if (ml.get_weight(name.c_str()) == nullptr) { + throw std::runtime_error("DeepSeek V4.1 is missing required Engram tensor " + name); + } + } + } + } + + throw std::runtime_error( + std::string("DeepSeek V4.1 tensor metadata is valid, but tensor payloads cannot be mapped: ") + + llama_dsv41_runtime_dependency_error()); +} + +[[noreturn]] std::unique_ptr llama_model_deepseek41::build_arch_graph(const llm_graph_params &) const { + throw std::runtime_error(llama_dsv41_runtime_dependency_error()); +} diff --git a/src/models/models.h b/src/models/models.h index 4bc2d46b6b78..231936009d49 100644 --- a/src/models/models.h +++ b/src/models/models.h @@ -1313,6 +1313,14 @@ struct llama_model_deepseek4 : public llama_model_base { std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; }; +struct llama_model_deepseek41 : public llama_model_deepseek4 { + llama_model_deepseek41(const struct llama_model_params & params) : llama_model_deepseek4(params) {} + void load_arch_hparams(llama_model_loader & ml) override; + [[noreturn]] void load_arch_tensors(llama_model_loader & ml) override; + + [[noreturn]] std::unique_ptr build_arch_graph(const llm_graph_params & params) const override; +}; + struct llama_model_deepseek2ocr : public llama_model_base { llama_model_deepseek2ocr(const struct llama_model_params & params) : llama_model_base(params) {} diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index ea937784c5a2..ec74cf2d3492 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -197,6 +197,7 @@ if (NOT WIN32 OR NOT BUILD_SHARED_LIBS) # llama_build_and_test(test-double-float.cpp) # SLOW llama_build_and_test(test-deepseek41-schema.cpp) + llama_build_and_test(test-deepseek41-runtime.cpp) llama_build_and_test(test-llama-archs.cpp) set(MODEL_DIR "${CMAKE_CURRENT_BINARY_DIR}/test-models/") diff --git a/tests/test-deepseek41-runtime.cpp b/tests/test-deepseek41-runtime.cpp new file mode 100644 index 000000000000..49289c409d35 --- /dev/null +++ b/tests/test-deepseek41-runtime.cpp @@ -0,0 +1,298 @@ +#include "../src/llama-dsv41.h" + +#include "ggml.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +static void check(bool condition, const std::string & message) { + if (!condition) { + std::fprintf(stderr, "%s\n", message.c_str()); + std::exit(1); + } +} + +static void expect_throw(const std::function & fn, const std::string & message) { + try { + fn(); + } catch (const std::runtime_error &) { + return; + } + check(false, message); +} + +static llama_dsv41_config valid_config() { + llama_dsv41_config config = {}; + config.n_ctx_train = LLAMA_DSV41_N_CTX; + config.n_embd = LLAMA_DSV41_N_EMBD; + config.n_layer = LLAMA_DSV41_N_LAYER; + config.n_vocab = LLAMA_DSV41_N_VOCAB; + config.n_head = LLAMA_DSV41_N_HEAD; + config.n_head_kv = LLAMA_DSV41_N_HEAD_KV; + config.n_head_dim = LLAMA_DSV41_N_HEAD_DIM; + config.n_rot = LLAMA_DSV41_N_ROT; + config.n_lora_q = LLAMA_DSV41_N_LORA_Q; + config.n_lora_o = LLAMA_DSV41_N_LORA_O; + config.n_o_group = LLAMA_DSV41_N_O_GROUP; + config.n_ff_dense = LLAMA_DSV41_N_FF_DENSE; + config.n_ff_expert = LLAMA_DSV41_N_FF_EXP; + config.n_expert = LLAMA_DSV41_N_EXPERT; + config.n_expert_used = LLAMA_DSV41_N_EXPERT_USED; + config.n_expert_shared = LLAMA_DSV41_N_EXPERT_SHARED; + config.indexer_n_head = LLAMA_DSV41_N_INDEX_HEAD; + config.indexer_head_size = LLAMA_DSV41_N_INDEX_HEAD_DIM; + config.indexer_top_k = LLAMA_DSV41_N_INDEX_TOP_K; + config.hc_count = LLAMA_DSV41_HC_MULT; + config.hc_sinkhorn_iters = LLAMA_DSV41_HC_SINKHORN_ITERS; + config.raw_window = LLAMA_DSV41_N_SWA; + config.candidate_source_layer = LLAMA_DSV41_CANDIDATE_SOURCE_LAYER; + config.candidate_topk_blocks = LLAMA_DSV41_CANDIDATE_TOPK_BLOCKS; + config.candidate_block_size = LLAMA_DSV41_CANDIDATE_BLOCK_SIZE; + config.f_norm_rms_eps = 1.0e-20f; + config.hc_eps = 1.0e-6f; + config.swiglu_clamp = 10.0f; + config.routed_scale = 1.5f; + config.rope_theta = 10000.0f; + config.compress_rope_theta = 160000.0f; + config.yarn_factor = 16.0f; + config.yarn_beta_fast = 32.0f; + config.yarn_beta_slow = 1.0f; + config.yarn_original_context = 65536; + config.expert_weights_norm = true; + config.hidden_act = "silu"; + config.scoring_func = "sqrtsoftplus"; + config.topk_method = "noaux_tc"; + for (uint32_t il = 0; il < LLAMA_DSV41_N_LAYER; ++il) { + config.compress_ratios.push_back(llama_dsv41_compress_ratio(il)); + } + config.kv_sources = { 2, 8, 14, 20 }; + config.index_sources = { 2, 8, 14, 20, 24, 28, 32, 36 }; + config.engram_layers = { 1, 14 }; + config.engram_rows = { 30000000, 5000000 }; + config.engram_encoding = LLAMA_DSV41_ENGRAM_ENCODING; + config.engram_compressed_vocab_size = LLAMA_DSV41_ENGRAM_COMPRESSED_VOCAB; + config.engram_pad_id = LLAMA_DSV41_ENGRAM_PAD_ID; + config.engram_token_map_size = LLAMA_DSV41_N_VOCAB; + config.engram_primes_size = LLAMA_DSV41_ENGRAM_PRIMES_COUNT; + config.engram_multipliers_size = LLAMA_DSV41_ENGRAM_MULTIPLIERS_COUNT; + return config; +} + +static void test_hparams() { + llama_dsv41_validate_config(valid_config()); + + llama_dsv41_config config = valid_config(); + config.compress_ratios.insert(config.compress_ratios.end(), 3, 0); + expect_throw([&]() { llama_dsv41_validate_config(config); }, "43-entry source-config compression layout was accepted"); + + config = valid_config(); + config.compress_ratios[20] = 2; + expect_throw([&]() { llama_dsv41_validate_config(config); }, "invalid ratio-1 boundary was accepted"); + + config = valid_config(); + config.kv_sources = { 2, 8, 20 }; + expect_throw([&]() { llama_dsv41_validate_config(config); }, "invalid KV source map was accepted"); + + config = valid_config(); + config.engram_primes_size = 24; + expect_throw([&]() { llama_dsv41_validate_config(config); }, "truncated Engram prime table was accepted"); + + const std::string dependency_error = llama_dsv41_runtime_dependency_error(); + check(dependency_error.find("disk-backed Engram") != std::string::npos, "dependency error omits Engram"); + check(dependency_error.find("routed-expert streaming") != std::string::npos, "dependency error omits expert streaming"); +} + +static void test_source_maps() { + check(llama_dsv41_compress_ratio(0) == 0, "layer 0 ratio mismatch"); + check(llama_dsv41_compress_ratio(1) == 0, "layer 1 ratio mismatch"); + check(llama_dsv41_compress_ratio(2) == 2, "layer 2 ratio mismatch"); + check(llama_dsv41_compress_ratio(19) == 2, "layer 19 ratio mismatch"); + check(llama_dsv41_compress_ratio(20) == 1, "layer 20 ratio mismatch"); + + check(llama_dsv41_kv_source_layer(0) == -1, "layer 0 unexpectedly has a KV source"); + check(llama_dsv41_kv_source_layer(2) == 2, "layer 2 KV source mismatch"); + check(llama_dsv41_kv_source_layer(7) == 2, "layer 7 KV source mismatch"); + check(llama_dsv41_kv_source_layer(8) == 8, "layer 8 KV source mismatch"); + check(llama_dsv41_kv_source_layer(19) == 14, "layer 19 KV source mismatch"); + check(llama_dsv41_kv_source_layer(39) == 20, "layer 39 KV source mismatch"); + + check(llama_dsv41_index_source_layer(19) == 14, "layer 19 index source mismatch"); + check(llama_dsv41_index_source_layer(20) == 20, "layer 20 index source mismatch"); + check(llama_dsv41_index_source_layer(23) == 20, "layer 23 index source mismatch"); + check(llama_dsv41_index_source_layer(24) == 24, "layer 24 index source mismatch"); + check(llama_dsv41_index_source_layer(39) == 36, "layer 39 index source mismatch"); +} + +static void test_compression() { + const auto ratio_2 = llama_dsv41_build_compression_plan({ 0, 1, 2 }, 2, 1024); + check(ratio_2.n_visible == std::vector({ 0, 1, 1 }), "ratio-2 visible counts mismatch"); + check(ratio_2.write_idxs == std::vector({ 0 }), "ratio-2 write index mismatch"); + check(ratio_2.write_pos == std::vector({ 0 }), "ratio-2 compressed position mismatch"); + check(ratio_2.state_persist_dst_idxs == std::vector({ 0, 1 }), "ratio-2 state rows mismatch"); + + const auto ratio_1 = llama_dsv41_build_compression_plan({ 19, 20 }, 1, 1024); + check(ratio_1.n_visible == std::vector({ 20, 21 }), "ratio-1 visible counts mismatch"); + check(ratio_1.write_idxs == std::vector({ 19, 20 }), "ratio-1 write indexes mismatch"); + check(ratio_1.write_pos == std::vector({ 19, 20 }), "ratio-1 compressed positions mismatch"); + + const auto layer_0 = llama_dsv41_build_layer_plan(0, { 0 }, 1024); + check(layer_0.ratio == 0 && layer_0.compression.write_idxs.empty(), "layer 0 must use raw attention only"); + const auto layer_2 = llama_dsv41_build_layer_plan(2, { 0, 1 }, 1024); + check(layer_2.ratio == 2 && layer_2.owns_kv_source, "layer 2 compression ownership mismatch"); + check(layer_2.compression.write_idxs == std::vector({ 0 }), "layer 2 graph compression mismatch"); + const auto layer_20 = llama_dsv41_build_layer_plan(20, { 20 }, 1024); + check(layer_20.ratio == 1 && layer_20.owns_kv_source, "layer 20 compression ownership mismatch"); + check(layer_20.builds_candidates && !layer_20.uses_candidates, "layer 20 candidate propagation mismatch"); + const auto layer_21 = llama_dsv41_build_layer_plan(21, { 21 }, 1024); + check(!layer_21.uses_candidates && layer_21.reuses_index_selection, "layer 21 index reuse mismatch"); + const auto layer_24 = llama_dsv41_build_layer_plan(24, { 24 }, 1024); + check(!layer_24.owns_kv_source && layer_24.owns_index_source, "layer 24 source ownership mismatch"); + check(layer_24.uses_candidates, "layer 24 must consume layer-20 candidates"); + check(llama_dsv41_build_layer_plan(39, { 39 }, 1024).collapses_output, "final layer output collapse missing"); + expect_throw([&]() { llama_dsv41_build_layer_plan(20, { 20, 22 }, 1024); }, "non-contiguous graph plan was accepted"); +} + +static void test_state() { + llama_dsv41_cache_state state(1024); + for (llama_pos pos = 0; pos <= 129; ++pos) { + state.append(pos); + } + check(state.position() == 129, "cache position mismatch"); + check(state.raw_slots()[0] == 128 && state.raw_slots()[1] == 129, "raw ring state mismatch"); + check(state.compressed_slots(2)[0] == 0, "ratio-2 first compressed row mismatch"); + check(state.compressed_slots(2)[64] == 128, "ratio-2 boundary row mismatch"); + check(state.pending_slots(2) == std::vector({ 128, 129 }), "ratio-2 pending rows mismatch"); + check(state.compressed_slots(20)[129] == 129, "ratio-1 direct row mismatch"); + state.set_candidate_blocks({ 4, 1 }); + check(state.candidate_blocks() == std::vector({ 4, 1 }), "candidate state mismatch"); + expect_throw([&]() { state.append(131); }, "non-contiguous cache append was accepted"); + state.clear(); + check(state.position() == -1 && state.raw_slots()[0] == -1, "cache clear mismatch"); + + llama_dsv41_cache_state small(1); + small.append(0); + expect_throw([&]() { small.append(1); }, "compressed cache overflow was accepted"); + check(small.position() == 0 && small.raw_slots()[1] == -1, "failed cache append mutated state"); + + const auto bytes = llama_dsv41_account_memory(32768, 1, 8192, 2, 2, 1234); + check(bytes.raw_kv > 0 && bytes.compressed_kv > 0 && bytes.index_keys > 0, "cache memory accounting is incomplete"); + check(bytes.compressor_carry > 0 && bytes.candidate_scores > 0 && bytes.candidate_ids > 0, "state memory accounting is incomplete"); + check(bytes.total() == bytes.raw_kv + bytes.compressed_kv + bytes.index_keys + bytes.compressor_carry + + bytes.candidate_scores + bytes.candidate_ids + bytes.position_state + bytes.graph_workspace, + "memory accounting total mismatch"); +} + +static void test_raw_ring() { + const auto at_127 = llama_dsv41_raw_ring_order(127, 128); + check(at_127.size() == 128 && at_127.front() == 0 && at_127.back() == 127, "raw ring at 127 mismatch"); + const auto at_128 = llama_dsv41_raw_ring_order(128, 128); + check(at_128.front() == 1 && at_128.back() == 0, "raw ring at 128 mismatch"); + const auto at_129 = llama_dsv41_raw_ring_order(129, 128); + check(at_129.front() == 2 && at_129.back() == 1, "raw ring at 129 mismatch"); +} + +static void test_candidates() { + for (uint32_t n_visible : { 1u, 7u, 8u, 9u, 127u, 16385u, 17017u }) { + std::vector scores(n_visible); + for (uint32_t i = 0; i < n_visible; ++i) { + scores[i] = -(float) i; + } + const auto blocks = llama_dsv41_select_candidate_blocks(scores, n_visible, 8, 2048); + const int32_t final_block = (int32_t) ((n_visible - 1)/8); + check(std::find(blocks.begin(), blocks.end(), final_block) != blocks.end(), "final partial candidate block was dropped"); + check(blocks.size() == std::min(2048, (n_visible + 7)/8), "candidate block count mismatch"); + const auto rows = llama_dsv41_candidate_rows(blocks, n_visible, 8); + check(std::find(rows.begin(), rows.end(), (int32_t) n_visible - 1) != rows.end(), "final visible row was filtered"); + check(std::all_of(rows.begin(), rows.end(), [&](int32_t row) { return row >= 0 && (uint32_t) row < n_visible; }), "candidate rows crossed causal visibility"); + } + + const auto tie = llama_dsv41_select_candidate_blocks(std::vector(24, 1.0f), 24, 8, 2); + check(tie == std::vector({ 0, 1 }), "candidate tie-break mismatch"); + + std::vector partial_scores(9, -100.0f); + partial_scores[0] = 100.0f; + const auto partial = llama_dsv41_select_candidate_blocks(partial_scores, 9, 8, 1); + check(partial == std::vector({ 1 }), "final partial candidate block was not forced"); +} + +static void test_output_collapse() { + const std::vector residual = { + 1.0f, 2.0f, + 3.0f, 4.0f, + 5.0f, 6.0f, + 7.0f, 8.0f, + }; + const auto result = llama_dsv41_output_collapse(residual, { 0.1f, 0.2f, 0.3f, 0.4f }, 2, 4); + check(result.size() == 2, "output collapse width mismatch"); + check(std::abs(result[0] - 5.0f) < 1.0e-6f, "output collapse first value mismatch"); + check(std::abs(result[1] - 6.0f) < 1.0e-6f, "output collapse second value mismatch"); +} + +static void test_graph_construction() { + ggml_init_params params = { + /*.mem_size =*/ 4*1024*1024, + /*.mem_buffer =*/ nullptr, + /*.no_alloc =*/ false, + }; + ggml_context * ctx = ggml_init(params); + check(ctx != nullptr, "failed to create graph test context"); + + ggml_tensor * kv = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 512, 2, 3); + ggml_tensor * gate = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 512, 2, 3); + ggml_tensor * pooled = llama_dsv41_build_ratio_pool(ctx, kv, gate, 2); + check(pooled->ne[0] == 512 && pooled->ne[1] == 3, "ratio-2 graph output shape mismatch"); + + ggml_tensor * direct = llama_dsv41_build_ratio_pool( + ctx, ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 512, 1, 3), nullptr, 1); + check(direct->ne[0] == 512 && direct->ne[1] == 3, "ratio-1 graph output shape mismatch"); + + ggml_tensor * compressed_scores = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 32, 64, 2); + ggml_tensor * raw_scores = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 128, 64, 2); + ggml_tensor * probs = llama_dsv41_build_shared_softmax(ctx, raw_scores, compressed_scores); + check(probs->ne[0] == 160 && probs->ne[1] == 64 && probs->ne[2] == 2, "shared-softmax graph shape mismatch"); + + ggml_tensor * raw_order = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 2); + ggml_tensor * compressed_order = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 1); + const float raw_values[] = { 1.0f, 2.0f }; + const float compressed_value = 3.0f; + std::memcpy(raw_order->data, raw_values, sizeof(raw_values)); + std::memcpy(compressed_order->data, &compressed_value, sizeof(compressed_value)); + ggml_tensor * ordered_probs = llama_dsv41_build_shared_softmax(ctx, raw_order, compressed_order); + ggml_cgraph * gf = ggml_new_graph(ctx); + ggml_build_forward_expand(gf, ordered_probs); + check(ggml_graph_compute_with_ctx(ctx, gf, 1) == GGML_STATUS_SUCCESS, "shared-softmax graph execution failed"); + const float * ordered = static_cast(ordered_probs->data); + check(ordered[0] < ordered[1] && ordered[1] < ordered[2], "shared-softmax segment order mismatch"); + + ggml_tensor * residual = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 32, 4, 2); + ggml_tensor * pre = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 4, 2); + ggml_tensor * collapsed = llama_dsv41_build_output_collapse(ctx, residual, pre, 32, 4, 2); + check(collapsed->type == GGML_TYPE_BF16, "output collapse BF16 boundary is missing"); + check(collapsed->ne[0] == 32 && collapsed->ne[1] == 2, "output collapse graph shape mismatch"); + + ggml_tensor * output_norm = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 32); + ggml_tensor * output = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 32, 64); + ggml_tensor * logits = llama_dsv41_build_output(ctx, residual, pre, output_norm, output, 1.0e-20f, 4); + check(logits->ne[0] == 64 && logits->ne[1] == 2, "final output graph shape mismatch"); + + ggml_free(ctx); +} + +int main() { + test_hparams(); + test_source_maps(); + test_compression(); + test_state(); + test_raw_ring(); + test_candidates(); + test_output_collapse(); + test_graph_construction(); + return 0; +} diff --git a/tests/test-deepseek41-schema.cpp b/tests/test-deepseek41-schema.cpp index 4bbca5e85e00..4ae136d90c1a 100644 --- a/tests/test-deepseek41-schema.cpp +++ b/tests/test-deepseek41-schema.cpp @@ -76,6 +76,7 @@ int main() { } const LLM_TN tn(LLM_ARCH_DEEPSEEK41); + check(tn(LLM_TENSOR_ATTN_KV_A_NORM, "weight", 2).str() == "blk.2.attn_kv_a_norm.weight", "V4.1 KV A norm tensor name failed"); check(tn(LLM_TENSOR_ENGRAM_EMBD, "weight", 1).str() == "blk.1.engram_embd.weight", "Engram embedding tensor name failed"); check(tn(LLM_TENSOR_ENGRAM_Q_NORM, "weight", 1).str() == "blk.1.engram_q_norm.weight", "Engram query norm tensor name failed"); check(tn(LLM_TENSOR_ENGRAM_K_NORM, "weight", 1).str() == "blk.1.engram_k_norm.weight", "Engram key norm tensor name failed"); diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index a3ecf08ff1d7..b9f63331163e 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -562,7 +562,7 @@ static bool arch_supported(const llm_arch arch) { return false; } if (arch == LLM_ARCH_DEEPSEEK41) { - return false; // GGUF schema only; the runtime graph is added by a dependent PR. + return false; // Published-model construction is gated until Engram and expert streaming are available. } // FIXME: these hit scheduler/view-backed-output issues with WebGPU on CI. #ifdef GGML_USE_WEBGPU From 96fb9288203ca13867742fb6b9ff72d81f6578ae Mon Sep 17 00:00:00 2001 From: Jerome Coste Date: Sat, 12 Sep 2026 08:52:56 -0700 Subject: [PATCH 2/3] deepseek41 : fix published attention metadata Assisted-by: GPT-5.6 Sol Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- src/llama-dsv41.cpp | 14 ++++++-------- src/llama-dsv41.h | 6 +++--- src/models/deepseek41.cpp | 8 ++++---- tests/test-deepseek41-runtime.cpp | 19 +++++++++++++++---- 4 files changed, 28 insertions(+), 19 deletions(-) diff --git a/src/llama-dsv41.cpp b/src/llama-dsv41.cpp index 8b54af6a2c11..769f7c9096b8 100644 --- a/src/llama-dsv41.cpp +++ b/src/llama-dsv41.cpp @@ -12,7 +12,7 @@ static constexpr int32_t DSV41_KV_SOURCES[] = { 2, 8, 14, 20 }; static constexpr int32_t DSV41_INDEX_SOURCES[] = { 2, 8, 14, 20, 24, 28, 32, 36 }; static constexpr uint32_t DSV41_ENGRAM_LAYERS[] = { 1, 14 }; -static constexpr uint32_t DSV41_ENGRAM_ROWS[] = { 30000000, 5000000 }; +static constexpr uint32_t DSV41_ENGRAM_ROWS[] = { 384006168, 384016682 }; static void dsv41_require(bool condition, const char * message) { if (!condition) { @@ -50,12 +50,12 @@ void llama_dsv41_validate_config(const llama_dsv41_config & config) { dsv41_require(config.hc_eps == 1.0e-6f, "hc_eps must be 1e-6"); dsv41_require(config.swiglu_clamp == 10.0f, "swiglu_clamp_limit must be 10"); dsv41_require(config.routed_scale == 1.5f, "routed_scaling_factor must be 1.5"); - dsv41_require(config.rope_theta == 10000.0f, "rope_theta must be 10000"); - dsv41_require(config.compress_rope_theta == 160000.0f, "compress_rope_theta must be 160000"); + dsv41_require(config.rope_theta == 10000, "rope_theta must be 10000"); + dsv41_require(config.compress_rope_theta == 160000, "compress_rope_theta must be 160000"); dsv41_require(config.yarn_factor == 16.0f, "rope_scaling.factor must be 16"); dsv41_require(config.yarn_beta_fast == 32.0f, "rope_scaling.beta_fast must be 32"); dsv41_require(config.yarn_beta_slow == 1.0f, "rope_scaling.beta_slow must be 1"); - dsv41_require(config.yarn_original_context == 65536, "rope_scaling.original_max_position_embeddings must be 65536"); + dsv41_require(config.yarn_original_context == 65536.0f, "rope_scaling.original_max_position_embeddings must be 65536"); dsv41_require(config.expert_weights_norm, "norm_topk_prob must be true"); dsv41_require(config.hidden_act == "silu", "hidden_act must be silu"); dsv41_require(config.scoring_func == "sqrtsoftplus", "scoring_func must be sqrtsoftplus"); @@ -68,7 +68,7 @@ void llama_dsv41_validate_config(const llama_dsv41_config & config) { dsv41_require(config.kv_sources == std::vector(std::begin(DSV41_KV_SOURCES), std::end(DSV41_KV_SOURCES)), "kv_source_layers must be [2,8,14,20]"); dsv41_require(config.index_sources == std::vector(std::begin(DSV41_INDEX_SOURCES), std::end(DSV41_INDEX_SOURCES)), "index_source_layers must be [2,8,14,20,24,28,32,36]"); dsv41_require(config.engram_layers == std::vector(std::begin(DSV41_ENGRAM_LAYERS), std::end(DSV41_ENGRAM_LAYERS)), "engram.layer_ids must be [1,14]"); - dsv41_require(config.engram_rows == std::vector(std::begin(DSV41_ENGRAM_ROWS), std::end(DSV41_ENGRAM_ROWS)), "engram.rows must be [30000000,5000000]"); + dsv41_require(config.engram_rows == std::vector(std::begin(DSV41_ENGRAM_ROWS), std::end(DSV41_ENGRAM_ROWS)), "engram.rows must be [384006168,384016682]"); dsv41_require(config.engram_encoding == LLAMA_DSV41_ENGRAM_ENCODING, "engram.encoding must be e4m3_e8m0_32_row264"); dsv41_require(config.engram_compressed_vocab_size == LLAMA_DSV41_ENGRAM_COMPRESSED_VOCAB, "engram.compressed_vocab_size must be 99092"); dsv41_require(config.engram_pad_id == LLAMA_DSV41_ENGRAM_PAD_ID, "engram.pad_id must be 2"); @@ -388,9 +388,7 @@ std::vector llama_dsv41_select_candidate_blocks( } } - if (n_visible%block_size != 0) { - block_scores.back() = std::numeric_limits::infinity(); - } + block_scores.back() = std::numeric_limits::infinity(); std::vector blocks(n_blocks); std::iota(blocks.begin(), blocks.end(), 0); diff --git a/src/llama-dsv41.h b/src/llama-dsv41.h index ebbd0a577a67..47e6df22c76a 100644 --- a/src/llama-dsv41.h +++ b/src/llama-dsv41.h @@ -71,12 +71,12 @@ struct llama_dsv41_config { float hc_eps; float swiglu_clamp; float routed_scale; - float rope_theta; - float compress_rope_theta; + uint32_t rope_theta; + uint32_t compress_rope_theta; float yarn_factor; float yarn_beta_fast; float yarn_beta_slow; - uint32_t yarn_original_context; + float yarn_original_context; bool expert_weights_norm; std::string hidden_act; std::string scoring_func; diff --git a/src/models/deepseek41.cpp b/src/models/deepseek41.cpp index aba6126e0ca9..03ac0a6ab375 100644 --- a/src/models/deepseek41.cpp +++ b/src/models/deepseek41.cpp @@ -95,7 +95,7 @@ void llama_model_deepseek41::load_arch_hparams(llama_model_loader & ml) { hparams.dsv4_o_lora_rank = config.n_lora_o; hparams.dsv4_hc_mult = config.hc_count; hparams.dsv4_hc_sinkhorn_iters = config.hc_sinkhorn_iters; - hparams.dsv4_compress_rope_base = config.compress_rope_theta; + hparams.dsv4_compress_rope_base = (float) config.compress_rope_theta; hparams.dsv4_hc_eps = config.hc_eps; hparams.dsv41_candidate_source_layer = config.candidate_source_layer; hparams.dsv41_candidate_topk_blocks = config.candidate_topk_blocks; @@ -104,11 +104,11 @@ void llama_model_deepseek41::load_arch_hparams(llama_model_loader & ml) { hparams.expert_weights_scale = config.routed_scale; hparams.expert_weights_norm = config.expert_weights_norm; hparams.expert_gating_func = LLAMA_EXPERT_GATING_FUNC_TYPE_SQRT_SOFTPLUS; - hparams.rope_freq_base_train = config.rope_theta; - hparams.rope_freq_base_train_swa = config.rope_theta; + hparams.rope_freq_base_train = (float) config.rope_theta; + hparams.rope_freq_base_train_swa = (float) config.rope_theta; hparams.rope_freq_scale_train = 1.0f/config.yarn_factor; hparams.rope_freq_scale_train_swa = hparams.rope_freq_scale_train; - hparams.n_ctx_orig_yarn = config.yarn_original_context; + hparams.n_ctx_orig_yarn = (uint32_t) config.yarn_original_context; hparams.yarn_beta_fast = config.yarn_beta_fast; hparams.yarn_beta_slow = config.yarn_beta_slow; hparams.yarn_ext_factor = 1.0f; diff --git a/tests/test-deepseek41-runtime.cpp b/tests/test-deepseek41-runtime.cpp index 49289c409d35..3b996208d472 100644 --- a/tests/test-deepseek41-runtime.cpp +++ b/tests/test-deepseek41-runtime.cpp @@ -10,6 +10,7 @@ #include #include #include +#include #include static void check(bool condition, const std::string & message) { @@ -75,7 +76,7 @@ static llama_dsv41_config valid_config() { config.kv_sources = { 2, 8, 14, 20 }; config.index_sources = { 2, 8, 14, 20, 24, 28, 32, 36 }; config.engram_layers = { 1, 14 }; - config.engram_rows = { 30000000, 5000000 }; + config.engram_rows = { 384006168, 384016682 }; config.engram_encoding = LLAMA_DSV41_ENGRAM_ENCODING; config.engram_compressed_vocab_size = LLAMA_DSV41_ENGRAM_COMPRESSED_VOCAB; config.engram_pad_id = LLAMA_DSV41_ENGRAM_PAD_ID; @@ -86,7 +87,12 @@ static llama_dsv41_config valid_config() { } static void test_hparams() { + static_assert(std::is_same_v); + static_assert(std::is_same_v); + static_assert(std::is_same_v); + llama_dsv41_validate_config(valid_config()); + check(valid_config().engram_rows == std::vector({ 384006168, 384016682 }), "published Engram rows mismatch"); llama_dsv41_config config = valid_config(); config.compress_ratios.insert(config.compress_ratios.end(), 3, 0); @@ -199,14 +205,14 @@ static void test_raw_ring() { } static void test_candidates() { - for (uint32_t n_visible : { 1u, 7u, 8u, 9u, 127u, 16385u, 17017u }) { + for (uint32_t n_visible : { 1u, 7u, 8u, 9u, 127u, 16385u, 16392u, 17017u }) { std::vector scores(n_visible); for (uint32_t i = 0; i < n_visible; ++i) { scores[i] = -(float) i; } const auto blocks = llama_dsv41_select_candidate_blocks(scores, n_visible, 8, 2048); const int32_t final_block = (int32_t) ((n_visible - 1)/8); - check(std::find(blocks.begin(), blocks.end(), final_block) != blocks.end(), "final partial candidate block was dropped"); + check(std::find(blocks.begin(), blocks.end(), final_block) != blocks.end(), "final visible candidate block was dropped"); check(blocks.size() == std::min(2048, (n_visible + 7)/8), "candidate block count mismatch"); const auto rows = llama_dsv41_candidate_rows(blocks, n_visible, 8); check(std::find(rows.begin(), rows.end(), (int32_t) n_visible - 1) != rows.end(), "final visible row was filtered"); @@ -214,12 +220,17 @@ static void test_candidates() { } const auto tie = llama_dsv41_select_candidate_blocks(std::vector(24, 1.0f), 24, 8, 2); - check(tie == std::vector({ 0, 1 }), "candidate tie-break mismatch"); + check(tie == std::vector({ 2, 0 }), "candidate tie-break or final block retention mismatch"); std::vector partial_scores(9, -100.0f); partial_scores[0] = 100.0f; const auto partial = llama_dsv41_select_candidate_blocks(partial_scores, 9, 8, 1); check(partial == std::vector({ 1 }), "final partial candidate block was not forced"); + + std::vector full_scores(16392, -100.0f); + full_scores[0] = 100.0f; + const auto full = llama_dsv41_select_candidate_blocks(full_scores, 16392, 8, 2048); + check(std::find(full.begin(), full.end(), 2048) != full.end(), "final full candidate block was not forced"); } static void test_output_collapse() { From c51c971392ca4ee0fe502c532bbe415b14f8a71a Mon Sep 17 00:00:00 2001 From: Jerome Coste Date: Mon, 14 Sep 2026 00:40:54 -0700 Subject: [PATCH 3/3] deepseek41 : fix output and candidate execution Assisted-by: GPT-5.6 Sol Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- src/llama-dsv41.cpp | 15 +++++++------ tests/test-deepseek41-runtime.cpp | 35 +++++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+), 6 deletions(-) diff --git a/src/llama-dsv41.cpp b/src/llama-dsv41.cpp index 769f7c9096b8..84dbf8188ba1 100644 --- a/src/llama-dsv41.cpp +++ b/src/llama-dsv41.cpp @@ -388,9 +388,8 @@ std::vector llama_dsv41_select_candidate_blocks( } } - block_scores.back() = std::numeric_limits::infinity(); - - std::vector blocks(n_blocks); + const int32_t final_block = (int32_t) n_blocks - 1; + std::vector blocks(n_blocks - 1); std::iota(blocks.begin(), blocks.end(), 0); std::stable_sort(blocks.begin(), blocks.end(), [&](int32_t a, int32_t b) { if (block_scores[a] != block_scores[b]) { @@ -399,8 +398,12 @@ std::vector llama_dsv41_select_candidate_blocks( return a < b; }); - blocks.resize(std::min(top_k_blocks, n_blocks)); - return blocks; + const uint32_t n_selected = std::min(top_k_blocks, n_blocks); + std::vector selected; + selected.reserve(n_selected); + selected.push_back(final_block); + selected.insert(selected.end(), blocks.begin(), blocks.begin() + n_selected - 1); + return selected; } std::vector llama_dsv41_candidate_rows( @@ -537,7 +540,7 @@ ggml_tensor * llama_dsv41_build_output( ggml_tensor * collapsed = llama_dsv41_build_output_collapse( ctx, residual, pre, residual->ne[0], hc_mult, residual->ne[2]); - ggml_tensor * normalized = ggml_rms_norm(ctx, collapsed, rms_eps); + ggml_tensor * normalized = ggml_rms_norm(ctx, ggml_cast(ctx, collapsed, GGML_TYPE_F32), rms_eps); normalized = ggml_mul(ctx, normalized, output_norm); return ggml_mul_mat(ctx, output, normalized); } diff --git a/tests/test-deepseek41-runtime.cpp b/tests/test-deepseek41-runtime.cpp index 3b996208d472..6c9fc0167ea4 100644 --- a/tests/test-deepseek41-runtime.cpp +++ b/tests/test-deepseek41-runtime.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -231,6 +232,22 @@ static void test_candidates() { full_scores[0] = 100.0f; const auto full = llama_dsv41_select_candidate_blocks(full_scores, 16392, 8, 2048); check(std::find(full.begin(), full.end(), 2048) != full.end(), "final full candidate block was not forced"); + + const auto inf_tie = llama_dsv41_select_candidate_blocks( + std::vector(24, std::numeric_limits::infinity()), 24, 8, 2); + check(std::find(inf_tie.begin(), inf_tie.end(), 2) != inf_tie.end(), "final candidate block lost an infinity tie"); + + const auto inf_boundary = llama_dsv41_select_candidate_blocks( + std::vector(16392, std::numeric_limits::infinity()), 16392, 8, 2048); + check(std::find(inf_boundary.begin(), inf_boundary.end(), 2048) != inf_boundary.end(), + "final full candidate block lost an infinity tie"); + std::vector unique_blocks = inf_boundary; + std::sort(unique_blocks.begin(), unique_blocks.end()); + check(std::adjacent_find(unique_blocks.begin(), unique_blocks.end()) == unique_blocks.end(), + "candidate selection contains duplicate blocks"); + check(std::all_of(unique_blocks.begin(), unique_blocks.end(), [](int32_t block) { + return block >= 0 && block <= 2048; + }), "candidate selection contains an out-of-range block"); } static void test_output_collapse() { @@ -293,6 +310,24 @@ static void test_graph_construction() { ggml_tensor * logits = llama_dsv41_build_output(ctx, residual, pre, output_norm, output, 1.0e-20f, 4); check(logits->ne[0] == 64 && logits->ne[1] == 2, "final output graph shape mismatch"); + ggml_tensor * exec_residual = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 32, 4, 1); + ggml_tensor * exec_pre = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 4, 1); + ggml_tensor * exec_norm = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 32); + ggml_tensor * exec_output = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 32, 2); + std::fill_n(static_cast(exec_residual->data), ggml_nelements(exec_residual), 1.0f); + std::fill_n(static_cast(exec_pre->data), ggml_nelements(exec_pre), 0.25f); + std::fill_n(static_cast(exec_norm->data), ggml_nelements(exec_norm), 1.0f); + std::fill_n(static_cast(exec_output->data), ggml_nelements(exec_output), 1.0f); + ggml_tensor * exec_logits = llama_dsv41_build_output( + ctx, exec_residual, exec_pre, exec_norm, exec_output, 1.0e-20f, 4); + ggml_cgraph * exec_gf = ggml_new_graph(ctx); + ggml_build_forward_expand(exec_gf, exec_logits); + check(ggml_graph_compute_with_ctx(ctx, exec_gf, 1) == GGML_STATUS_SUCCESS, "final output graph execution failed"); + const float * exec_values = static_cast(exec_logits->data); + check(std::isfinite(exec_values[0]) && std::isfinite(exec_values[1]), "final output graph produced non-finite logits"); + check(std::abs(exec_values[0] - 32.0f) < 1.0e-4f && std::abs(exec_values[1] - 32.0f) < 1.0e-4f, + "final output graph numeric mismatch"); + ggml_free(ctx); }