diff --git a/include/llama.h b/include/llama.h index 177fc10a91..a04177f9f7 100644 --- a/include/llama.h +++ b/include/llama.h @@ -733,7 +733,7 @@ extern "C" { // Removes all tokens that belong to the specified sequence and have positions in [p0, p1) // Returns false if a partial sequence cannot be removed. Removing a whole sequence never fails - // seq_id < 0 : match any sequence + // seq_id < 0 : match any sequence [TAG_LLAMA_SEQ_ID_NEG] // p0 < 0 : [0, p1] // p1 < 0 : [p0, inf) LLAMA_API bool llama_memory_seq_rm( diff --git a/src/llama-context.cpp b/src/llama-context.cpp index 52f8d53672..0402044da6 100644 --- a/src/llama-context.cpp +++ b/src/llama-context.cpp @@ -3218,8 +3218,6 @@ size_t llama_context::state_read_data(llama_io_read_i & io) { } size_t llama_context::state_seq_write_data(llama_io_write_i & io, llama_seq_id seq_id, llama_state_seq_flags flags) { - GGML_UNUSED(seq_id); - if (memory) { memory->state_write(io, seq_id, flags); } @@ -3228,8 +3226,6 @@ size_t llama_context::state_seq_write_data(llama_io_write_i & io, llama_seq_id s } size_t llama_context::state_seq_read_data(llama_io_read_i & io, llama_seq_id seq_id, llama_state_seq_flags flags) { - GGML_UNUSED(seq_id); - if (memory) { memory->state_read(io, seq_id, flags); } diff --git a/src/llama-kv-cache-dsv4.cpp b/src/llama-kv-cache-dsv4.cpp index 5caa05e8b0..58f78e4384 100644 --- a/src/llama-kv-cache-dsv4.cpp +++ b/src/llama-kv-cache-dsv4.cpp @@ -599,6 +599,33 @@ static llama_kv_cache_dsv4_context::comp_plan dsv4_build_comp_plan( } } + if (ratio == DSV4_HCA_RATIO && !plan.state_pos.empty() && plan.state_write_idxs.empty()) { + assert(kv_size > 0); + // the last slot must not be live, or the dummy write would corrupt it; + // a full stream implies a completed block, which implies real writes + assert(plan.n_kv < (int64_t) kv_size); + + // Keep the compress/write ops in the graph when no HCA block completes + // in this ubatch. The dummy block writes to the last cache slot and is + // masked out. + uint32_t i = 0; + while (i < ubatch.n_tokens && ubatch.pos[i] < 0) { + ++i; + } + assert(i < ubatch.n_tokens); + + const llama_seq_id seq_id = ubatch.seq_id[i][0]; + const int64_t cache_off = dsv4_stream_offset(n_stream, seq_id, kv_size); + const int32_t source_idx = state_source_idx(seq_id, ubatch.pos[i]); + + plan.state_write_idxs.push_back(cache_off + kv_size - 1); + plan.state_write_pos .push_back(0); + + for (uint32_t j = 0; j < ratio; ++j) { + plan.state_read_idxs.push_back(source_idx); + } + } + if (overlap) { // [ all blocks' prev-window indices | all blocks' cur-window indices ] plan.state_read_idxs.reserve(overlap_prev_reads.size() + overlap_cur_reads.size()); @@ -608,7 +635,10 @@ static llama_kv_cache_dsv4_context::comp_plan dsv4_build_comp_plan( overlap_cur_reads.begin(), overlap_cur_reads.end()); } - plan.n_kv = GGML_PAD(plan.n_kv, 256u); + // Keep the mask (and with it the compressed-attention branch) present even + // before the first block is visible, so the graph topology never changes. + // Padded slots are masked out; comp cache buffers are zero-initialized. + plan.n_kv = std::max(GGML_PAD(plan.n_kv, 256u), 256); std::sort(persist_rows.begin(), persist_rows.end(), [](const persist_row & a, const persist_row & b) { @@ -620,16 +650,26 @@ static llama_kv_cache_dsv4_context::comp_plan dsv4_build_comp_plan( plan.state_persist_dst_idxs.push_back(row.dst); } - if (n_rs_seq > 0) { - for (uint32_t s = 0; s < ubatch.n_seqs_unq; ++s) { - const llama_seq_id seq_id = ubatch.seq_id_unq[s]; - if (seq_id < 0 || (uint32_t) seq_id >= n_stream) { - continue; + // Emit restore/snapshot entries for all layout streams so that the + // graph tensor sizes do not depend on the ubatch's sequence count. + // Streams not present in the ubatch get no-op entries. + for (uint32_t stream = 0; stream < n_stream; ++stream) { + llama_seq_id seq_id = -1; + if (n_stream == 1) { + // a unified stream serves any single sequence + seq_id = ubatch.n_seqs_unq > 0 ? ubatch.seq_id_unq[0] : -1; + } else { + for (uint32_t s = 0; s < ubatch.n_seqs_unq; ++s) { + if (ubatch.seq_id_unq[s] == (llama_seq_id) stream) { + seq_id = ubatch.seq_id_unq[s]; + break; + } + } } - const int64_t stream_off = dsv4_stream_offset(n_stream, seq_id, state_size); - const uint32_t rollback = (uint32_t) seq_id < rs_idx.size() ? rs_idx[seq_id] : 0; + const int64_t stream_off = (int64_t) stream*state_size; + const uint32_t rollback = seq_id >= 0 && (uint32_t) seq_id < rs_idx.size() ? rs_idx[seq_id] : 0; // Keep the restore graph fixed-width when no rollback is pending. const int64_t src_plane = rollback > 0 && rollback <= n_rs_seq ? (int64_t) rollback*state_rows : 0; for (uint32_t r = 0; r < state_size; ++r) { @@ -639,35 +679,33 @@ static llama_kv_cache_dsv4_context::comp_plan dsv4_build_comp_plan( std::vector token_idxs; token_idxs.reserve(ubatch.n_tokens); - for (uint32_t i = 0; i < ubatch.n_tokens; ++i) { - if (dsv4_token_has_seq(ubatch, i, seq_id)) { - token_idxs.push_back(i); + if (seq_id >= 0) { + for (uint32_t i = 0; i < ubatch.n_tokens; ++i) { + if (dsv4_token_has_seq(ubatch, i, seq_id)) { + token_idxs.push_back(i); + } } } - if (token_idxs.empty()) { - continue; - } const uint32_t n_seq_tokens = (uint32_t) token_idxs.size(); const int64_t scratch_off = (int64_t) state_rows*(1 + n_rs_seq); for (uint32_t d = 1; d <= n_rs_seq; ++d) { const int64_t dst_plane = (int64_t) d*state_rows; + const uint32_t prefix = d <= n_seq_tokens ? n_seq_tokens - d : 0; for (uint32_t r = 0; r < state_size; ++r) { - int32_t src; - if (d <= n_seq_tokens) { - const uint32_t prefix = n_seq_tokens - d; - src = (int32_t) (stream_off + r); + int32_t src = (int32_t) (stream_off + r); - for (uint32_t j = 0; j < prefix; ++j) { - const uint32_t i_tok = token_idxs[j]; - if (ubatch.pos[i_tok] >= 0 && (uint32_t) (ubatch.pos[i_tok]%state_size) == r) { - src = (int32_t) (scratch_off + i_tok); - } + for (uint32_t j = 0; j < prefix; ++j) { + const uint32_t i_tok = token_idxs[j]; + if (ubatch.pos[i_tok] >= 0 && (uint32_t) (ubatch.pos[i_tok]%state_size) == r) { + src = (int32_t) (scratch_off + i_tok); } - } else { - const int64_t src_plane = (int64_t) (d - n_seq_tokens)*state_rows; - src = (int32_t) (src_plane + stream_off + r); + } + + if (n_seq_tokens == 0) { + // no-op: copy the snapshot plane onto itself + src = (int32_t) (dst_plane + stream_off + r); } plan.state_snapshot_src_idxs.push_back(src); @@ -683,10 +721,16 @@ static llama_kv_cache_dsv4_context::comp_plan dsv4_build_comp_plan( }(); if (debug) { - LLAMA_LOG_INFO("%s: ratio=%u, n_tokens=%u, state_persist_dst=%s, state_write_pos=%s\n", - __func__, ratio, ubatch.n_tokens, + LLAMA_LOG_DEBUG("%s: ratio=%u, n_tokens=%u, n_seqs_unq=%u, state_persist_dst=%s, state_write_pos=%s\n", + __func__, ratio, ubatch.n_tokens, ubatch.n_seqs_unq, dsv4_plan_positions(plan.state_persist_dst_idxs).c_str(), dsv4_plan_positions(plan.state_write_pos).c_str()); + for (uint32_t s = 0; s < ubatch.n_seqs_unq; ++s) { + const llama_seq_id seq_id = ubatch.seq_id_unq[s]; + const uint32_t rollback = seq_id >= 0 && (uint32_t) seq_id < rs_idx.size() ? rs_idx[seq_id] : 0; + LLAMA_LOG_DEBUG("%s: seq %d pos [%d, %d] rollback=%u\n", __func__, seq_id, + ubatch.pos[0], ubatch.pos[ubatch.n_tokens - 1], rollback); + } } return plan; @@ -704,8 +748,17 @@ static std::vector dsv4_build_comp_plans std::vector plans; plans.reserve(ubatches.size()); + // the first ubatch touching a seq consumes its rollback restore + std::vector rs(rs_idx); for (const llama_ubatch & ubatch : ubatches) { - plans.push_back(dsv4_build_comp_plan(ubatch, ratio, overlap, state_size, kv_size, n_stream, n_rs_seq, rs_idx)); + plans.push_back(dsv4_build_comp_plan(ubatch, ratio, overlap, state_size, kv_size, n_stream, n_rs_seq, rs)); + + for (uint32_t s = 0; s < ubatch.n_seqs_unq; ++s) { + const llama_seq_id seq_id = ubatch.seq_id_unq[s]; + if (seq_id >= 0 && (size_t) seq_id < rs.size()) { + rs[seq_id] = 0; + } + } } return plans; @@ -803,16 +856,15 @@ static llama_kv_cache_dsv4_context::comp_plan dsv4_build_reserve_comp_plan( return plan; } - const uint32_t n_seqs = std::max(1, ubatch.n_seqs); - const uint32_t n_seq_tokens = std::max(1, ubatch.n_seq_tokens); - const uint64_t n_blocks_u64 = (uint64_t) n_seqs*((n_seq_tokens + ratio - 1)/ratio); - const size_t n_blocks = (size_t) std::max(1, n_blocks_u64); - GGML_ASSERT((uint64_t) n_blocks == std::max(1, n_blocks_u64)); + // worst case over every seq split: sum of per-seq ceil(tokens/ratio) is at + // most floor(n_tokens/ratio) + n_seqs + const uint32_t n_seqs = std::max(1, ubatch.n_seqs); + const size_t n_blocks = (size_t) ubatch.n_tokens/ratio + n_seqs; const uint64_t state_rows = (uint64_t) state_size*n_stream; const size_t n_persist = (size_t) std::min(ubatch.n_tokens, state_rows); - const size_t n_restore = n_rs_seq > 0 ? (size_t) state_size*std::max(1, ubatch.n_seqs_unq) : 0; - const size_t n_snapshot = (size_t) n_rs_seq*state_size*std::max(1, ubatch.n_seqs_unq); + const size_t n_restore = n_rs_seq > 0 ? (size_t) state_size*n_stream : 0; + const size_t n_snapshot = (size_t) n_rs_seq*state_size*n_stream; plan.state_pos .resize(ubatch.n_tokens); plan.state_persist_src_idxs.resize(n_persist); @@ -1356,7 +1408,9 @@ llama_memory_context_ptr llama_kv_cache_dsv4::init_batch( if (has_coupled) { ubatch = balloc.split_seq(n_ubatch); } else { - ubatch = balloc.split_equal(n_ubatch, raw_per_seq || comp_per_seq, 0); + // [TAG_RECURRENT_ROLLBACK_SPLITS] + // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch + ubatch = balloc.split_equal(n_ubatch, raw_per_seq || comp_per_seq, n_rs_seq > 0 ? n_rs_seq + 1 : 0); } if (ubatch.n_tokens == 0) { @@ -1433,6 +1487,11 @@ bool llama_kv_cache_dsv4::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos p1 return false; } + // pending rollback is single-use: stacked partial removals don't compose + if (rs_idx[seq_id] != 0) { + return false; + } + const bool res = kv_raw->seq_rm(seq_id, p0, p1); if (res) { rs_idx[seq_id] = (uint32_t) rollback; @@ -1594,9 +1653,7 @@ void llama_kv_cache_dsv4::state_read(llama_io_read_i & io, llama_seq_id seq_id, kv_raw->state_read(io, seq_id, flags); if (!partial_only) { - kv_csa->clear(true); - kv_hca->clear(true); - kv_lid->clear(true); + clear_compressed(seq_id, true); dsv4_state_read_k_cache(io, kv_csa.get(), seq_id, flags); dsv4_state_read_k_cache(io, kv_hca.get(), seq_id, flags); diff --git a/src/llama-kv-cache.cpp b/src/llama-kv-cache.cpp index 2e2bd7dc6d..ec0f5a7531 100644 --- a/src/llama-kv-cache.cpp +++ b/src/llama-kv-cache.cpp @@ -383,6 +383,7 @@ bool llama_kv_cache::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos p1) { return true; } + // TODO: fix incosistent handling of `seq_id < 0` and `seq_id == -1` in the codebase [TAG_LLAMA_SEQ_ID_NEG] GGML_ASSERT(seq_id == -1 || (seq_id >= 0 && (size_t) seq_id < seq_to_stream.size())); if (p0 < 0) { @@ -2043,6 +2044,7 @@ void llama_kv_cache::state_read(llama_io_read_i & io, llama_seq_id seq_id, llama GGML_UNUSED(flags); + // TODO: fix incosistent handling of `seq_id < 0` and `seq_id == -1` in the codebase [TAG_LLAMA_SEQ_ID_NEG] GGML_ASSERT(seq_id == -1 || (seq_id >= 0 && (size_t) seq_id < seq_to_stream.size())); uint32_t n_stream_cur; diff --git a/src/llama-memory-recurrent.cpp b/src/llama-memory-recurrent.cpp index ef82eb976c..e2990972ef 100644 --- a/src/llama-memory-recurrent.cpp +++ b/src/llama-memory-recurrent.cpp @@ -158,13 +158,14 @@ bool llama_memory_recurrent::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos p1 = std::numeric_limits::max(); } + if ((uint32_t) seq_id >= this->n_seq_max) { + LLAMA_LOG_ERROR("%s: invalid seq_id (%d) - larger than n_seq_max (%d)\n", __func__, seq_id, this->n_seq_max); + return false; + } + const bool rm_all = p0 == 0 && p1 == std::numeric_limits::max(); if (rm_all) { - if (seq_id >= 0) { - set_rs_idx(seq_id, 0); - } else { - std::fill(rs_idx.begin(), rs_idx.end(), 0); - } + set_rs_idx(seq_id, 0); } // models like Mamba or RWKV can't have a state partially erased at the end @@ -181,7 +182,9 @@ bool llama_memory_recurrent::seq_rm(llama_seq_id seq_id, llama_pos p0, llama_pos // partial rollback via per-token snapshot index (bounded by n_rs_seq) if (0 < p0 && p0 <= cell.pos && p1 > cell.pos) { const llama_pos rollback = cell.pos - (p0 - 1); - if (rollback >= 1 && rollback <= (llama_pos) n_rs_seq) { + // pending rollback is single-use + const bool pending = rs_idx[seq_id] != 0; + if (!pending && rollback >= 1 && rollback <= (llama_pos) n_rs_seq) { set_rs_idx(seq_id, (uint32_t) rollback); cell.pos = p0 - 1; return true; @@ -390,10 +393,17 @@ llama_pos llama_memory_recurrent::seq_pos_max(llama_seq_id seq_id) const { } void llama_memory_recurrent::set_rs_idx(llama_seq_id seq_id, uint32_t idx) { - if (seq_id < 0 || (size_t) seq_id >= rs_idx.size()) { + if (seq_id < 0) { + std::fill(rs_idx.begin(), rs_idx.end(), 0); return; } - rs_idx[seq_id] = (idx > n_rs_seq) ? n_rs_seq : idx; + + assert(n_seq_max == rs_idx.size()); + + GGML_ASSERT((uint32_t) seq_id < n_seq_max); + GGML_ASSERT(idx <= n_rs_seq); + + rs_idx[seq_id] = idx; } std::map llama_memory_recurrent::memory_breakdown() const { @@ -742,6 +752,7 @@ void llama_memory_recurrent::state_write(llama_io_write_i & io, llama_seq_id seq uint32_t cell_range_begin = size; for (uint32_t i = 0; i < size; ++i) { const auto & cell = cells[i]; + // TODO: fix incosistent handling of `seq_id < 0` and `seq_id == -1` in the codebase [TAG_LLAMA_SEQ_ID_NEG] if ((seq_id == -1 && !cell.is_empty()) || cell.has_seq_id(seq_id)) { ++cell_count; uint32_t rs_idx_cur = 0; @@ -827,6 +838,7 @@ void llama_memory_recurrent::state_read(llama_io_read_i & io, llama_seq_id seq_i } if (!res) { + // TODO: fix incosistent handling of `seq_id < 0` and `seq_id == -1` in the codebase [TAG_LLAMA_SEQ_ID_NEG] if (seq_id == -1) { clear(true); } else { @@ -836,11 +848,7 @@ void llama_memory_recurrent::state_read(llama_io_read_i & io, llama_seq_id seq_i } if (n_rs_seq != 0) { - if (seq_id == -1) { - std::fill(rs_idx.begin(), rs_idx.end(), 0); - } else { - set_rs_idx(seq_id, 0); - } + set_rs_idx(seq_id, 0); } } diff --git a/src/llama-model-saver.cpp b/src/llama-model-saver.cpp index 0d39e6de89..2eb5b7aafd 100644 --- a/src/llama-model-saver.cpp +++ b/src/llama-model-saver.cpp @@ -293,6 +293,14 @@ void llama_model_saver::add_kv_from_model() { add_kv(LLM_KV_ATTENTION_INDEXER_LOCAL_BLOCKS, hparams.indexer_local_blocks); add_kv(LLM_KV_ATTENTION_INDEXER_TYPES, hparams.is_indexer_full_impl, true); add_kv(LLM_KV_ATTENTION_RECURRENT_LAYERS, hparams.is_recr_impl, true); + add_kv(LLM_KV_ATTENTION_OUTPUT_GROUP_COUNT, hparams.dsv4_o_group_count); + add_kv(LLM_KV_ATTENTION_OUTPUT_LORA_RANK, hparams.dsv4_o_lora_rank); + add_kv(LLM_KV_ATTENTION_COMPRESS_ROPE_FREQ_BASE, hparams.dsv4_compress_rope_base); + add_kv(LLM_KV_ATTENTION_COMPRESS_RATIOS, hparams.dsv4_compress_ratios, true); + add_kv(LLM_KV_HYPER_CONNECTION_COUNT, hparams.dsv4_hc_mult); + add_kv(LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS, hparams.dsv4_hc_sinkhorn_iters); + add_kv(LLM_KV_HYPER_CONNECTION_EPSILON, hparams.dsv4_hc_eps); + add_kv(LLM_KV_HASH_LAYER_COUNT, hparams.dsv4_hash_layer_count); const float rope_scaling_factor = hparams.rope_freq_scale_train == 1.0f ? 0.0f : 1.0f/hparams.rope_freq_scale_train; @@ -422,6 +430,9 @@ void llama_model_saver::add_tensors_from_model() { add_tensor(model->cls_out); add_tensor(model->cls_out_b); add_tensor(model->cls_norm); + add_tensor(model->hc_head_fn); + add_tensor(model->hc_head_base); + add_tensor(model->hc_head_scale); for (const struct llama_layer & layer : model->layers) { for (size_t i = 0; i < sizeof(layer)/sizeof(struct ggml_tensor *); ++i) { diff --git a/src/models/deepseek4.cpp b/src/models/deepseek4.cpp index 1c278e435e..fc816e2aeb 100644 --- a/src/models/deepseek4.cpp +++ b/src/models/deepseek4.cpp @@ -1,3 +1,4 @@ +#include "llama-hparams.h" #include "models.h" #include "llama-kv-cache-dsv4.h" @@ -58,6 +59,7 @@ void llama_model_deepseek4::load_arch_hparams(llama_model_loader & ml) { if (n_compress_ratios < hparams.n_layer_all) { throw std::runtime_error("DeepSeek-V4 compress_ratios is shorter than block_count"); } + GGML_ASSERT(n_compress_ratios <= LLAMA_MAX_LAYERS); ml.get_arr(LLM_KV_ATTENTION_COMPRESS_RATIOS, hparams.dsv4_compress_ratios); ml.get_key(LLM_KV_EXPERT_GATING_FUNC, hparams.expert_gating_func); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index e517d2c635..cb6ae29707 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -228,6 +228,15 @@ if (NOT WIN32 OR NOT BUILD_SHARED_LIBS) set_tests_properties(test-recurrent-state-rollback-nemotron-h PROPERTIES FIXTURES_REQUIRED generate-models ) + llama_test( + test-recurrent-state-rollback + NAME test-recurrent-state-rollback-dsv4 + LABEL main + ARGS -m "${MODEL_DIR}/deepseek4-moe.gguf" + ) + set_tests_properties(test-recurrent-state-rollback-dsv4 PROPERTIES + FIXTURES_REQUIRED generate-models + ) endif() llama_build_and_test(test-chat-peg-parser.cpp peg-parser/simple-tokenize.cpp) diff --git a/tests/test-llama-archs.cpp b/tests/test-llama-archs.cpp index 18676f2be6..dff8c4668b 100644 --- a/tests/test-llama-archs.cpp +++ b/tests/test-llama-archs.cpp @@ -101,6 +101,11 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) { n_head = 1; n_ff = 96; n_layer = 22; // hparams.n_layer_kv_from_start = 20 is hardcoded + } else if (arch == LLM_ARCH_DEEPSEEK4) { + n_embd = 128; + n_head = 1; + n_ff = 192; + n_layer = 3; // uncompressed + csa + hca, one layer of each ratio kind } else if (arch == LLM_ARCH_STEP35 || arch == LLM_ARCH_LAGUNA) { n_embd = 160; // exercise per-head tensor split granularity with head size 80 } else if (arch == LLM_ARCH_QWEN3 || arch == LLM_ARCH_MUSE_GLIMMER || arch == LLM_ARCH_AFMOE) { @@ -203,6 +208,10 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) { } ms.add_kv(LLM_KV_ATTENTION_INDEXER_TYPES, indexer_types); } + } else if (arch == LLM_ARCH_DEEPSEEK4) { + ms.add_kv(LLM_KV_ATTENTION_KEY_LENGTH, uint32_t(128)); + ms.add_kv(LLM_KV_ATTENTION_VALUE_LENGTH, uint32_t(128)); + ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, uint32_t(64)); } else if (arch == LLM_ARCH_MINIMAX_M3) { // partial rotary: n_rot must not exceed the indexer key length (64) ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, uint32_t(64)); @@ -239,6 +248,20 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) { // MSA requires one indexer head per GQA (KV) head, unlike the DSA archs where the // indexer head count is independent of the main attention head count. + if (arch == LLM_ARCH_DEEPSEEK4) { + ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE, 2.5f); + ms.add_kv(LLM_KV_EXPERT_WEIGHTS_NORM, true); + ms.add_kv(LLM_KV_SWIGLU_CLAMP_EXP, 7.0f); + ms.add_kv(LLM_KV_ATTENTION_OUTPUT_GROUP_COUNT, uint32_t(1)); + ms.add_kv(LLM_KV_ATTENTION_OUTPUT_LORA_RANK, uint32_t(64)); + ms.add_kv(LLM_KV_ATTENTION_COMPRESS_ROPE_FREQ_BASE, 10000.0f); + ms.add_kv(LLM_KV_HYPER_CONNECTION_COUNT, uint32_t(4)); + ms.add_kv(LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS, uint32_t(4)); + ms.add_kv(LLM_KV_HYPER_CONNECTION_EPSILON, 1e-6f); + ms.add_kv(LLM_KV_HASH_LAYER_COUNT, uint32_t(0)); + ms.add_kv(LLM_KV_ATTENTION_COMPRESS_RATIOS, std::vector({0, 4, 128})); + } + ms.add_kv(LLM_KV_ATTENTION_INDEXER_HEAD_COUNT, arch == LLM_ARCH_MINIMAX_M3 ? n_head : uint32_t(1)); ms.add_kv(LLM_KV_ATTENTION_INDEXER_KEY_LENGTH, uint32_t(64)); ms.add_kv(LLM_KV_ATTENTION_INDEXER_TOP_K, uint32_t(8)); @@ -257,7 +280,7 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) { ms.add_kv(LLM_KV_EXPERT_COUNT, uint32_t(2)); ms.add_kv(LLM_KV_EXPERT_USED_COUNT, uint32_t(1)); ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, uint32_t(1)); - ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, uint32_t(2)); // sigmoid + ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, arch == LLM_ARCH_DEEPSEEK4 ? uint32_t(4) : uint32_t(2)); // sqrtsoftplus : sigmoid ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f); ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1)); } @@ -395,6 +418,7 @@ static bool moe_mandatory(const llm_arch arch) { case LLM_ARCH_DEEPSEEK2: case LLM_ARCH_DEEPSEEK32: case LLM_ARCH_DOTS3NOTE: + case LLM_ARCH_DEEPSEEK4: case LLM_ARCH_GLM4_MOE: case LLM_ARCH_GLM_DSA: case LLM_ARCH_EXAONE_MOE: @@ -480,9 +504,6 @@ static bool arch_supported(const llm_arch arch) { if (arch == LLM_ARCH_DEEPSEEK2OCR) { return false; } - if (arch == LLM_ARCH_DEEPSEEK4) { - return false; - } // FIXME: these hit scheduler/view-backed-output issues with WebGPU on CI. #ifdef GGML_USE_WEBGPU diff --git a/tests/test-recurrent-state-rollback.cpp b/tests/test-recurrent-state-rollback.cpp index 5d1f0140b6..c6f599e584 100644 --- a/tests/test-recurrent-state-rollback.cpp +++ b/tests/test-recurrent-state-rollback.cpp @@ -35,6 +35,178 @@ static bool decode_one(llama_context * ctx, llama_token tok, llama_pos pos) { return ok; } +// Roll back multiple sequences, then replay them in a single batch whose +// per-seq token count exceeds n_ubatch: each seq's replay spans several +// ubatches while its rollback restore is still pending. Compared against a +// reference context that never advanced past the rollback point and decodes +// the identical replay batch. +static bool test_multi_seq_split_replay(const common_params & params, llama_model * model, const int n_vocab) { + constexpr uint32_t n_seqs = 2; + constexpr uint32_t n_ubatch = 16; + constexpr uint32_t n_prompt = 19; + constexpr uint32_t n_rollback = 3; + constexpr uint32_t n_replay = 40; // > n_ubatch so each seq spans multiple ubatches + constexpr llama_pos p0 = n_prompt - n_rollback; + + const auto make_ctx_multi = [&]() { + auto cparams = common_context_params_to_llama(params); + cparams.n_seq_max = n_seqs; + cparams.n_rs_seq = 8; + cparams.n_ctx = 256; + cparams.n_batch = 256; + cparams.n_ubatch = n_ubatch; + cparams.kv_unified = false; + return llama_init_from_model(model, cparams); + }; + + llama_context * ctx_roll = make_ctx_multi(); + llama_context * ctx_ref = make_ctx_multi(); + if (ctx_roll == nullptr || ctx_ref == nullptr) { + fprintf(stderr, "%s : failed to init multi-seq contexts\n", __func__); + return false; + } + + const auto cleanup = [&]() { + llama_free(ctx_roll); + llama_free(ctx_ref); + }; + + if (llama_n_rs_seq(ctx_roll) < n_rollback) { + fprintf(stderr, "%s : skipping because n_rs_seq is too small\n", __func__); + cleanup(); + return true; + } + + const auto tok = [&](uint32_t seq, llama_pos pos) { + return (llama_token) ((7*(uint32_t) pos + 31*seq + 1) % (uint32_t) n_vocab); + }; + + bool ok = true; + + // both contexts decode the identical [0, p0) prefill; only ctx_roll decodes + // the tail, which is then rolled back so its restore is pending at replay + for (uint32_t s = 0; s < n_seqs && ok; ++s) { + llama_batch batch = llama_batch_init(n_prompt, 0, 1); + for (llama_pos pos = 0; pos < (llama_pos) p0; ++pos) { + common_batch_add(batch, tok(s, pos), pos, { (llama_seq_id) s }, false); + } + ok = ok && llama_decode(ctx_roll, batch) == 0; + ok = ok && llama_decode(ctx_ref, batch) == 0; + + common_batch_clear(batch); + for (llama_pos pos = p0; pos < (llama_pos) n_prompt; ++pos) { + common_batch_add(batch, tok(s, pos), pos, { (llama_seq_id) s }, false); + } + ok = ok && llama_decode(ctx_roll, batch) == 0; + llama_batch_free(batch); + + ok = ok && llama_memory_seq_rm(llama_get_memory(ctx_roll), (llama_seq_id) s, p0, -1); + + // a second partial removal while one is pending must be refused + ok = ok && !llama_memory_seq_rm(llama_get_memory(ctx_roll), (llama_seq_id) s, p0 - 1, -1); + } + if (!ok) { + fprintf(stderr, "%s : multi-seq prefill/rollback failed\n", __func__); + cleanup(); + return false; + } + + llama_batch batch = llama_batch_init(n_seqs*n_replay, 0, 1); + for (uint32_t s = 0; s < n_seqs; ++s) { + for (uint32_t i = 0; i < n_replay; ++i) { + const llama_pos pos = p0 + (llama_pos) i; + common_batch_add(batch, tok(s, pos), pos, { (llama_seq_id) s }, true); + } + } + ok = llama_decode(ctx_roll, batch) == 0; + ok = ok && llama_decode(ctx_ref, batch) == 0; + llama_batch_free(batch); + if (!ok) { + fprintf(stderr, "%s : multi-seq replay decode failed\n", __func__); + cleanup(); + return false; + } + + // identical ubatch shapes from bit-exact states: a correct implementation + // matches bitwise, so eps only allows backend scheduling noise + constexpr float eps = 1e-7f; + + float diff_max = 0.0f; + uint32_t seq_first = 0; + int32_t pos_first = -1; + for (uint32_t i = 0; i < n_seqs*n_replay; ++i) { + const float * l_roll = llama_get_logits_ith(ctx_roll, i); + const float * l_ref = llama_get_logits_ith(ctx_ref, i); + if (l_roll == nullptr || l_ref == nullptr) { + fprintf(stderr, "%s : missing multi-seq logits at index %u\n", __func__, i); + cleanup(); + return false; + } + for (int t = 0; t < n_vocab; ++t) { + const float diff = std::fabs(l_roll[t] - l_ref[t]); + if (diff > eps && pos_first < 0) { + seq_first = i/n_replay; + pos_first = p0 + (int32_t) (i%n_replay); + } + diff_max = std::max(diff_max, diff); + } + } + + if (diff_max > eps) { + fprintf(stderr, "%s : multi-seq split replay logits mismatch (max diff %g, first at seq %u pos %d)\n", + __func__, (double) diff_max, seq_first, pos_first); + cleanup(); + return false; + } + + fprintf(stderr, "%s : multi-seq split replay matched (max diff %g)\n", __func__, (double) diff_max); + + // seq-1-only decodes must be independent of seq 0's content: diverge seq 0 + // in ctx_ref only, then compare identical seq-1-only continuations bitwise + constexpr uint32_t n_tail = 4; + + { + llama_batch batch_tail = llama_batch_init(n_tail, 0, 1); + for (uint32_t i = 0; i < n_tail; ++i) { + const llama_pos pos = p0 + (llama_pos) (n_replay + i); + common_batch_add(batch_tail, tok(0, pos + 7), pos, { 0 }, false); + } + ok = llama_decode(ctx_ref, batch_tail) == 0; + llama_batch_free(batch_tail); + } + + float diff_tail = 0.0f; + for (uint32_t i = 0; i < n_tail && ok; ++i) { + const llama_pos pos = p0 + (llama_pos) (n_replay + i); + llama_batch batch_one = llama_batch_init(1, 0, 1); + common_batch_add(batch_one, tok(1, pos), pos, { 1 }, true); + ok = llama_decode(ctx_roll, batch_one) == 0; + ok = ok && llama_decode(ctx_ref, batch_one) == 0; + llama_batch_free(batch_one); + if (!ok) { + break; + } + + const float * l_roll = llama_get_logits_ith(ctx_roll, 0); + const float * l_ref = llama_get_logits_ith(ctx_ref, 0); + ok = l_roll != nullptr && l_ref != nullptr; + for (int t = 0; ok && t < n_vocab; ++t) { + diff_tail = std::max(diff_tail, std::fabs(l_roll[t] - l_ref[t])); + } + } + + if (!ok || diff_tail > eps) { + fprintf(stderr, "%s : seq-1-only decode leaked seq 0 state (ok=%d, max diff %g)\n", + __func__, ok ? 1 : 0, (double) diff_tail); + cleanup(); + return false; + } + + fprintf(stderr, "%s : seq-1-only decode independent of seq 0 (max diff %g)\n", __func__, (double) diff_tail); + cleanup(); + return true; +} + int main(int argc, char ** argv) { std::setlocale(LC_NUMERIC, "C"); @@ -220,5 +392,10 @@ int main(int argc, char ** argv) { llama_free(ctx_src); llama_free(ctx_dst); llama_free(ctx_dirty); + + if (!test_multi_seq_split_replay(params, model, n_vocab)) { + return 1; + } + return 0; }