Implementing DeepSeek V4
This commit is contained in:
+1
-1
@@ -2631,7 +2631,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
|||||||
).set_env("LLAMA_ARG_LOAD_MODE"));
|
).set_env("LLAMA_ARG_LOAD_MODE"));
|
||||||
add_opt(common_arg(
|
add_opt(common_arg(
|
||||||
{"--longhaul"},
|
{"--longhaul"},
|
||||||
"stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache",
|
"stream routed Qwen3.5 MoE, DeepSeek V4, Laguna, or Inkling experts through a bounded CPU or Metal cache",
|
||||||
[](common_params & params) {
|
[](common_params & params) {
|
||||||
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
|
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
|
||||||
}
|
}
|
||||||
|
|||||||
+9
-1
@@ -31,7 +31,7 @@ The normal startup warmup is skipped automatically in longhaul mode. Routed expe
|
|||||||
Longhaul currently requires:
|
Longhaul currently requires:
|
||||||
|
|
||||||
- all repeating layers on CPU, or all repeating layers on Metal
|
- all repeating layers on CPU, or all repeating layers on Metal
|
||||||
- Qwen3.5 MoE, Laguna, or Inkling architecture
|
- Qwen3.5 MoE, DeepSeek V4, Laguna, or Inkling architecture
|
||||||
- text generation without embeddings or LoRA adapters
|
- text generation without embeddings or LoRA adapters
|
||||||
|
|
||||||
CPU mode is available wherever the CPU backend is supported. Metal mode requires
|
CPU mode is available wherever the CPU backend is supported. Metal mode requires
|
||||||
@@ -63,6 +63,14 @@ once, and independent expert slices are read concurrently where the platform
|
|||||||
supports positional reads. CPU and shared Metal buffers are populated directly;
|
supports positional reads. CPU and shared Metal buffers are populated directly;
|
||||||
private Metal buffers use a staged fallback.
|
private Metal buffers use a staged fallback.
|
||||||
|
|
||||||
|
DeepSeek V4 keeps its shared expert, router weights, learned routing bias, and
|
||||||
|
hash-routing tables resident. Longhaul streams the routed gate, up, and down
|
||||||
|
expert banks. DeepSeek V4 Flash selects six routed experts per token; caches
|
||||||
|
with fewer than six slots use the existing staged MoE path. The startup log is
|
||||||
|
the authoritative source for slot capacity because the bytes per slot depend on
|
||||||
|
the GGUF tensor types and shard layout. MTP tensors remain outside the current
|
||||||
|
scope.
|
||||||
|
|
||||||
Inkling keeps its two shared experts resident and streams only the routed
|
Inkling keeps its two shared experts resident and streams only the routed
|
||||||
256-expert banks. Inkling-Small selects six routed experts per token. In the
|
256-expert banks. Inkling-Small selects six routed experts per token. In the
|
||||||
seven-shard Q8_0 model, one cache slot across all 40 MoE layers uses about
|
seven-shard Q8_0 model, one cache slot across all 40 MoE layers uses about
|
||||||
|
|||||||
+3
-2
@@ -1358,8 +1358,9 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
||||||
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA && arch != LLM_ARCH_INKLING) {
|
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_DEEPSEEK4 &&
|
||||||
throw std::runtime_error("longhaul currently requires a qwen35moe, laguna, or inkling model");
|
arch != LLM_ARCH_LAGUNA && arch != LLM_ARCH_INKLING) {
|
||||||
|
throw std::runtime_error("longhaul currently requires a qwen35moe, deepseek4, laguna, or inkling model");
|
||||||
}
|
}
|
||||||
if (hparams.n_layer_nextn != 0) {
|
if (hparams.n_layer_nextn != 0) {
|
||||||
throw std::runtime_error("longhaul does not support MTP tensors");
|
throw std::runtime_error("longhaul does not support MTP tensors");
|
||||||
|
|||||||
@@ -64,6 +64,8 @@ void llama_model_deepseek4::load_arch_hparams(llama_model_loader & ml) {
|
|||||||
void llama_model_deepseek4::load_arch_tensors(llama_model_loader &) {
|
void llama_model_deepseek4::load_arch_tensors(llama_model_loader &) {
|
||||||
LLAMA_LOAD_LOCALS;
|
LLAMA_LOAD_LOCALS;
|
||||||
|
|
||||||
|
const int expert_flags = params.load_mode == LLAMA_LOAD_MODE_LONGHAUL ? TENSOR_LONGHAUL : 0;
|
||||||
|
|
||||||
const int64_t q_lora_rank = hparams.n_lora_q;
|
const int64_t q_lora_rank = hparams.n_lora_q;
|
||||||
const int64_t n_ff_exp = hparams.n_ff_exp;
|
const int64_t n_ff_exp = hparams.n_ff_exp;
|
||||||
const int64_t n_expert_shared = hparams.n_expert_shared;
|
const int64_t n_expert_shared = hparams.n_expert_shared;
|
||||||
@@ -136,9 +138,9 @@ void llama_model_deepseek4::load_arch_tensors(llama_model_loader &) {
|
|||||||
}
|
}
|
||||||
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
|
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
|
||||||
|
|
||||||
layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i), {n_embd, n_ff_exp, n_expert}, 0);
|
layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i), {n_embd, n_ff_exp, n_expert}, expert_flags);
|
||||||
layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd, n_expert}, 0);
|
layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd, n_expert}, expert_flags);
|
||||||
layer.ffn_up_exps = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i), {n_embd, n_ff_exp, n_expert}, 0);
|
layer.ffn_up_exps = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i), {n_embd, n_ff_exp, n_expert}, expert_flags);
|
||||||
|
|
||||||
layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), {n_embd, n_ff_exp * n_expert_shared}, 0);
|
layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), {n_embd, n_ff_exp * n_expert_shared}, 0);
|
||||||
layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_exp * n_expert_shared, n_embd }, 0);
|
layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_exp * n_expert_shared, n_embd }, 0);
|
||||||
|
|||||||
@@ -291,6 +291,22 @@ llama_test(
|
|||||||
set_tests_properties(test-longhaul-cpu-inkling-six-slots PROPERTIES
|
set_tests_properties(test-longhaul-cpu-inkling-six-slots PROPERTIES
|
||||||
FIXTURES_REQUIRED generate-models
|
FIXTURES_REQUIRED generate-models
|
||||||
)
|
)
|
||||||
|
llama_test(
|
||||||
|
test-longhaul
|
||||||
|
NAME test-longhaul-cpu-deepseek4
|
||||||
|
ARGS --cpu-model "${MODEL_DIR}/deepseek4-moe.gguf" 147456 1
|
||||||
|
)
|
||||||
|
set_tests_properties(test-longhaul-cpu-deepseek4 PROPERTIES
|
||||||
|
FIXTURES_REQUIRED generate-models
|
||||||
|
)
|
||||||
|
llama_test(
|
||||||
|
test-longhaul
|
||||||
|
NAME test-longhaul-cpu-deepseek4-six-slots
|
||||||
|
ARGS --cpu-model "${MODEL_DIR}/deepseek4-moe.gguf" 884736 6
|
||||||
|
)
|
||||||
|
set_tests_properties(test-longhaul-cpu-deepseek4-six-slots PROPERTIES
|
||||||
|
FIXTURES_REQUIRED generate-models
|
||||||
|
)
|
||||||
if (APPLE AND GGML_METAL)
|
if (APPLE AND GGML_METAL)
|
||||||
llama_test(
|
llama_test(
|
||||||
test-longhaul
|
test-longhaul
|
||||||
@@ -308,6 +324,22 @@ if (APPLE AND GGML_METAL)
|
|||||||
set_tests_properties(test-longhaul-metal-inkling-six-slots PROPERTIES
|
set_tests_properties(test-longhaul-metal-inkling-six-slots PROPERTIES
|
||||||
FIXTURES_REQUIRED generate-models
|
FIXTURES_REQUIRED generate-models
|
||||||
)
|
)
|
||||||
|
llama_test(
|
||||||
|
test-longhaul
|
||||||
|
NAME test-longhaul-metal-deepseek4
|
||||||
|
ARGS --metal-model "${MODEL_DIR}/deepseek4-moe.gguf" 147456 1
|
||||||
|
)
|
||||||
|
set_tests_properties(test-longhaul-metal-deepseek4 PROPERTIES
|
||||||
|
FIXTURES_REQUIRED generate-models
|
||||||
|
)
|
||||||
|
llama_test(
|
||||||
|
test-longhaul
|
||||||
|
NAME test-longhaul-metal-deepseek4-six-slots
|
||||||
|
ARGS --metal-model "${MODEL_DIR}/deepseek4-moe.gguf" 884736 6
|
||||||
|
)
|
||||||
|
set_tests_properties(test-longhaul-metal-deepseek4-six-slots PROPERTIES
|
||||||
|
FIXTURES_REQUIRED generate-models
|
||||||
|
)
|
||||||
endif()
|
endif()
|
||||||
llama_build_and_test(test-token-cache.cpp)
|
llama_build_and_test(test-token-cache.cpp)
|
||||||
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
||||||
|
|||||||
+56
-16
@@ -60,6 +60,12 @@ static void set_tensor_data(struct ggml_tensor * tensor, void * userdata) {
|
|||||||
tmp[i] = ggml_fp32_to_fp16(dis(gen));
|
tmp[i] = ggml_fp32_to_fp16(dis(gen));
|
||||||
}
|
}
|
||||||
ggml_backend_tensor_set(tensor, tmp.data(), 0, ggml_nbytes(tensor));
|
ggml_backend_tensor_set(tensor, tmp.data(), 0, ggml_nbytes(tensor));
|
||||||
|
} else if (tensor->type == GGML_TYPE_I32 && strstr(tensor->name, "ffn_gate_tid2eid") != nullptr) {
|
||||||
|
std::vector<int32_t> tmp(ne);
|
||||||
|
for (int64_t i = 0; i < ne; i++) {
|
||||||
|
tmp[i] = i % tensor->ne[0];
|
||||||
|
}
|
||||||
|
ggml_backend_tensor_set(tensor, tmp.data(), 0, ggml_nbytes(tensor));
|
||||||
} else {
|
} else {
|
||||||
GGML_ABORT("fatal error");
|
GGML_ABORT("fatal error");
|
||||||
}
|
}
|
||||||
@@ -110,6 +116,10 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
n_embd = 128;
|
n_embd = 128;
|
||||||
n_head = 1;
|
n_head = 1;
|
||||||
n_ff = 192;
|
n_ff = 192;
|
||||||
|
} else if (arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
|
n_embd = 64;
|
||||||
|
n_head = 4;
|
||||||
|
n_ff = 96;
|
||||||
} else if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
|
} else if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
|
||||||
n_layer = 3;
|
n_layer = 3;
|
||||||
} else if (arch == LLM_ARCH_CHAMELEON) {
|
} else if (arch == LLM_ARCH_CHAMELEON) {
|
||||||
@@ -158,6 +168,9 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
}
|
}
|
||||||
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT, n_head_per_layer);
|
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT, n_head_per_layer);
|
||||||
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT_KV, n_head_per_layer);
|
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT_KV, n_head_per_layer);
|
||||||
|
} else if (arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT, n_head);
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT_KV, uint32_t(1));
|
||||||
} else {
|
} else {
|
||||||
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT, n_head);
|
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT, n_head);
|
||||||
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT_KV, n_head);
|
ms.add_kv(LLM_KV_ATTENTION_HEAD_COUNT_KV, n_head);
|
||||||
@@ -174,6 +187,10 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, uint32_t(64));
|
ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, uint32_t(64));
|
||||||
ms.add_kv(LLM_KV_ATTENTION_KEY_LENGTH_MLA, uint32_t(192));
|
ms.add_kv(LLM_KV_ATTENTION_KEY_LENGTH_MLA, uint32_t(192));
|
||||||
ms.add_kv(LLM_KV_ATTENTION_VALUE_LENGTH_MLA, uint32_t(128));
|
ms.add_kv(LLM_KV_ATTENTION_VALUE_LENGTH_MLA, uint32_t(128));
|
||||||
|
} else if (arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_KEY_LENGTH, n_embd_head);
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_VALUE_LENGTH, n_embd_head);
|
||||||
|
ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, n_embd_head/2);
|
||||||
} else if (arch == LLM_ARCH_MINIMAX_M3) {
|
} else if (arch == LLM_ARCH_MINIMAX_M3) {
|
||||||
// partial rotary: n_rot must not exceed the indexer key length (64)
|
// partial rotary: n_rot must not exceed the indexer key length (64)
|
||||||
ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, uint32_t(64));
|
ms.add_kv(LLM_KV_ROPE_DIMENSION_COUNT, uint32_t(64));
|
||||||
@@ -183,7 +200,7 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
ms.add_kv(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, 1e-5f);
|
ms.add_kv(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, 1e-5f);
|
||||||
ms.add_kv(LLM_KV_ATTENTION_GROUPNORM_EPS, 1e-5f);
|
ms.add_kv(LLM_KV_ATTENTION_GROUPNORM_EPS, 1e-5f);
|
||||||
ms.add_kv(LLM_KV_ATTENTION_GROUPNORM_GROUPS, uint32_t(8));
|
ms.add_kv(LLM_KV_ATTENTION_GROUPNORM_GROUPS, uint32_t(8));
|
||||||
ms.add_kv(LLM_KV_ATTENTION_Q_LORA_RANK, uint32_t(512));
|
ms.add_kv(LLM_KV_ATTENTION_Q_LORA_RANK, arch == LLM_ARCH_DEEPSEEK4 ? uint32_t(32) : uint32_t(512));
|
||||||
ms.add_kv(LLM_KV_ATTENTION_KV_LORA_RANK, uint32_t(512));
|
ms.add_kv(LLM_KV_ATTENTION_KV_LORA_RANK, uint32_t(512));
|
||||||
ms.add_kv(LLM_KV_ATTENTION_RELATIVE_BUCKETS_COUNT, uint32_t(8));
|
ms.add_kv(LLM_KV_ATTENTION_RELATIVE_BUCKETS_COUNT, uint32_t(8));
|
||||||
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW, n_ctx/8);
|
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW, n_ctx/8);
|
||||||
@@ -226,15 +243,31 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
if (moe) {
|
if (moe) {
|
||||||
ms.add_kv(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, n_ff);
|
ms.add_kv(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, n_ff);
|
||||||
ms.add_kv(LLM_KV_INTERLEAVE_MOE_LAYER_STEP, uint32_t(2));
|
ms.add_kv(LLM_KV_INTERLEAVE_MOE_LAYER_STEP, uint32_t(2));
|
||||||
ms.add_kv(LLM_KV_EXPERT_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(8) : uint32_t(2));
|
ms.add_kv(LLM_KV_EXPERT_COUNT, arch == LLM_ARCH_INKLING || arch == LLM_ARCH_DEEPSEEK4 ? uint32_t(8) : uint32_t(2));
|
||||||
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(6) : uint32_t(1));
|
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, arch == LLM_ARCH_INKLING || arch == LLM_ARCH_DEEPSEEK4 ? uint32_t(6) : uint32_t(1));
|
||||||
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(2) : uint32_t(1));
|
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(2) : uint32_t(1));
|
||||||
ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, uint32_t(2)); // sigmoid
|
ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, arch == LLM_ARCH_DEEPSEEK4 ? uint32_t(4) : uint32_t(2)); // sqrtsoftplus or sigmoid
|
||||||
ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f);
|
ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f);
|
||||||
ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1));
|
ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1));
|
||||||
if (arch == LLM_ARCH_INKLING) {
|
if (arch == LLM_ARCH_INKLING || arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE, 1.0f);
|
ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE, 1.0f);
|
||||||
}
|
}
|
||||||
|
if (arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
|
ms.add_kv(LLM_KV_EXPERT_WEIGHTS_NORM, true);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
|
ms.add_kv(LLM_KV_SWIGLU_CLAMP_EXP, 10.0f);
|
||||||
|
ms.add_kv(LLM_KV_SWIGLU_CLAMP_SHEXP, 10.0f);
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_OUTPUT_GROUP_COUNT, uint32_t(2));
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_OUTPUT_LORA_RANK, uint32_t(16));
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_COMPRESS_ROPE_FREQ_BASE, 160000.0f);
|
||||||
|
ms.add_kv(LLM_KV_ATTENTION_COMPRESS_RATIOS, std::vector<uint32_t>{0, 4});
|
||||||
|
ms.add_kv(LLM_KV_HYPER_CONNECTION_COUNT, uint32_t(4));
|
||||||
|
ms.add_kv(LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS, uint32_t(2));
|
||||||
|
ms.add_kv(LLM_KV_HYPER_CONNECTION_EPSILON, 1.0e-6f);
|
||||||
|
ms.add_kv(LLM_KV_HASH_LAYER_COUNT, uint32_t(1));
|
||||||
}
|
}
|
||||||
|
|
||||||
if (arch == LLM_ARCH_INKLING) {
|
if (arch == LLM_ARCH_INKLING) {
|
||||||
@@ -266,6 +299,19 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
ms.add_kv(LLM_KV_WKV_HEAD_SIZE, n_embd/n_head);
|
ms.add_kv(LLM_KV_WKV_HEAD_SIZE, n_embd/n_head);
|
||||||
ms.add_kv(LLM_KV_SHORTCONV_L_CACHE, uint32_t(3));
|
ms.add_kv(LLM_KV_SHORTCONV_L_CACHE, uint32_t(3));
|
||||||
|
|
||||||
|
if (arch == LLM_ARCH_DEEPSEEK4) {
|
||||||
|
ggml_init_params params = {
|
||||||
|
/*.mem_size =*/ ggml_tensor_overhead(),
|
||||||
|
/*.mem_buffer =*/ nullptr,
|
||||||
|
/*.no_alloc =*/ true,
|
||||||
|
};
|
||||||
|
ggml_context_ptr tensor_ctx(ggml_init(params));
|
||||||
|
ggml_tensor * tid2eid = ggml_new_tensor_2d(
|
||||||
|
tensor_ctx.get(), GGML_TYPE_I32, 6, n_vocab);
|
||||||
|
ggml_set_name(tid2eid, "blk.0.ffn_gate_tid2eid.weight");
|
||||||
|
gguf_add_tensor(ms.gguf_ctx, tid2eid);
|
||||||
|
}
|
||||||
|
|
||||||
for (uint32_t il = 0; il < n_layer; il++) {
|
for (uint32_t il = 0; il < n_layer; il++) {
|
||||||
ggml_tensor t;
|
ggml_tensor t;
|
||||||
memset(&t, 0, sizeof(ggml_tensor));
|
memset(&t, 0, sizeof(ggml_tensor));
|
||||||
@@ -370,6 +416,7 @@ static bool moe_mandatory(const llm_arch arch) {
|
|||||||
case LLM_ARCH_DEEPSEEK:
|
case LLM_ARCH_DEEPSEEK:
|
||||||
case LLM_ARCH_DEEPSEEK2:
|
case LLM_ARCH_DEEPSEEK2:
|
||||||
case LLM_ARCH_DEEPSEEK32:
|
case LLM_ARCH_DEEPSEEK32:
|
||||||
|
case LLM_ARCH_DEEPSEEK4:
|
||||||
case LLM_ARCH_GLM4_MOE:
|
case LLM_ARCH_GLM4_MOE:
|
||||||
case LLM_ARCH_GLM_DSA:
|
case LLM_ARCH_GLM_DSA:
|
||||||
case LLM_ARCH_EXAONE_MOE:
|
case LLM_ARCH_EXAONE_MOE:
|
||||||
@@ -450,10 +497,6 @@ static bool arch_supported(const llm_arch arch) {
|
|||||||
if (arch == LLM_ARCH_DEEPSEEK2OCR) {
|
if (arch == LLM_ARCH_DEEPSEEK2OCR) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
if (arch == LLM_ARCH_DEEPSEEK4) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// FIXME: these hit scheduler/view-backed-output issues with WebGPU on CI.
|
// FIXME: these hit scheduler/view-backed-output issues with WebGPU on CI.
|
||||||
#ifdef GGML_USE_WEBGPU
|
#ifdef GGML_USE_WEBGPU
|
||||||
if (arch == LLM_ARCH_DEEPSEEK32 || arch == LLM_ARCH_GLM_DSA) {
|
if (arch == LLM_ARCH_DEEPSEEK32 || arch == LLM_ARCH_GLM_DSA) {
|
||||||
@@ -515,15 +558,11 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
|
|||||||
const std::string path = dir + "/" + llm_arch_name(arch) + (moe ? "-moe.gguf" : "-dense.gguf");
|
const std::string path = dir + "/" + llm_arch_name(arch) + (moe ? "-moe.gguf" : "-dense.gguf");
|
||||||
LOG_INF("%s: Saving %s model (%s) to %s...\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense", path.c_str());
|
LOG_INF("%s: Saving %s model (%s) to %s...\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense", path.c_str());
|
||||||
const bool private_fixture =
|
const bool private_fixture =
|
||||||
arch == LLM_ARCH_LAGUNA || arch == LLM_ARCH_INKLING;
|
arch == LLM_ARCH_DEEPSEEK4 || arch == LLM_ARCH_LAGUNA || arch == LLM_ARCH_INKLING;
|
||||||
if (llama_model_saver_supports_arch(arch) && !private_fixture) {
|
if (llama_model_saver_supports_arch(arch) && !private_fixture) {
|
||||||
llama_model_save_to_file(model_and_ctx.first.get(), path.c_str());
|
llama_model_save_to_file(model_and_ctx.first.get(), path.c_str());
|
||||||
} else {
|
} else {
|
||||||
// Laguna and Inkling are not supported by the production
|
// Preserve exact fixture metadata for architectures the production saver cannot round-trip yet.
|
||||||
// model saver yet.
|
|
||||||
// Preserve the exact synthetic fixture metadata and attach the
|
|
||||||
// initialized model tensors so longhaul can be tested from a
|
|
||||||
// real file without broadening the saver API.
|
|
||||||
gguf_context_ptr fixture(gguf_init_empty());
|
gguf_context_ptr fixture(gguf_init_empty());
|
||||||
// Model initialization normalizes the caller's GGUF context,
|
// Model initialization normalizes the caller's GGUF context,
|
||||||
// so regenerate the exact private metadata for the file.
|
// so regenerate the exact private metadata for the file.
|
||||||
@@ -669,7 +708,8 @@ static int test_backends(const llm_arch target_arch, const size_t seed, const gg
|
|||||||
FILE * file = tmpfile(); // Can be null on Windows without administrator privileges.
|
FILE * file = tmpfile(); // Can be null on Windows without administrator privileges.
|
||||||
// FIXME: when adding a tensor to a gguf_context a copy is made, this changes the pointer which the meta backend
|
// FIXME: when adding a tensor to a gguf_context a copy is made, this changes the pointer which the meta backend
|
||||||
// in turn uses to map the tensors to their simple equivalents - this is fundamentally incompatible
|
// in turn uses to map the tensors to their simple equivalents - this is fundamentally incompatible
|
||||||
if (file != nullptr && llama_model_saver_supports_arch(arch) && dc.split_mode != LLAMA_SPLIT_MODE_TENSOR) {
|
if (file != nullptr && llama_model_saver_supports_arch(arch) && arch != LLM_ARCH_DEEPSEEK4 &&
|
||||||
|
dc.split_mode != LLAMA_SPLIT_MODE_TENSOR) {
|
||||||
GGML_ASSERT(model_and_ctx_dev.first && model_and_ctx_dev.second);
|
GGML_ASSERT(model_and_ctx_dev.first && model_and_ctx_dev.second);
|
||||||
llama_model_saver ms = llama_model_saver(model_and_ctx_dev.first.get());
|
llama_model_saver ms = llama_model_saver(model_and_ctx_dev.first.get());
|
||||||
ms.add_kv_from_model();
|
ms.add_kv_from_model();
|
||||||
|
|||||||
+1
-1
@@ -61,7 +61,7 @@
|
|||||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
| `--longhaul` | stream routed Qwen3.5 MoE, DeepSeek V4, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||||
|
|||||||
@@ -144,7 +144,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
|
|||||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
| `--longhaul` | stream routed Qwen3.5 MoE, DeepSeek V4, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||||
|
|||||||
@@ -78,7 +78,7 @@ For the full list of features, please refer to [server's changelog](https://gith
|
|||||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
| `--longhaul` | stream routed Qwen3.5 MoE, DeepSeek V4, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||||
|
|||||||
Reference in New Issue
Block a user