This commit is contained in:
Owen Qwen
2026-07-30 21:20:05 -05:00
parent d5458aa44e
commit 22f42628dd
25 changed files with 1038 additions and 28 deletions
+1 -1
View File
@@ -2631,7 +2631,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
).set_env("LLAMA_ARG_LOAD_MODE"));
add_opt(common_arg(
{"--longhaul"},
"stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache",
"stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache",
[](common_params & params) {
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
}
+119
View File
@@ -2530,6 +2530,118 @@ static common_chat_params common_chat_params_init_minimax_m3(const common_chat_t
return data;
}
// Inkling's TML format can emit multiple typed content blocks in one turn.
// <|end_message|> closes a block while content_model_end_sampling closes the
// generation, so the generic single-block parser is not sufficient.
static common_chat_params common_chat_params_init_inkling(
const common_chat_template & tmpl,
const autoparser::generation_params & inputs) {
common_chat_params data;
const std::string MSG_MODEL = "<|message_model|>";
const std::string MSG_USER = "<|message_user|>";
const std::string MSG_SYSTEM = "<|message_system|>";
const std::string MSG_TOOL = "<|message_tool|>";
const std::string THINK = "<|content_thinking|>";
const std::string TEXT = "<|content_text|>";
const std::string END_MESSAGE = "<|end_message|>";
const std::string END_SAMPLING = "<|content_model_end_sampling|>";
const std::string INVOKE_TOOL = "<|content_invoke_tool_json|>";
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);
data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs);
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
data.supports_thinking = true;
data.thinking_start_tag = THINK;
data.thinking_end_tags = {END_MESSAGE};
data.preserved_tokens = {
MSG_MODEL, MSG_USER, MSG_SYSTEM, MSG_TOOL,
THINK, TEXT, END_MESSAGE, END_SAMPLING, INVOKE_TOOL,
};
data.message_delimiters = {
{ COMMON_CHAT_ROLE_ASSISTANT, MSG_MODEL },
{ COMMON_CHAT_ROLE_USER, MSG_USER },
{ COMMON_CHAT_ROLE_SYSTEM, MSG_SYSTEM },
{ COMMON_CHAT_ROLE_TOOL, MSG_TOOL },
};
const bool has_tools = inputs.tools.is_array() && !inputs.tools.empty();
const bool extract_reasoning =
inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
if (inputs.has_continuation()) {
const auto & msg = inputs.continue_msg;
data.generation_prompt = MSG_MODEL + THINK + msg.reasoning_content;
if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
data.generation_prompt += END_MESSAGE + TEXT + msg.render_content();
}
data.prompt += data.generation_prompt;
}
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
auto generation_prompt = p.literal(MSG_MODEL);
auto end = p.end();
common_peg_parser reasoning_block = p.eps();
if (extract_reasoning) {
reasoning_block =
p.literal(THINK) +
p.reasoning(p.until_one_of({ END_MESSAGE, TEXT, END_SAMPLING })) +
p.optional(p.literal(END_MESSAGE));
} else {
reasoning_block = p.content(
p.literal(THINK) +
p.until_one_of({ END_MESSAGE, TEXT, END_SAMPLING }) +
p.optional(p.literal(END_MESSAGE)));
}
auto reasoning = p.optional(reasoning_block);
auto text_block =
p.optional(p.literal(MSG_MODEL)) +
p.optional(p.literal(TEXT)) +
p.content(p.until_one_of({ THINK, END_MESSAGE, END_SAMPLING })) +
p.optional(p.literal(END_MESSAGE));
auto text_content =
p.one_or_more(p.choice({ reasoning_block, text_block }));
if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
return generation_prompt + reasoning + text_content +
p.optional(p.literal(END_SAMPLING)) + end;
}
auto tool_section = p.standard_json_tools(
INVOKE_TOOL, END_MESSAGE, inputs.tools,
/* parallel_tool_calls = */ false,
/* force_tool_calls = */ true,
/* name_key = */ "name",
/* args_key = */ "args",
/* array_wrapped = */ false,
/* function_is_key = */ false,
/* call_id_key = */ "",
/* gen_call_id_key = */ "",
/* parameters_order = */ {},
/* accept_openai_wrapper = */ false);
auto tool_block =
p.optional(p.literal(MSG_MODEL)) +
p.until_one_of({ INVOKE_TOOL, TEXT, THINK, END_MESSAGE, END_SAMPLING }) +
tool_section;
auto tool_calls = inputs.parallel_tool_calls
? p.one_or_more(tool_block)
: tool_block;
auto mixed_body =
p.one_or_more(p.choice({ tool_block, reasoning_block, text_block }));
auto body = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED
? tool_calls
: mixed_body;
return generation_prompt + reasoning + body +
p.optional(p.literal(END_SAMPLING)) + end;
});
data.parser = parser.save();
return data;
}
namespace workaround {
static void map_developer_role_to_system(json & messages) {
@@ -2947,6 +3059,13 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
return common_chat_params_init_cohere2moe(tmpl, params);
}
if (src.find("<|content_thinking|>") != std::string::npos &&
src.find("<|content_text|>") != std::string::npos &&
src.find("<|message_model|>") != std::string::npos) {
LOG_DBG("Using specialized template: Inkling\n");
return common_chat_params_init_inkling(tmpl, params);
}
if (is_lfm2_template(src)) {
LOG_DBG("Using specialized template: LFM2\n");
return common_chat_params_init_lfm2(tmpl, params, /* tool_list_tokens = */ true);
+16 -1
View File
@@ -31,7 +31,7 @@ The normal startup warmup is skipped automatically in longhaul mode. Routed expe
Longhaul currently requires:
- all repeating layers on CPU, or all repeating layers on Metal
- Qwen3.5 MoE or Laguna architecture
- Qwen3.5 MoE, Laguna, or Inkling architecture
- text generation without embeddings or LoRA adapters
CPU mode is available wherever the CPU backend is supported. Metal mode requires
@@ -63,6 +63,21 @@ once, and independent expert slices are read concurrently where the platform
supports positional reads. CPU and shared Metal buffers are populated directly;
private Metal buffers use a staged fallback.
Inkling keeps its two shared experts resident and streams only the routed
256-expert banks. Inkling-Small selects six routed experts per token. For the
current three-shard UD-IQ1_S model, a 2 GiB cache provides seven routed-expert
slots and therefore executes each MoE layer in one stage:
```sh
llama-cli \
--model /path/to/Inkling-Small-UD-IQ1_S-00001-of-00003.gguf \
--longhaul \
--longhaul-cache 2
```
Add `--n-gpu-layers 0` to run entirely on CPU; on macOS the default all-Metal
placement is supported.
## Benchmarking prompt processing
`llama-bench` accepts the longhaul load mode and cache budget:
+32
View File
@@ -144,6 +144,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
{ LLM_ARCH_TALKIE, "talkie" },
{ LLM_ARCH_MELLUM, "mellum" },
{ LLM_ARCH_NANBEIGE, "nanbeige" },
{ LLM_ARCH_INKLING, "inkling" },
{ LLM_ARCH_UNKNOWN, "(unknown)" },
};
@@ -320,6 +321,15 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
{ LLM_KV_NORM_BEFORE_FC, "%s.norm_before_fc" },
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
{ LLM_KV_INKLING_D_REL, "%s.d_rel" },
{ LLM_KV_INKLING_REL_EXTENT, "%s.rel_extent" },
{ LLM_KV_INKLING_REL_EXTENT_SWA, "%s.rel_extent_swa" },
{ LLM_KV_INKLING_SHORTCONV_KERNEL, "%s.shortconv_kernel" },
{ LLM_KV_INKLING_DENSE_BLOCK_COUNT, "%s.dense_block_count" },
{ LLM_KV_INKLING_LOGIT_SCALE_DENOM, "%s.logit_scale_denom" },
{ LLM_KV_INKLING_LOG_SCALING_N_FLOOR, "%s.log_scaling_n_floor" },
{ LLM_KV_INKLING_LOG_SCALING_ALPHA, "%s.log_scaling_alpha" },
{ LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, "%s.unpadded_vocab_size" },
// sentence-transformers dense modules feature dims
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
{ LLM_KV_DENSE_2_FEAT_OUT, "%s.dense_2_feat_out" },
@@ -592,9 +602,19 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
{ LLM_TENSOR_SHORTCONV_CONV, "blk.%d.shortconv.conv" },
{ LLM_TENSOR_SHORTCONV_INPROJ, "blk.%d.shortconv.in_proj" },
{ LLM_TENSOR_SHORTCONV_OUTPROJ, "blk.%d.shortconv.out_proj" },
{ LLM_TENSOR_ATTN_R, "blk.%d.attn_r" },
{ LLM_TENSOR_ATTN_REL_PROJ, "blk.%d.attn_rel_proj" },
{ LLM_TENSOR_SHORTCONV_K, "blk.%d.shortconv_k" },
{ LLM_TENSOR_SHORTCONV_V, "blk.%d.shortconv_v" },
{ LLM_TENSOR_SHORTCONV_ATTN, "blk.%d.shortconv_attn" },
{ LLM_TENSOR_SHORTCONV_MLP, "blk.%d.shortconv_mlp" },
{ LLM_TENSOR_FFN_GSCALE, "blk.%d.ffn_gscale" },
{ LLM_TENSOR_FFN_GATE_CHEXPS, "blk.%d.ffn_gate_chexps" },
{ LLM_TENSOR_FFN_DOWN_CHEXPS, "blk.%d.ffn_down_chexps" },
{ LLM_TENSOR_FFN_UP_CHEXPS, "blk.%d.ffn_up_chexps" },
{ LLM_TENSOR_FFN_GATE_SHEXPS, "blk.%d.ffn_gate_shexp" },
{ LLM_TENSOR_FFN_DOWN_SHEXPS, "blk.%d.ffn_down_shexp" },
{ LLM_TENSOR_FFN_UP_SHEXPS, "blk.%d.ffn_up_shexp" },
{ LLM_TENSOR_VISEXP_ATTN_QKV, "blk.%d.vis_attn_qkv" },
{ LLM_TENSOR_VISEXP_ATTN_OUT, "blk.%d.vis_attn_output" },
{ LLM_TENSOR_VISEXP_FFN_GATE, "blk.%d.vis_gate" },
@@ -795,6 +815,9 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
{LLM_TENSOR_FFN_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_GATE_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_DOWN_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_DOWN_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_GATE_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_UP_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_GATE_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_UP_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
{LLM_TENSOR_FFN_EXP_PROBS_B, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ADD}},
@@ -836,6 +859,13 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
{LLM_TENSOR_SHORTCONV_CONV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
{LLM_TENSOR_SHORTCONV_INPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_SHORTCONV_OUTPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_ATTN_R, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_ATTN_REL_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_SHORTCONV_K, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
{LLM_TENSOR_SHORTCONV_V, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
{LLM_TENSOR_SHORTCONV_ATTN, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
{LLM_TENSOR_SHORTCONV_MLP, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
{LLM_TENSOR_FFN_GSCALE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
{LLM_TENSOR_VISEXP_ATTN_QKV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_VISEXP_ATTN_OUT, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
{LLM_TENSOR_VISEXP_FFN_GATE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
@@ -968,6 +998,7 @@ bool llm_arch_is_hybrid(const llm_arch & arch) {
case LLM_ARCH_KIMI_LINEAR:
case LLM_ARCH_QWEN35:
case LLM_ARCH_QWEN35MOE:
case LLM_ARCH_INKLING:
return true;
default:
return false;
@@ -1024,6 +1055,7 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
case LLM_ARCH_MINIMAX_M3:
case LLM_ARCH_MISTRAL4:
case LLM_ARCH_KIMI_LINEAR:
case LLM_ARCH_INKLING:
return false;
default:
return true;
+20
View File
@@ -149,6 +149,7 @@ enum llm_arch {
LLM_ARCH_MINIMAX_M3,
LLM_ARCH_DFLASH,
LLM_ARCH_NANBEIGE,
LLM_ARCH_INKLING,
LLM_ARCH_UNKNOWN,
};
@@ -366,6 +367,15 @@ enum llm_kv {
LLM_KV_NORM_BEFORE_FC,
LLM_KV_SHORTCONV_L_CACHE,
LLM_KV_INKLING_D_REL,
LLM_KV_INKLING_REL_EXTENT,
LLM_KV_INKLING_REL_EXTENT_SWA,
LLM_KV_INKLING_SHORTCONV_KERNEL,
LLM_KV_INKLING_DENSE_BLOCK_COUNT,
LLM_KV_INKLING_LOGIT_SCALE_DENOM,
LLM_KV_INKLING_LOG_SCALING_N_FLOOR,
LLM_KV_INKLING_LOG_SCALING_ALPHA,
LLM_KV_INKLING_UNPADDED_VOCAB_SIZE,
LLM_KV_XIELU_ALPHA_N,
LLM_KV_XIELU_ALPHA_P,
@@ -434,6 +444,9 @@ enum llm_tensor {
LLM_TENSOR_FFN_DOWN_CHEXPS,
LLM_TENSOR_FFN_GATE_CHEXPS,
LLM_TENSOR_FFN_UP_CHEXPS,
LLM_TENSOR_FFN_DOWN_SHEXPS,
LLM_TENSOR_FFN_GATE_SHEXPS,
LLM_TENSOR_FFN_UP_SHEXPS,
LLM_TENSOR_FFN_EXP_PROBS_B,
LLM_TENSOR_FFN_LATENT_DOWN,
LLM_TENSOR_FFN_LATENT_UP,
@@ -595,6 +608,13 @@ enum llm_tensor {
LLM_TENSOR_SHORTCONV_CONV,
LLM_TENSOR_SHORTCONV_INPROJ,
LLM_TENSOR_SHORTCONV_OUTPROJ,
LLM_TENSOR_ATTN_R,
LLM_TENSOR_ATTN_REL_PROJ,
LLM_TENSOR_SHORTCONV_K,
LLM_TENSOR_SHORTCONV_V,
LLM_TENSOR_SHORTCONV_ATTN,
LLM_TENSOR_SHORTCONV_MLP,
LLM_TENSOR_FFN_GSCALE,
LLM_TENSOR_VISEXP_ATTN_QKV,
LLM_TENSOR_VISEXP_ATTN_OUT,
LLM_TENSOR_VISEXP_FFN_GATE,
+113
View File
@@ -1765,6 +1765,119 @@ ggml_tensor * llm_graph_context::build_ffn(
return cur;
}
ggml_tensor * llm_graph_context::build_moe_experts(
ggml_tensor * cur,
ggml_tensor * selected_experts,
ggml_tensor * weights,
ggml_tensor * up_exps,
ggml_tensor * gate_exps,
ggml_tensor * down_exps,
int64_t n_expert_used,
llm_ffn_op_type type_op,
int il) const {
GGML_ASSERT(cur != nullptr);
GGML_ASSERT(selected_experts != nullptr);
GGML_ASSERT(weights != nullptr);
GGML_ASSERT(up_exps != nullptr && down_exps != nullptr);
const int64_t n_embd = cur->ne[0];
const int64_t n_tokens = cur->ne[1];
// Ensure routing and weights are available before a longhaul cache
// callback resolves and loads the first expert chunk.
ggml_build_forward_expand(gf, weights);
ggml_tensor * moe_inp = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
const int64_t chunk_size = longhaul
? std::min<int64_t>(n_expert_used, longhaul->capacity())
: n_expert_used;
GGML_ASSERT(chunk_size > 0);
ggml_tensor * moe_out = nullptr;
int chunk = 0;
for (int64_t first = 0; first < n_expert_used; first += chunk_size, ++chunk) {
const int64_t count = std::min<int64_t>(chunk_size, n_expert_used - first);
ggml_tensor * selected_chunk = selected_experts;
ggml_tensor * weights_chunk = weights;
if (count != n_expert_used) {
selected_chunk = ggml_view_2d(
ctx0, selected_experts, count, n_tokens,
selected_experts->nb[1], first*selected_experts->nb[0]);
weights_chunk = ggml_view_3d(
ctx0, weights, 1, count, n_tokens,
weights->nb[1], weights->nb[2], first*weights->nb[1]);
}
ggml_tensor * selected_moe = selected_chunk;
if (longhaul) {
selected_moe = ggml_dup(ctx0, selected_chunk);
ggml_format_name(selected_moe, "longhaul.remap.%d.%d", il, chunk);
ggml_build_forward_expand(gf, selected_moe);
}
ggml_tensor * up = build_lora_mm_id(up_exps, moe_inp, selected_moe);
ggml_tensor * act = gate_exps
? build_lora_mm_id(gate_exps, moe_inp, selected_moe)
: up;
switch (type_op) {
case LLM_FFN_SILU:
act = gate_exps
? ggml_swiglu_split(ctx0, act, up)
: ggml_silu(ctx0, act);
break;
case LLM_FFN_GELU:
act = gate_exps
? ggml_geglu_split(ctx0, act, up)
: ggml_gelu(ctx0, act);
break;
case LLM_FFN_RELU:
act = gate_exps
? ggml_reglu_split(ctx0, act, up)
: ggml_relu(ctx0, act);
break;
default:
GGML_ABORT("unsupported routed expert activation");
}
ggml_tensor * experts = build_lora_mm_id(
down_exps, act, selected_moe);
experts = ggml_mul(ctx0, experts, weights_chunk);
ggml_build_forward_expand(gf, experts);
ggml_tensor * chunk_out = nullptr;
for (int64_t i = 0; i < count; ++i) {
ggml_tensor * expert = ggml_view_2d(
ctx0, experts, n_embd, n_tokens,
experts->nb[2], i*experts->nb[1]);
ggml_build_forward_expand(gf, expert);
chunk_out = chunk_out ? ggml_add(ctx0, chunk_out, expert) : expert;
ggml_build_forward_expand(gf, chunk_out);
}
if (count == 1) {
chunk_out = ggml_cont(ctx0, chunk_out);
}
ggml_tensor * contribution = chunk_out;
if (longhaul) {
ggml_tensor * release = ggml_cont(ctx0, chunk_out);
ggml_format_name(release, "longhaul.release.%d.%d", il, chunk);
ggml_build_forward_expand(gf, release);
if (chunk_size < n_expert_used) {
contribution = release;
}
}
moe_out = moe_out ? ggml_add(ctx0, moe_out, contribution) : contribution;
ggml_build_forward_expand(gf, moe_out);
}
cb(moe_out, "ffn_moe_out", il);
return moe_out;
}
ggml_tensor * llm_graph_context::build_moe_ffn(
ggml_tensor * cur,
ggml_tensor * gate_inp,
+14
View File
@@ -1000,6 +1000,20 @@ struct llm_graph_context {
llm_ffn_gate_type type_gate,
int il) const;
// Execute experts for an architecture that supplies its own selected IDs
// and weights. In longhaul mode this transparently stages those experts
// through the bounded cache.
ggml_tensor * build_moe_experts(
ggml_tensor * cur,
ggml_tensor * selected_experts,
ggml_tensor * weights,
ggml_tensor * up_exps,
ggml_tensor * gate_exps,
ggml_tensor * down_exps,
int64_t n_expert_used,
llm_ffn_op_type type_op,
int il) const;
// build MoE FFN without bias tensors
ggml_tensor * build_moe_ffn(
ggml_tensor * cur,
+4
View File
@@ -191,6 +191,10 @@ uint32_t llama_hparams::n_embd_k_idx(uint32_t il) const {
}
uint32_t llama_hparams::n_embd_r() const {
if (n_embd_r_impl != 0) {
return n_embd_r_impl;
}
if (wkv_head_size != 0) {
// for RWKV models
return token_shift_count * n_embd;
+12
View File
@@ -80,6 +80,18 @@ struct llama_hparams {
uint32_t n_shortconv_l_cache = 0;
// Explicit recurrent-state size override. Inkling packs four independent
// short-convolution histories into the recurrent cache for every layer.
uint32_t n_embd_r_impl = 0;
// Inkling relative-attention and output metadata.
uint32_t inkling_d_rel = 0;
uint32_t inkling_rel_extent = 0;
uint32_t inkling_rel_extent_swa = 0;
uint32_t inkling_log_n_floor = 0;
float inkling_log_alpha = 0.0f;
uint32_t inkling_unpadded_n_vocab = 0;
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_arr;
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_kv_arr;
std::array<uint32_t, LLAMA_MAX_LAYERS> n_ff_arr;
+38
View File
@@ -1931,6 +1931,37 @@ void llama_kv_cache::set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch
}
}
void llama_kv_cache::set_input_pos_rel_flat(
ggml_tensor * dst,
const llama_ubatch * ubatch,
uint32_t extent) const {
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
GGML_ASSERT(dst->ne[1] == (int64_t) ubatch->n_tokens);
int32_t * data = (int32_t *) dst->data;
const int64_t n_kv = dst->ne[0];
// Each query owns an extent+1-row block in the flattened relative-logit
// table. The final row is the zero-bias sentinel for empty/out-of-band KV
// cells.
for (int64_t i = 0; i < (int64_t) ubatch->n_tokens; ++i) {
const llama_seq_id seq_id = ubatch->seq_id[i][0];
const auto & cells = v_cells[seq_to_stream[seq_id]];
const llama_pos p1 = ubatch->pos[i];
for (int64_t j = 0; j < n_kv; ++j) {
int32_t rel = (int32_t) extent;
if (!cells.is_empty(j)) {
const llama_pos d = p1 - cells.pos_get(j);
if (d >= 0 && d < (llama_pos) extent) {
rel = (int32_t) d;
}
}
data[i*n_kv + j] = (int32_t) (i*(extent + 1)) + rel;
}
}
}
void llama_kv_cache::set_input_k_rot(ggml_tensor * dst) const {
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
@@ -2896,6 +2927,13 @@ void llama_kv_cache_context::set_input_pos_bucket(ggml_tensor * dst, const llama
kv->set_input_pos_bucket(dst, ubatch);
}
void llama_kv_cache_context::set_input_pos_rel_flat(
ggml_tensor * dst,
const llama_ubatch * ubatch,
uint32_t extent) const {
kv->set_input_pos_rel_flat(dst, ubatch, extent);
}
void llama_kv_cache_context::set_input_k_rot(ggml_tensor * dst) const {
kv->set_input_k_rot(dst);
}
+4
View File
@@ -215,6 +215,8 @@ public:
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
void set_input_pos_rel_flat(
ggml_tensor * dst, const llama_ubatch * ubatch, uint32_t extent) const;
void set_input_k_rot(ggml_tensor * dst) const;
void set_input_v_rot(ggml_tensor * dst) const;
@@ -405,6 +407,8 @@ public:
void set_input_k_shift (ggml_tensor * dst) const;
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
void set_input_pos_rel_flat(
ggml_tensor * dst, const llama_ubatch * ubatch, uint32_t extent) const;
void set_input_k_rot(ggml_tensor * dst) const;
void set_input_v_rot(ggml_tensor * dst) const;
+1
View File
@@ -29,6 +29,7 @@ bool llama_model_saver_supports_arch(llm_arch arch) {
case LLM_ARCH_STEP35:
case LLM_ARCH_MELLUM:
case LLM_ARCH_LAGUNA:
case LLM_ARCH_INKLING:
return false;
default:
return true;
+6 -3
View File
@@ -312,6 +312,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
return new llama_model_kimi_linear(params);
case LLM_ARCH_STEP35:
return new llama_model_step35(params);
case LLM_ARCH_INKLING:
return new llama_model_inkling(params);
default:
throw std::runtime_error(std::string("unsupported model architecture: '") + llm_arch_name(arch) + "'");
}
@@ -1356,8 +1358,8 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
}
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA) {
throw std::runtime_error("longhaul currently requires a qwen35moe or laguna model");
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA && arch != LLM_ARCH_INKLING) {
throw std::runtime_error("longhaul currently requires a qwen35moe, laguna, or inkling model");
}
if (hparams.n_layer_nextn != 0) {
throw std::runtime_error("longhaul does not support MTP tensors");
@@ -2165,7 +2167,7 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,
// layer filters, so pick the right one here
llama_memory_hybrid::layer_filter_cb filter_attn = nullptr;
llama_memory_hybrid::layer_filter_cb filter_recr = nullptr;
if (arch == LLM_ARCH_FALCON_H1) {
if (arch == LLM_ARCH_FALCON_H1 || arch == LLM_ARCH_INKLING) {
filter_attn = [&](uint32_t) { return true; };
filter_recr = [&](uint32_t) { return true; };
} else if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
@@ -2506,6 +2508,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
case LLM_ARCH_NEMOTRON_H:
case LLM_ARCH_NEMOTRON_H_MOE:
case LLM_ARCH_KIMI_LINEAR:
case LLM_ARCH_INKLING:
return LLAMA_ROPE_TYPE_NONE;
// use what we call a normal RoPE, operating on pairs of consecutive head values
+9
View File
@@ -534,6 +534,15 @@ struct llama_layer {
struct llama_layer_shortconv shortconv;
struct llama_layer_nextn nextn;
// Inkling relative attention and short-convolution weights.
struct ggml_tensor * wr = nullptr;
struct ggml_tensor * attn_rel_proj = nullptr;
struct ggml_tensor * shortconv_k = nullptr;
struct ggml_tensor * shortconv_v = nullptr;
struct ggml_tensor * shortconv_attn = nullptr;
struct ggml_tensor * shortconv_mlp = nullptr;
struct ggml_tensor * ffn_gscale = nullptr;
};
struct llama_device {
+17 -1
View File
@@ -434,6 +434,13 @@ struct llm_tokenizer_bpe : llm_tokenizer {
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))*((?=[\\p{L}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))+((?=[\\p{L}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
};
break;
case LLAMA_VOCAB_PRE_TYPE_INKLING:
// o200k-family expression with combining marks included in
// letter lookaheads, as used by the Inkling tokenizer.
regex_exprs = {
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}\\p{M}])([^a-z]))*((?=[\\p{L}\\p{M}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}\\p{M}])([^a-z]))+((?=[\\p{L}\\p{M}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
};
break;
case LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI:
// Same lookaheads as GPT4O but with \p{M} added so combining marks
// (diacritics) attach to their base letters. Avoids excessive
@@ -2324,6 +2331,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
tokenizer_pre == "talkie") {
pre_type = LLAMA_VOCAB_PRE_TYPE_GPT4O;
clean_spaces = false;
} else if (
tokenizer_pre == "inkling") {
pre_type = LLAMA_VOCAB_PRE_TYPE_INKLING;
clean_spaces = false;
} else if (
tokenizer_pre == "granite-embed-multi-97m") {
pre_type = LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI;
@@ -2607,7 +2618,12 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
if (suppress_idx != -1) {
const int n = gguf_get_arr_n(ctx, suppress_idx);
const int32_t * data = (const int32_t *) gguf_get_arr_data(ctx, suppress_idx);
suppress_tokens.assign(data, data + n);
suppress_tokens.reserve(n);
for (int i = 0; i < n; ++i) {
if (data[i] >= 0 && data[i] < (int32_t) id_to_token.size()) {
suppress_tokens.push_back(data[i]);
}
}
}
}
+1
View File
@@ -65,6 +65,7 @@ enum llama_vocab_pre_type {
LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI = 54,
LLAMA_VOCAB_PRE_TYPE_MELLUM2 = 55,
LLAMA_VOCAB_PRE_TYPE_LAGUNA = 56,
LLAMA_VOCAB_PRE_TYPE_INKLING = 57,
};
struct LLM_KV;
+489
View File
@@ -0,0 +1,489 @@
// Inkling: relative-position attention with per-layer packed short-convolution
// state and a mixture of routed and shared experts.
#include "models.h"
#include "../llama-memory-hybrid-iswa.h"
#include "../llama-memory-recurrent.h"
#include <algorithm>
#include <cmath>
void llama_model_inkling::load_arch_hparams(llama_model_loader & ml) {
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp);
ml.get_key(LLM_KV_EXPERT_SHARED_COUNT, hparams.n_expert_shared);
ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE, hparams.expert_weights_scale);
ml.get_key(LLM_KV_EXPERT_GATING_FUNC, hparams.expert_gating_func, false);
ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa);
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, hparams.is_swa_impl, hparams.n_layer());
for (uint32_t il = 0; il < hparams.n_layer(); ++il) {
hparams.is_recr_impl[il] = 1;
}
ml.get_key(LLM_KV_INKLING_D_REL, hparams.inkling_d_rel);
ml.get_key(LLM_KV_INKLING_REL_EXTENT, hparams.inkling_rel_extent);
ml.get_key(LLM_KV_INKLING_REL_EXTENT_SWA, hparams.inkling_rel_extent_swa);
ml.get_key(LLM_KV_INKLING_SHORTCONV_KERNEL, hparams.n_shortconv_l_cache);
ml.get_key(LLM_KV_INKLING_DENSE_BLOCK_COUNT, hparams.n_layer_dense_lead);
float logit_scale_denom = 0.0f;
ml.get_key(LLM_KV_INKLING_LOGIT_SCALE_DENOM, logit_scale_denom);
GGML_ASSERT(logit_scale_denom != 0.0f);
hparams.f_logit_scale = 1.0f/logit_scale_denom;
ml.get_key(LLM_KV_INKLING_LOG_SCALING_N_FLOOR, hparams.inkling_log_n_floor, false);
ml.get_key(LLM_KV_INKLING_LOG_SCALING_ALPHA, hparams.inkling_log_alpha, false);
ml.get_key(LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, hparams.inkling_unpadded_n_vocab, false);
GGML_ASSERT(hparams.n_shortconv_l_cache > 1);
GGML_ASSERT(hparams.inkling_d_rel > 0);
GGML_ASSERT(hparams.inkling_rel_extent > 0 && hparams.inkling_rel_extent_swa > 0);
// Four streams per layer: K, V, attention output, and MLP output.
const uint32_t d_conv = hparams.n_shortconv_l_cache - 1;
hparams.n_embd_r_impl = d_conv *
(hparams.n_embd_k_gqa_max() + hparams.n_embd_v_gqa_max() + 2*hparams.n_embd);
type = LLM_TYPE_UNKNOWN;
}
void llama_model_inkling::load_arch_tensors(llama_model_loader &) {
LLAMA_LOAD_LOCALS;
const int64_t head_dim = hparams.n_embd_head_k();
const int64_t d_rel = hparams.inkling_d_rel;
const int64_t K = hparams.n_shortconv_l_cache;
const int64_t n_ff_exp = hparams.n_ff_exp;
const int64_t n_shexp = hparams.n_expert_shared;
const int expert_flags = params.load_mode == LLAMA_LOAD_MODE_LONGHAUL ? TENSOR_LONGHAUL : 0;
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
tok_norm = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD_NORM, "weight", 0), {n_embd}, 0);
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, 0);
for (int i = 0; i < n_layer; ++i) {
auto & layer = layers[i];
const int64_t n_head_kv_i = hparams.n_head_kv(i);
const int64_t kvw = n_head_kv_i*head_dim;
const int64_t rel_extent = hparams.is_swa(i)
? hparams.inkling_rel_extent_swa
: hparams.inkling_rel_extent;
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
layer.wq = create_tensor(tn(LLM_TENSOR_ATTN_Q, "weight", i), {n_embd, n_head*head_dim}, 0);
layer.wk = create_tensor(tn(LLM_TENSOR_ATTN_K, "weight", i), {n_embd, kvw}, 0);
layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V, "weight", i), {n_embd, kvw}, 0);
layer.wr = create_tensor(tn(LLM_TENSOR_ATTN_R, "weight", i), {n_embd, n_head*d_rel}, 0);
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_head*head_dim, n_embd}, 0);
layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {head_dim}, 0);
layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {head_dim}, 0);
layer.attn_rel_proj = create_tensor(
tn(LLM_TENSOR_ATTN_REL_PROJ, "weight", i), {rel_extent, d_rel}, 0);
layer.shortconv_k = create_tensor(tn(LLM_TENSOR_SHORTCONV_K, "weight", i), {K, kvw}, 0);
layer.shortconv_v = create_tensor(tn(LLM_TENSOR_SHORTCONV_V, "weight", i), {K, kvw}, 0);
layer.shortconv_attn = create_tensor(tn(LLM_TENSOR_SHORTCONV_ATTN, "weight", i), {K, n_embd}, 0);
layer.shortconv_mlp = create_tensor(tn(LLM_TENSOR_SHORTCONV_MLP, "weight", i), {K, n_embd}, 0);
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
layer.ffn_gscale = create_tensor(tn(LLM_TENSOR_FFN_GSCALE, "weight", i), {1}, 0);
if (i < (int) hparams.n_layer_dense_lead) {
const int64_t n_ff_i = hparams.n_ff(i);
layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff_i}, 0);
layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff_i}, 0);
layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff_i, n_embd}, 0);
} else {
GGML_ASSERT(n_expert > 0 && n_expert_used > 0 && n_shexp > 0);
// The final rows are shared-expert sink logits.
layer.ffn_gate_inp = create_tensor(
tn(LLM_TENSOR_FFN_GATE_INP, "weight", i), {n_embd, n_expert + n_shexp}, 0);
layer.ffn_exp_probs_b = create_tensor(
tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias", i), {n_expert}, 0);
layer.ffn_gate_exps = create_tensor(
tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i),
{n_embd, n_ff_exp, n_expert}, expert_flags);
layer.ffn_up_exps = create_tensor(
tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i),
{n_embd, n_ff_exp, n_expert}, expert_flags);
layer.ffn_down_exps = create_tensor(
tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i),
{n_ff_exp, n_embd, n_expert}, expert_flags);
// Shared experts stay resident and are never part of the longhaul cache.
layer.ffn_gate_shexp = create_tensor(
tn(LLM_TENSOR_FFN_GATE_SHEXPS, "weight", i),
{n_embd, n_ff_exp, n_shexp}, 0);
layer.ffn_up_shexp = create_tensor(
tn(LLM_TENSOR_FFN_UP_SHEXPS, "weight", i),
{n_embd, n_ff_exp, n_shexp}, 0);
layer.ffn_down_shexp = create_tensor(
tn(LLM_TENSOR_FFN_DOWN_SHEXPS, "weight", i),
{n_ff_exp, n_embd, n_shexp}, 0);
}
}
}
class llm_graph_input_inkling : public llm_graph_input_i {
public:
llm_graph_input_inkling(
const llama_hparams & hparams,
const llama_memory_hybrid_iswa_context * mctx) :
hparams(hparams), mctx(mctx) {
}
void set_input(const llama_ubatch * ubatch) override {
if (tau) {
GGML_ASSERT(ggml_backend_buffer_is_host(tau->buffer));
float * data = (float *) tau->data;
const float n_floor = (float) hparams.inkling_log_n_floor;
const float alpha = hparams.inkling_log_alpha;
for (int64_t i = 0; i < (int64_t) ubatch->n_tokens; ++i) {
const float eff = (float) (ubatch->pos[i] + 1)/n_floor;
data[i] = 1.0f + alpha*logf(std::max(eff, 1.0f));
}
}
if (rel_idx) {
mctx->get_attn()->get_base()->set_input_pos_rel_flat(
rel_idx, ubatch, hparams.inkling_rel_extent);
}
if (rel_idx_swa) {
mctx->get_attn()->get_swa()->set_input_pos_rel_flat(
rel_idx_swa, ubatch, hparams.inkling_rel_extent_swa);
}
if (vocab_mask) {
GGML_ASSERT(ggml_backend_buffer_is_host(vocab_mask->buffer));
float * data = (float *) vocab_mask->data;
for (int64_t id = 0; id < vocab_mask->ne[0]; ++id) {
data[id] = id < (int64_t) hparams.inkling_unpadded_n_vocab ? 0.0f : -INFINITY;
}
}
if (shexp_idx) {
GGML_ASSERT(ggml_backend_buffer_is_host(shexp_idx->buffer));
int32_t * data = (int32_t *) shexp_idx->data;
for (int64_t j = 0; j < shexp_idx->ne[1]; ++j) {
for (int64_t s = 0; s < shexp_idx->ne[0]; ++s) {
data[j*shexp_idx->ne[0] + s] = (int32_t) s;
}
}
}
}
ggml_tensor * tau = nullptr;
ggml_tensor * rel_idx = nullptr;
ggml_tensor * rel_idx_swa = nullptr;
ggml_tensor * vocab_mask = nullptr;
ggml_tensor * shexp_idx = nullptr;
private:
const llama_hparams hparams;
const llama_memory_hybrid_iswa_context * mctx;
};
std::unique_ptr<llm_graph_context> llama_model_inkling::build_arch_graph(
const llm_graph_params & params) const {
return std::make_unique<graph>(*this, params);
}
llama_model_inkling::graph::graph(
const llama_model & model,
const llm_graph_params & params) :
llm_graph_context(params) {
const int64_t head_dim = hparams.n_embd_head_k();
const int64_t d_rel = hparams.inkling_d_rel;
const int64_t d_conv = hparams.n_shortconv_l_cache - 1;
const int64_t n_embd_r = hparams.n_embd_r();
const int64_t kw_max = hparams.n_embd_k_gqa_max();
const int64_t vw_max = hparams.n_embd_v_gqa_max();
const int64_t off_k = 0;
const int64_t off_v = d_conv*kw_max;
const int64_t off_attn = d_conv*(kw_max + vw_max);
const int64_t off_mlp = d_conv*(kw_max + vw_max + n_embd);
const auto * mctx_hyb = static_cast<const llama_memory_hybrid_iswa_context *>(mctx);
const auto * mctx_recr = mctx_hyb->get_recr();
const auto * mctx_attn = mctx_hyb->get_attn();
const uint32_t kv_head = mctx_recr->get_head();
const int64_t n_seq_tokens = ubatch.n_seq_tokens;
const int64_t n_seqs = ubatch.n_seqs;
GGML_ASSERT(n_seqs != 0);
GGML_ASSERT(ubatch.equal_seqs());
GGML_ASSERT(ubatch.n_tokens == n_seq_tokens*n_seqs);
bool has_global = false;
bool has_local = false;
for (int il = 0; il < n_layer; ++il) {
has_local |= hparams.is_swa(il);
has_global |= !hparams.is_swa(il);
}
auto inp = std::make_unique<llm_graph_input_inkling>(hparams, mctx_hyb);
if (hparams.inkling_log_n_floor > 0 && has_global) {
inp->tau = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 1, 1, n_tokens);
ggml_set_input(inp->tau);
ggml_set_name(inp->tau, "inkling_tau");
}
if (has_global) {
inp->rel_idx = ggml_new_tensor_2d(
ctx0, GGML_TYPE_I32, mctx_attn->get_base()->get_n_kv(), n_tokens);
ggml_set_input(inp->rel_idx);
ggml_set_name(inp->rel_idx, "inkling_rel_idx");
}
if (has_local) {
inp->rel_idx_swa = ggml_new_tensor_2d(
ctx0, GGML_TYPE_I32, mctx_attn->get_swa()->get_n_kv(), n_tokens);
ggml_set_input(inp->rel_idx_swa);
ggml_set_name(inp->rel_idx_swa, "inkling_rel_idx_swa");
}
const int64_t n_vocab = model.vocab.n_tokens();
if (!cparams.embeddings &&
hparams.inkling_unpadded_n_vocab > 0 &&
(int64_t) hparams.inkling_unpadded_n_vocab < n_vocab) {
inp->vocab_mask = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_vocab);
ggml_set_input(inp->vocab_mask);
ggml_set_name(inp->vocab_mask, "inkling_vocab_mask");
}
if (hparams.n_expert_shared > 0 && (uint32_t) n_layer > hparams.n_layer_dense_lead) {
inp->shexp_idx = ggml_new_tensor_2d(
ctx0, GGML_TYPE_I32, hparams.n_expert_shared, n_tokens);
ggml_set_input(inp->shexp_idx);
ggml_set_name(inp->shexp_idx, "inkling_shexp_idx");
}
ggml_tensor * tau = inp->tau;
ggml_tensor * rel_idx = inp->rel_idx;
ggml_tensor * rel_idx_swa = inp->rel_idx_swa;
ggml_tensor * vocab_mask = inp->vocab_mask;
ggml_tensor * shexp_idx = inp->shexp_idx;
res->add_input(std::move(inp));
auto * inp_hybrid = build_inp_mem_hybrid_iswa();
ggml_tensor * conv_rs_cur = nullptr;
auto build_sconv = [&](ggml_tensor * x2d, ggml_tensor * kernel, int64_t off, int il) {
const int64_t w = x2d->ne[0];
ggml_tensor * x3 = ggml_reshape_3d(ctx0, x2d, w, n_seq_tokens, n_seqs);
ggml_tensor * xt = ggml_transpose(ctx0, x3);
ggml_tensor * conv_state = mctx_recr->get_r_l(il);
GGML_ASSERT(conv_rs_cur != nullptr);
const size_t sz = ggml_element_size(conv_rs_cur);
ggml_tensor * state = ggml_view_3d(
ctx0, conv_rs_cur, d_conv, w, n_seqs,
d_conv*sz, conv_rs_cur->nb[1], off*sz);
ggml_tensor * sx = ggml_concat(ctx0, state, xt, 0);
ggml_tensor * new_state = ggml_view_3d(
ctx0, sx, d_conv, w, n_seqs,
sx->nb[1], sx->nb[2], (sx->ne[0] - d_conv)*sx->nb[0]);
ggml_tensor * state_dst = ggml_view_3d(
ctx0, conv_state, d_conv, w, n_seqs,
d_conv*sz, n_embd_r*sz, (kv_head*n_embd_r + off)*sz);
ggml_build_forward_expand(gf, ggml_cpy(ctx0, new_state, state_dst));
ggml_tensor * conv_out = ggml_ssm_conv(ctx0, sx, kernel);
return ggml_reshape_2d(ctx0, ggml_add(ctx0, x3, conv_out), w, n_seq_tokens*n_seqs);
};
auto build_attn_block = [&](ggml_tensor * cur, int il) {
const auto & layer = model.layers[il];
const bool is_swa = hparams.is_swa(il);
const int64_t n_head_kv = hparams.n_head_kv(il);
const int64_t rel_extent = is_swa
? hparams.inkling_rel_extent_swa
: hparams.inkling_rel_extent;
ggml_tensor * q = build_lora_mm(layer.wq, cur);
ggml_tensor * k = build_lora_mm(layer.wk, cur);
ggml_tensor * v = build_lora_mm(layer.wv, cur);
ggml_tensor * r = build_lora_mm(layer.wr, cur);
k = build_sconv(k, layer.shortconv_k, off_k, il);
v = build_sconv(v, layer.shortconv_v, off_v, il);
q = ggml_reshape_3d(ctx0, q, head_dim, n_head, n_tokens);
k = ggml_reshape_3d(ctx0, k, head_dim, n_head_kv, n_tokens);
v = ggml_reshape_3d(ctx0, v, head_dim, n_head_kv, n_tokens);
q = build_norm(q, layer.attn_q_norm, nullptr, LLM_NORM_RMS, il);
k = build_norm(k, layer.attn_k_norm, nullptr, LLM_NORM_RMS, il);
if (tau && !is_swa) {
q = ggml_mul(ctx0, q, tau);
}
ggml_tensor * r2 = ggml_reshape_2d(ctx0, r, d_rel, n_head*n_tokens);
ggml_tensor * proj = ggml_cont(ctx0, ggml_transpose(ctx0, layer.attn_rel_proj));
ggml_tensor * rel = ggml_mul_mat(ctx0, proj, r2);
ggml_mul_mat_set_prec(rel, GGML_PREC_F32);
rel = ggml_reshape_3d(ctx0, rel, rel_extent, n_head, n_tokens);
if (tau && !is_swa) {
rel = ggml_mul(ctx0, rel, tau);
}
// Scale Q rather than build_attn's joint logits so the relative term
// remains unscaled, matching the reference implementation.
q = ggml_scale(ctx0, q, 1.0f/(float) head_dim);
rel = ggml_pad(ctx0, rel, 1, 0, 0, 0);
rel = ggml_cont(ctx0, ggml_permute(ctx0, rel, 1, 0, 2, 3));
rel = ggml_reshape_2d(ctx0, rel, n_head, (rel_extent + 1)*n_tokens);
ggml_tensor * idx = is_swa ? rel_idx_swa : rel_idx;
GGML_ASSERT(idx != nullptr);
const int64_t n_kv = idx->ne[0];
ggml_tensor * idx1 = ggml_reshape_1d(ctx0, idx, n_kv*n_tokens);
ggml_tensor * kq_b = ggml_get_rows(ctx0, rel, idx1);
kq_b = ggml_reshape_3d(ctx0, kq_b, n_head, n_kv, n_tokens);
kq_b = ggml_cont(ctx0, ggml_permute(ctx0, kq_b, 2, 0, 1, 3));
auto * inp_attn = inp_hybrid->get_attn();
const int64_t n_stream =
(is_swa ? inp_attn->get_kq_mask_swa() : inp_attn->get_kq_mask())->ne[3];
if (n_stream > 1) {
kq_b = ggml_view_4d(
ctx0, kq_b, n_kv, n_tokens/n_stream, n_head, n_stream,
kq_b->nb[1], kq_b->nb[2], (n_tokens/n_stream)*kq_b->nb[1], 0);
}
return build_attn(
inp_attn, layer.wo, nullptr, nullptr,
q, k, v, kq_b, nullptr, nullptr, 1.0f, il);
};
auto build_dense_ffn = [&](ggml_tensor * cur, int il) {
cur = build_ffn(
cur,
model.layers[il].ffn_up, nullptr, nullptr,
model.layers[il].ffn_gate, nullptr, nullptr,
model.layers[il].ffn_down, nullptr, nullptr,
nullptr, LLM_FFN_SILU, LLM_FFN_PAR, il);
return ggml_mul(ctx0, cur, model.layers[il].ffn_gscale);
};
auto build_moe = [&](ggml_tensor * cur, int il) {
const auto & layer = model.layers[il];
const int64_t n_shexp = hparams.n_expert_shared;
ggml_tensor * logits = build_lora_mm(layer.ffn_gate_inp, cur);
ggml_mul_mat_set_prec(logits, GGML_PREC_F32);
const size_t lsz = ggml_element_size(logits);
ggml_tensor * routed = ggml_cont(ctx0, ggml_view_2d(
ctx0, logits, n_expert, n_tokens, logits->nb[1], 0));
ggml_tensor * shared_logits = ggml_view_2d(
ctx0, logits, n_shexp, n_tokens, logits->nb[1], n_expert*lsz);
ggml_tensor * scores = ggml_add(
ctx0, ggml_sigmoid(ctx0, routed), layer.ffn_exp_probs_b);
ggml_tensor * selected = ggml_argsort_top_k(
ctx0, scores, n_expert_used);
ggml_tensor * routed3 = ggml_reshape_3d(ctx0, routed, 1, n_expert, n_tokens);
ggml_tensor * topk_logits = ggml_reshape_2d(
ctx0, ggml_get_rows(ctx0, routed3, selected), n_expert_used, n_tokens);
ggml_tensor * all_logits = ggml_concat(ctx0, topk_logits, shared_logits, 0);
ggml_tensor * w = ggml_neg(
ctx0, ggml_softplus(ctx0, ggml_neg(ctx0, all_logits)));
w = ggml_soft_max(ctx0, w);
w = ggml_scale(ctx0, w, hparams.expert_weights_scale);
w = ggml_mul(ctx0, w, layer.ffn_gscale);
const size_t wsz = ggml_element_size(w);
ggml_tensor * weights = ggml_cont(ctx0, ggml_view_2d(
ctx0, w, n_expert_used, n_tokens, w->nb[1], 0));
weights = ggml_reshape_3d(ctx0, weights, 1, n_expert_used, n_tokens);
ggml_tensor * moe_out = build_moe_experts(
cur, selected, weights,
layer.ffn_up_exps, layer.ffn_gate_exps, layer.ffn_down_exps,
n_expert_used, LLM_FFN_SILU, il);
GGML_ASSERT(shexp_idx != nullptr);
ggml_tensor * xr = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
ggml_tensor * gs = build_lora_mm_id(layer.ffn_gate_shexp, xr, shexp_idx);
ggml_tensor * us = build_lora_mm_id(layer.ffn_up_shexp, xr, shexp_idx);
ggml_tensor * hs = ggml_swiglu_split(ctx0, gs, us);
ggml_tensor * gammas = ggml_cont(ctx0, ggml_view_2d(
ctx0, w, n_shexp, n_tokens, w->nb[1], n_expert_used*wsz));
hs = ggml_mul(ctx0, hs, ggml_reshape_3d(ctx0, gammas, 1, n_shexp, n_tokens));
ggml_tensor * ds = build_lora_mm_id(layer.ffn_down_shexp, hs, shexp_idx);
for (int64_t s = 0; s < n_shexp; ++s) {
ggml_tensor * e = ggml_view_2d(
ctx0, ds, n_embd, n_tokens, ds->nb[2], s*ds->nb[1]);
moe_out = ggml_add(ctx0, moe_out, e);
}
return moe_out;
};
ggml_tensor * cur = build_inp_embd(model.tok_embd);
if (ubatch.token) {
cur = build_norm(cur, model.tok_norm, nullptr, LLM_NORM_RMS, -1);
}
ggml_build_forward_expand(gf, cur);
for (int il = 0; il < n_layer; ++il) {
conv_rs_cur = build_rs(
inp_hybrid->get_recr(), mctx_recr->get_r_l(il), n_embd_r, n_seqs);
ggml_tensor * attn_in = build_norm(
cur, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il);
ggml_tensor * attn_out = build_attn_block(attn_in, il);
attn_out = build_sconv(
attn_out, model.layers[il].shortconv_attn, off_attn, il);
cur = ggml_add(ctx0, cur, attn_out);
ggml_tensor * ffn_in = build_norm(
cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);
ggml_tensor * ffn_out = il < (int) hparams.n_layer_dense_lead
? build_dense_ffn(ffn_in, il)
: build_moe(ffn_in, il);
ffn_out = build_sconv(
ffn_out, model.layers[il].shortconv_mlp, off_mlp, il);
cur = ggml_add(ctx0, cur, ffn_out);
cur = build_cvec(cur, il);
cb(cur, "l_out", il);
}
ggml_tensor * inp_out_ids = build_inp_out_ids();
if (inp_out_ids) {
cur = ggml_get_rows(ctx0, cur, inp_out_ids);
}
cur = build_norm(cur, model.output_norm, nullptr, LLM_NORM_RMS, -1);
res->t_embd = cur;
if (!cparams.embeddings) {
cur = ggml_scale(ctx0, cur, hparams.f_logit_scale);
cur = build_lora_mm(model.output, cur);
if (model.output->type == GGML_TYPE_F32) {
ggml_mul_mat_set_prec(cur, GGML_PREC_F32);
}
if (vocab_mask) {
cur = ggml_add(ctx0, cur, vocab_mask);
}
res->t_logits = cur;
}
ggml_build_forward_expand(gf, cur);
}
+12
View File
@@ -1865,6 +1865,18 @@ struct llama_model_lfm2moe : public llama_model_base {
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};
struct llama_model_inkling : public llama_model_base {
llama_model_inkling(const struct llama_model_params & params) : llama_model_base(params) {}
void load_arch_hparams(llama_model_loader & ml) override;
void load_arch_tensors(llama_model_loader & ml) override;
struct graph : public llm_graph_context {
graph(const llama_model & model, const llm_graph_params & params);
};
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
};
struct llama_model_smallthinker : public llama_model_base {
llama_model_smallthinker(const struct llama_model_params & params) : llama_model_base(params) {}
+18
View File
@@ -275,6 +275,24 @@ llama_test(
set_tests_properties(test-longhaul-cpu-laguna PROPERTIES
FIXTURES_REQUIRED generate-models
)
llama_test(
test-longhaul
NAME test-longhaul-cpu-inkling
ARGS --cpu-model "${MODEL_DIR}/inkling-moe.gguf" 73728
)
set_tests_properties(test-longhaul-cpu-inkling PROPERTIES
FIXTURES_REQUIRED generate-models
)
if (APPLE AND GGML_METAL)
llama_test(
test-longhaul
NAME test-longhaul-metal-inkling
ARGS --metal-model "${MODEL_DIR}/inkling-moe.gguf" 73728
)
set_tests_properties(test-longhaul-metal-inkling PROPERTIES
FIXTURES_REQUIRED generate-models
)
endif()
llama_build_and_test(test-token-cache.cpp)
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
+49 -1
View File
@@ -91,6 +91,7 @@ static void test_normalize_quotes_with_embedded_quotes(testing & t);
static void test_tagged_args_with_embedded_quotes(testing & t);
static void test_role_markers_all_templates(testing & t);
static void test_inkling_tml_parser(testing & t);
int main(int argc, char * argv[]) {
testing t(std::cout);
@@ -117,6 +118,7 @@ int main(int argc, char * argv[]) {
t.test("standard_json_tools", test_standard_json_tools_formats);
t.test("normalize_quotes_to_json", test_normalize_quotes_to_json);
t.test("tagged_args_embedded_quotes", test_tagged_args_with_embedded_quotes);
t.test("inkling_tml", test_inkling_tml_parser);
t.test("role_markers_all_templates", test_role_markers_all_templates);
return t.summary();
@@ -2075,6 +2077,53 @@ static void test_role_markers_all_templates(testing & t) {
}
}
static void test_inkling_tml_parser(testing & t) {
const std::string tmpl_src =
"{# <|content_thinking|><|content_text|> #}"
"{%- for message in messages -%}"
"<|message_user|><|content_text|>{{ message.content }}<|end_message|>"
"{%- endfor -%}"
"{%- if add_generation_prompt -%}<|message_model|>{%- endif -%}";
common_chat_templates_ptr tmpls =
common_chat_templates_init(nullptr, tmpl_src);
common_chat_templates_inputs inputs;
common_chat_msg user_message;
user_message.role = "user";
user_message.content = "Hello";
inputs.messages = { user_message };
inputs.add_generation_prompt = true;
inputs.reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
inputs.use_jinja = true;
const common_chat_params params =
common_chat_templates_apply(tmpls.get(), inputs);
t.assert_equal(
"Inkling uses the native PEG parser",
COMMON_CHAT_FORMAT_PEG_NATIVE,
params.format);
t.assert_true("Inkling exposes thinking blocks", params.supports_thinking);
t.assert_equal(
"Inkling generation marker",
"<|message_model|>",
params.generation_prompt);
common_peg_arena arena;
arena.load(params.parser);
common_chat_parser_params parser_params(params);
parser_params.reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
const common_chat_msg parsed = common_chat_peg_parse(
arena,
"<|content_thinking|>plan<|end_message|>"
"<|message_model|><|content_text|>answer<|end_message|>"
"<|content_model_end_sampling|>",
false,
parser_params);
t.assert_equal("reasoning block", "plan", parsed.reasoning_content);
t.assert_equal("text block", "answer", parsed.content);
}
// Test that reproduces the Seed-OSS template issue with embedded quotes
static void test_tagged_args_with_embedded_quotes(testing & t) {
json tools = build_edit_tool();
@@ -2192,4 +2241,3 @@ static void test_tagged_args_with_embedded_quotes(testing & t) {
}
}
}
+40 -7
View File
@@ -114,6 +114,11 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
n_layer = 3;
} else if (arch == LLM_ARCH_CHAMELEON) {
n_vocab = 10240;
} else if (arch == LLM_ARCH_INKLING) {
n_embd = 64;
n_head = 2;
n_ff = 96;
n_layer = 2;
}
const uint32_t n_embd_head = n_embd / n_head;
@@ -198,6 +203,10 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
pattern.push_back(il % 2);
}
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, pattern);
} else if (arch == LLM_ARCH_INKLING) {
ms.add_kv(
LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN,
std::vector<uint32_t>({1, 0}));
} else {
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, uint32_t(2));
}
@@ -217,12 +226,27 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
if (moe) {
ms.add_kv(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, n_ff);
ms.add_kv(LLM_KV_INTERLEAVE_MOE_LAYER_STEP, uint32_t(2));
ms.add_kv(LLM_KV_EXPERT_COUNT, uint32_t(2));
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, uint32_t(1));
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, uint32_t(1));
ms.add_kv(LLM_KV_EXPERT_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(8) : uint32_t(2));
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(6) : uint32_t(1));
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(2) : uint32_t(1));
ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, uint32_t(2)); // sigmoid
ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f);
ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1));
if (arch == LLM_ARCH_INKLING) {
ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE, 1.0f);
}
}
if (arch == LLM_ARCH_INKLING) {
ms.add_kv(LLM_KV_INKLING_D_REL, uint32_t(8));
ms.add_kv(LLM_KV_INKLING_REL_EXTENT, uint32_t(32));
ms.add_kv(LLM_KV_INKLING_REL_EXTENT_SWA, uint32_t(16));
ms.add_kv(LLM_KV_INKLING_SHORTCONV_KERNEL, uint32_t(4));
ms.add_kv(LLM_KV_INKLING_DENSE_BLOCK_COUNT, uint32_t(1));
ms.add_kv(LLM_KV_INKLING_LOGIT_SCALE_DENOM, 1.0f);
ms.add_kv(LLM_KV_INKLING_LOG_SCALING_N_FLOOR, uint32_t(16));
ms.add_kv(LLM_KV_INKLING_LOG_SCALING_ALPHA, 0.1f);
ms.add_kv(LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, n_vocab);
}
ms.add_kv(LLM_KV_POSNET_EMBEDDING_LENGTH, n_embd);
@@ -372,6 +396,7 @@ static bool moe_mandatory(const llm_arch arch) {
case LLM_ARCH_MISTRAL4:
case LLM_ARCH_MELLUM:
case LLM_ARCH_LAGUNA:
case LLM_ARCH_INKLING:
return true;
default:
return false;
@@ -478,7 +503,9 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
continue;
}
const bool can_save_fixture =
llama_model_saver_supports_arch(arch) || arch == LLM_ARCH_LAGUNA;
llama_model_saver_supports_arch(arch) ||
arch == LLM_ARCH_LAGUNA ||
arch == LLM_ARCH_INKLING;
if (!can_save_fixture || !arch_supported(arch)) {
LOG_INF("%s: %s model (%s) is unsupported, skipping\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense");
continue;
@@ -487,15 +514,21 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
auto model_and_ctx = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {});
const std::string path = dir + "/" + llm_arch_name(arch) + (moe ? "-moe.gguf" : "-dense.gguf");
LOG_INF("%s: Saving %s model (%s) to %s...\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense", path.c_str());
if (llama_model_saver_supports_arch(arch)) {
const bool private_fixture =
arch == LLM_ARCH_LAGUNA || arch == LLM_ARCH_INKLING;
if (llama_model_saver_supports_arch(arch) && !private_fixture) {
llama_model_save_to_file(model_and_ctx.first.get(), path.c_str());
} else {
// Laguna is not supported by the production model saver yet.
// Laguna and Inkling are not supported by the production
// model saver yet.
// Preserve the exact synthetic fixture metadata and attach the
// initialized model tensors so longhaul can be tested from a
// real file without broadening the saver API.
gguf_context_ptr fixture(gguf_init_empty());
gguf_set_kv(fixture.get(), gguf_ctx.get());
// Model initialization normalizes the caller's GGUF context,
// so regenerate the exact private metadata for the file.
gguf_context_ptr fixture_metadata = get_gguf_ctx(arch, moe);
gguf_set_kv(fixture.get(), fixture_metadata.get());
for (const auto & item : llama_internal_get_tensor_map(model_and_ctx.first.get())) {
gguf_add_tensor(fixture.get(), item.second);
}
+20 -11
View File
@@ -191,12 +191,13 @@ struct cpu_decode_result {
bool expert_buffer_is_host = false;
};
static cpu_decode_result decode_cpu_model(
static cpu_decode_result decode_model(
const std::string & path,
llama_load_mode load_mode,
uint64_t cache_bytes) {
uint64_t cache_bytes,
int32_t n_gpu_layers) {
llama_model_params model_params = llama_model_default_params();
model_params.n_gpu_layers = 0;
model_params.n_gpu_layers = n_gpu_layers;
model_params.load_mode = load_mode;
model_params.longhaul_cache_bytes = cache_bytes;
model_params.use_extra_bufts = true;
@@ -256,19 +257,24 @@ static cpu_decode_result decode_cpu_model(
return result;
}
static void test_cpu_model(
static void test_model(
testing & t,
const std::string & path,
uint64_t cache_bytes) {
uint64_t cache_bytes,
int32_t n_gpu_layers) {
const cpu_decode_result regular =
decode_cpu_model(path, LLAMA_LOAD_MODE_NONE, 0);
decode_model(path, LLAMA_LOAD_MODE_NONE, 0, n_gpu_layers);
const cpu_decode_result longhaul =
decode_cpu_model(path, LLAMA_LOAD_MODE_LONGHAUL, cache_bytes);
decode_model(path, LLAMA_LOAD_MODE_LONGHAUL, cache_bytes, n_gpu_layers);
t.assert_equal(1u, longhaul.capacity);
t.assert_true("longhaul loaded at least one expert", longhaul.misses > 0);
t.assert_true("longhaul read expert bytes", longhaul.bytes_read > 0);
t.assert_true("streamed experts use a directly writable host buffer", longhaul.expert_buffer_is_host);
if (n_gpu_layers == 0) {
t.assert_true(
"streamed CPU experts use a directly writable host buffer",
longhaul.expert_buffer_is_host);
}
t.assert_equal(regular.logits.size(), longhaul.logits.size());
double squared_error = 0.0;
@@ -296,11 +302,14 @@ int main(int argc, char ** argv) {
llama_log_set([](ggml_log_level, const char *, void *) {}, nullptr);
}
if (argc == 4 && std::string(argv[1]) == "--cpu-model") {
if (argc == 4 &&
(std::string(argv[1]) == "--cpu-model" ||
std::string(argv[1]) == "--metal-model")) {
const std::string path = argv[2];
const uint64_t cache_bytes = std::stoull(argv[3]);
t.test("cpu_model", [&](testing & current) {
test_cpu_model(current, path, cache_bytes);
const bool metal = std::string(argv[1]) == "--metal-model";
t.test(metal ? "metal_model" : "cpu_model", [&](testing & current) {
test_model(current, path, cache_bytes, metal ? 99 : 0);
});
const int result = t.summary();
llama_backend_free();
+1 -1
View File
@@ -61,7 +61,7 @@
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
| `--longhaul` | stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
+1 -1
View File
@@ -144,7 +144,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
| `--longhaul` | stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
+1 -1
View File
@@ -78,7 +78,7 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
| `--longhaul` | stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |