Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
22f42628dd |
+1
-1
@@ -2631,7 +2631,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
).set_env("LLAMA_ARG_LOAD_MODE"));
|
||||
add_opt(common_arg(
|
||||
{"--longhaul"},
|
||||
"stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache",
|
||||
"stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache",
|
||||
[](common_params & params) {
|
||||
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
|
||||
}
|
||||
|
||||
+119
@@ -2530,6 +2530,118 @@ static common_chat_params common_chat_params_init_minimax_m3(const common_chat_t
|
||||
return data;
|
||||
}
|
||||
|
||||
// Inkling's TML format can emit multiple typed content blocks in one turn.
|
||||
// <|end_message|> closes a block while content_model_end_sampling closes the
|
||||
// generation, so the generic single-block parser is not sufficient.
|
||||
static common_chat_params common_chat_params_init_inkling(
|
||||
const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs) {
|
||||
common_chat_params data;
|
||||
|
||||
const std::string MSG_MODEL = "<|message_model|>";
|
||||
const std::string MSG_USER = "<|message_user|>";
|
||||
const std::string MSG_SYSTEM = "<|message_system|>";
|
||||
const std::string MSG_TOOL = "<|message_tool|>";
|
||||
const std::string THINK = "<|content_thinking|>";
|
||||
const std::string TEXT = "<|content_text|>";
|
||||
const std::string END_MESSAGE = "<|end_message|>";
|
||||
const std::string END_SAMPLING = "<|content_model_end_sampling|>";
|
||||
const std::string INVOKE_TOOL = "<|content_invoke_tool_json|>";
|
||||
|
||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);
|
||||
data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs);
|
||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
||||
data.supports_thinking = true;
|
||||
data.thinking_start_tag = THINK;
|
||||
data.thinking_end_tags = {END_MESSAGE};
|
||||
data.preserved_tokens = {
|
||||
MSG_MODEL, MSG_USER, MSG_SYSTEM, MSG_TOOL,
|
||||
THINK, TEXT, END_MESSAGE, END_SAMPLING, INVOKE_TOOL,
|
||||
};
|
||||
data.message_delimiters = {
|
||||
{ COMMON_CHAT_ROLE_ASSISTANT, MSG_MODEL },
|
||||
{ COMMON_CHAT_ROLE_USER, MSG_USER },
|
||||
{ COMMON_CHAT_ROLE_SYSTEM, MSG_SYSTEM },
|
||||
{ COMMON_CHAT_ROLE_TOOL, MSG_TOOL },
|
||||
};
|
||||
|
||||
const bool has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
const bool extract_reasoning =
|
||||
inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
|
||||
if (inputs.has_continuation()) {
|
||||
const auto & msg = inputs.continue_msg;
|
||||
data.generation_prompt = MSG_MODEL + THINK + msg.reasoning_content;
|
||||
if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
|
||||
data.generation_prompt += END_MESSAGE + TEXT + msg.render_content();
|
||||
}
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto generation_prompt = p.literal(MSG_MODEL);
|
||||
auto end = p.end();
|
||||
|
||||
common_peg_parser reasoning_block = p.eps();
|
||||
if (extract_reasoning) {
|
||||
reasoning_block =
|
||||
p.literal(THINK) +
|
||||
p.reasoning(p.until_one_of({ END_MESSAGE, TEXT, END_SAMPLING })) +
|
||||
p.optional(p.literal(END_MESSAGE));
|
||||
} else {
|
||||
reasoning_block = p.content(
|
||||
p.literal(THINK) +
|
||||
p.until_one_of({ END_MESSAGE, TEXT, END_SAMPLING }) +
|
||||
p.optional(p.literal(END_MESSAGE)));
|
||||
}
|
||||
auto reasoning = p.optional(reasoning_block);
|
||||
|
||||
auto text_block =
|
||||
p.optional(p.literal(MSG_MODEL)) +
|
||||
p.optional(p.literal(TEXT)) +
|
||||
p.content(p.until_one_of({ THINK, END_MESSAGE, END_SAMPLING })) +
|
||||
p.optional(p.literal(END_MESSAGE));
|
||||
auto text_content =
|
||||
p.one_or_more(p.choice({ reasoning_block, text_block }));
|
||||
|
||||
if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
return generation_prompt + reasoning + text_content +
|
||||
p.optional(p.literal(END_SAMPLING)) + end;
|
||||
}
|
||||
|
||||
auto tool_section = p.standard_json_tools(
|
||||
INVOKE_TOOL, END_MESSAGE, inputs.tools,
|
||||
/* parallel_tool_calls = */ false,
|
||||
/* force_tool_calls = */ true,
|
||||
/* name_key = */ "name",
|
||||
/* args_key = */ "args",
|
||||
/* array_wrapped = */ false,
|
||||
/* function_is_key = */ false,
|
||||
/* call_id_key = */ "",
|
||||
/* gen_call_id_key = */ "",
|
||||
/* parameters_order = */ {},
|
||||
/* accept_openai_wrapper = */ false);
|
||||
auto tool_block =
|
||||
p.optional(p.literal(MSG_MODEL)) +
|
||||
p.until_one_of({ INVOKE_TOOL, TEXT, THINK, END_MESSAGE, END_SAMPLING }) +
|
||||
tool_section;
|
||||
auto tool_calls = inputs.parallel_tool_calls
|
||||
? p.one_or_more(tool_block)
|
||||
: tool_block;
|
||||
auto mixed_body =
|
||||
p.one_or_more(p.choice({ tool_block, reasoning_block, text_block }));
|
||||
auto body = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED
|
||||
? tool_calls
|
||||
: mixed_body;
|
||||
|
||||
return generation_prompt + reasoning + body +
|
||||
p.optional(p.literal(END_SAMPLING)) + end;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
return data;
|
||||
}
|
||||
|
||||
namespace workaround {
|
||||
|
||||
static void map_developer_role_to_system(json & messages) {
|
||||
@@ -2947,6 +3059,13 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
return common_chat_params_init_cohere2moe(tmpl, params);
|
||||
}
|
||||
|
||||
if (src.find("<|content_thinking|>") != std::string::npos &&
|
||||
src.find("<|content_text|>") != std::string::npos &&
|
||||
src.find("<|message_model|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Inkling\n");
|
||||
return common_chat_params_init_inkling(tmpl, params);
|
||||
}
|
||||
|
||||
if (is_lfm2_template(src)) {
|
||||
LOG_DBG("Using specialized template: LFM2\n");
|
||||
return common_chat_params_init_lfm2(tmpl, params, /* tool_list_tokens = */ true);
|
||||
|
||||
+16
-1
@@ -31,7 +31,7 @@ The normal startup warmup is skipped automatically in longhaul mode. Routed expe
|
||||
Longhaul currently requires:
|
||||
|
||||
- all repeating layers on CPU, or all repeating layers on Metal
|
||||
- Qwen3.5 MoE or Laguna architecture
|
||||
- Qwen3.5 MoE, Laguna, or Inkling architecture
|
||||
- text generation without embeddings or LoRA adapters
|
||||
|
||||
CPU mode is available wherever the CPU backend is supported. Metal mode requires
|
||||
@@ -63,6 +63,21 @@ once, and independent expert slices are read concurrently where the platform
|
||||
supports positional reads. CPU and shared Metal buffers are populated directly;
|
||||
private Metal buffers use a staged fallback.
|
||||
|
||||
Inkling keeps its two shared experts resident and streams only the routed
|
||||
256-expert banks. Inkling-Small selects six routed experts per token. For the
|
||||
current three-shard UD-IQ1_S model, a 2 GiB cache provides seven routed-expert
|
||||
slots and therefore executes each MoE layer in one stage:
|
||||
|
||||
```sh
|
||||
llama-cli \
|
||||
--model /path/to/Inkling-Small-UD-IQ1_S-00001-of-00003.gguf \
|
||||
--longhaul \
|
||||
--longhaul-cache 2
|
||||
```
|
||||
|
||||
Add `--n-gpu-layers 0` to run entirely on CPU; on macOS the default all-Metal
|
||||
placement is supported.
|
||||
|
||||
## Benchmarking prompt processing
|
||||
|
||||
`llama-bench` accepts the longhaul load mode and cache budget:
|
||||
|
||||
@@ -144,6 +144,7 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
|
||||
{ LLM_ARCH_TALKIE, "talkie" },
|
||||
{ LLM_ARCH_MELLUM, "mellum" },
|
||||
{ LLM_ARCH_NANBEIGE, "nanbeige" },
|
||||
{ LLM_ARCH_INKLING, "inkling" },
|
||||
{ LLM_ARCH_UNKNOWN, "(unknown)" },
|
||||
};
|
||||
|
||||
@@ -320,6 +321,15 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
|
||||
{ LLM_KV_NORM_BEFORE_FC, "%s.norm_before_fc" },
|
||||
|
||||
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
|
||||
{ LLM_KV_INKLING_D_REL, "%s.d_rel" },
|
||||
{ LLM_KV_INKLING_REL_EXTENT, "%s.rel_extent" },
|
||||
{ LLM_KV_INKLING_REL_EXTENT_SWA, "%s.rel_extent_swa" },
|
||||
{ LLM_KV_INKLING_SHORTCONV_KERNEL, "%s.shortconv_kernel" },
|
||||
{ LLM_KV_INKLING_DENSE_BLOCK_COUNT, "%s.dense_block_count" },
|
||||
{ LLM_KV_INKLING_LOGIT_SCALE_DENOM, "%s.logit_scale_denom" },
|
||||
{ LLM_KV_INKLING_LOG_SCALING_N_FLOOR, "%s.log_scaling_n_floor" },
|
||||
{ LLM_KV_INKLING_LOG_SCALING_ALPHA, "%s.log_scaling_alpha" },
|
||||
{ LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, "%s.unpadded_vocab_size" },
|
||||
// sentence-transformers dense modules feature dims
|
||||
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
|
||||
{ LLM_KV_DENSE_2_FEAT_OUT, "%s.dense_2_feat_out" },
|
||||
@@ -592,9 +602,19 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
|
||||
{ LLM_TENSOR_SHORTCONV_CONV, "blk.%d.shortconv.conv" },
|
||||
{ LLM_TENSOR_SHORTCONV_INPROJ, "blk.%d.shortconv.in_proj" },
|
||||
{ LLM_TENSOR_SHORTCONV_OUTPROJ, "blk.%d.shortconv.out_proj" },
|
||||
{ LLM_TENSOR_ATTN_R, "blk.%d.attn_r" },
|
||||
{ LLM_TENSOR_ATTN_REL_PROJ, "blk.%d.attn_rel_proj" },
|
||||
{ LLM_TENSOR_SHORTCONV_K, "blk.%d.shortconv_k" },
|
||||
{ LLM_TENSOR_SHORTCONV_V, "blk.%d.shortconv_v" },
|
||||
{ LLM_TENSOR_SHORTCONV_ATTN, "blk.%d.shortconv_attn" },
|
||||
{ LLM_TENSOR_SHORTCONV_MLP, "blk.%d.shortconv_mlp" },
|
||||
{ LLM_TENSOR_FFN_GSCALE, "blk.%d.ffn_gscale" },
|
||||
{ LLM_TENSOR_FFN_GATE_CHEXPS, "blk.%d.ffn_gate_chexps" },
|
||||
{ LLM_TENSOR_FFN_DOWN_CHEXPS, "blk.%d.ffn_down_chexps" },
|
||||
{ LLM_TENSOR_FFN_UP_CHEXPS, "blk.%d.ffn_up_chexps" },
|
||||
{ LLM_TENSOR_FFN_GATE_SHEXPS, "blk.%d.ffn_gate_shexp" },
|
||||
{ LLM_TENSOR_FFN_DOWN_SHEXPS, "blk.%d.ffn_down_shexp" },
|
||||
{ LLM_TENSOR_FFN_UP_SHEXPS, "blk.%d.ffn_up_shexp" },
|
||||
{ LLM_TENSOR_VISEXP_ATTN_QKV, "blk.%d.vis_attn_qkv" },
|
||||
{ LLM_TENSOR_VISEXP_ATTN_OUT, "blk.%d.vis_attn_output" },
|
||||
{ LLM_TENSOR_VISEXP_FFN_GATE, "blk.%d.vis_gate" },
|
||||
@@ -795,6 +815,9 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
||||
{LLM_TENSOR_FFN_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_GATE_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_DOWN_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_DOWN_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_GATE_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_UP_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_GATE_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_UP_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||
{LLM_TENSOR_FFN_EXP_PROBS_B, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ADD}},
|
||||
@@ -836,6 +859,13 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
||||
{LLM_TENSOR_SHORTCONV_CONV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
||||
{LLM_TENSOR_SHORTCONV_INPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_SHORTCONV_OUTPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_ATTN_R, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_ATTN_REL_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_SHORTCONV_K, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
||||
{LLM_TENSOR_SHORTCONV_V, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
||||
{LLM_TENSOR_SHORTCONV_ATTN, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
||||
{LLM_TENSOR_SHORTCONV_MLP, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
||||
{LLM_TENSOR_FFN_GSCALE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
||||
{LLM_TENSOR_VISEXP_ATTN_QKV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_VISEXP_ATTN_OUT, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
{LLM_TENSOR_VISEXP_FFN_GATE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||
@@ -968,6 +998,7 @@ bool llm_arch_is_hybrid(const llm_arch & arch) {
|
||||
case LLM_ARCH_KIMI_LINEAR:
|
||||
case LLM_ARCH_QWEN35:
|
||||
case LLM_ARCH_QWEN35MOE:
|
||||
case LLM_ARCH_INKLING:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
@@ -1024,6 +1055,7 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
|
||||
case LLM_ARCH_MINIMAX_M3:
|
||||
case LLM_ARCH_MISTRAL4:
|
||||
case LLM_ARCH_KIMI_LINEAR:
|
||||
case LLM_ARCH_INKLING:
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
|
||||
@@ -149,6 +149,7 @@ enum llm_arch {
|
||||
LLM_ARCH_MINIMAX_M3,
|
||||
LLM_ARCH_DFLASH,
|
||||
LLM_ARCH_NANBEIGE,
|
||||
LLM_ARCH_INKLING,
|
||||
LLM_ARCH_UNKNOWN,
|
||||
};
|
||||
|
||||
@@ -366,6 +367,15 @@ enum llm_kv {
|
||||
LLM_KV_NORM_BEFORE_FC,
|
||||
|
||||
LLM_KV_SHORTCONV_L_CACHE,
|
||||
LLM_KV_INKLING_D_REL,
|
||||
LLM_KV_INKLING_REL_EXTENT,
|
||||
LLM_KV_INKLING_REL_EXTENT_SWA,
|
||||
LLM_KV_INKLING_SHORTCONV_KERNEL,
|
||||
LLM_KV_INKLING_DENSE_BLOCK_COUNT,
|
||||
LLM_KV_INKLING_LOGIT_SCALE_DENOM,
|
||||
LLM_KV_INKLING_LOG_SCALING_N_FLOOR,
|
||||
LLM_KV_INKLING_LOG_SCALING_ALPHA,
|
||||
LLM_KV_INKLING_UNPADDED_VOCAB_SIZE,
|
||||
|
||||
LLM_KV_XIELU_ALPHA_N,
|
||||
LLM_KV_XIELU_ALPHA_P,
|
||||
@@ -434,6 +444,9 @@ enum llm_tensor {
|
||||
LLM_TENSOR_FFN_DOWN_CHEXPS,
|
||||
LLM_TENSOR_FFN_GATE_CHEXPS,
|
||||
LLM_TENSOR_FFN_UP_CHEXPS,
|
||||
LLM_TENSOR_FFN_DOWN_SHEXPS,
|
||||
LLM_TENSOR_FFN_GATE_SHEXPS,
|
||||
LLM_TENSOR_FFN_UP_SHEXPS,
|
||||
LLM_TENSOR_FFN_EXP_PROBS_B,
|
||||
LLM_TENSOR_FFN_LATENT_DOWN,
|
||||
LLM_TENSOR_FFN_LATENT_UP,
|
||||
@@ -595,6 +608,13 @@ enum llm_tensor {
|
||||
LLM_TENSOR_SHORTCONV_CONV,
|
||||
LLM_TENSOR_SHORTCONV_INPROJ,
|
||||
LLM_TENSOR_SHORTCONV_OUTPROJ,
|
||||
LLM_TENSOR_ATTN_R,
|
||||
LLM_TENSOR_ATTN_REL_PROJ,
|
||||
LLM_TENSOR_SHORTCONV_K,
|
||||
LLM_TENSOR_SHORTCONV_V,
|
||||
LLM_TENSOR_SHORTCONV_ATTN,
|
||||
LLM_TENSOR_SHORTCONV_MLP,
|
||||
LLM_TENSOR_FFN_GSCALE,
|
||||
LLM_TENSOR_VISEXP_ATTN_QKV,
|
||||
LLM_TENSOR_VISEXP_ATTN_OUT,
|
||||
LLM_TENSOR_VISEXP_FFN_GATE,
|
||||
|
||||
@@ -1765,6 +1765,119 @@ ggml_tensor * llm_graph_context::build_ffn(
|
||||
return cur;
|
||||
}
|
||||
|
||||
ggml_tensor * llm_graph_context::build_moe_experts(
|
||||
ggml_tensor * cur,
|
||||
ggml_tensor * selected_experts,
|
||||
ggml_tensor * weights,
|
||||
ggml_tensor * up_exps,
|
||||
ggml_tensor * gate_exps,
|
||||
ggml_tensor * down_exps,
|
||||
int64_t n_expert_used,
|
||||
llm_ffn_op_type type_op,
|
||||
int il) const {
|
||||
GGML_ASSERT(cur != nullptr);
|
||||
GGML_ASSERT(selected_experts != nullptr);
|
||||
GGML_ASSERT(weights != nullptr);
|
||||
GGML_ASSERT(up_exps != nullptr && down_exps != nullptr);
|
||||
|
||||
const int64_t n_embd = cur->ne[0];
|
||||
const int64_t n_tokens = cur->ne[1];
|
||||
|
||||
// Ensure routing and weights are available before a longhaul cache
|
||||
// callback resolves and loads the first expert chunk.
|
||||
ggml_build_forward_expand(gf, weights);
|
||||
|
||||
ggml_tensor * moe_inp = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
|
||||
const int64_t chunk_size = longhaul
|
||||
? std::min<int64_t>(n_expert_used, longhaul->capacity())
|
||||
: n_expert_used;
|
||||
GGML_ASSERT(chunk_size > 0);
|
||||
|
||||
ggml_tensor * moe_out = nullptr;
|
||||
int chunk = 0;
|
||||
|
||||
for (int64_t first = 0; first < n_expert_used; first += chunk_size, ++chunk) {
|
||||
const int64_t count = std::min<int64_t>(chunk_size, n_expert_used - first);
|
||||
|
||||
ggml_tensor * selected_chunk = selected_experts;
|
||||
ggml_tensor * weights_chunk = weights;
|
||||
if (count != n_expert_used) {
|
||||
selected_chunk = ggml_view_2d(
|
||||
ctx0, selected_experts, count, n_tokens,
|
||||
selected_experts->nb[1], first*selected_experts->nb[0]);
|
||||
weights_chunk = ggml_view_3d(
|
||||
ctx0, weights, 1, count, n_tokens,
|
||||
weights->nb[1], weights->nb[2], first*weights->nb[1]);
|
||||
}
|
||||
|
||||
ggml_tensor * selected_moe = selected_chunk;
|
||||
if (longhaul) {
|
||||
selected_moe = ggml_dup(ctx0, selected_chunk);
|
||||
ggml_format_name(selected_moe, "longhaul.remap.%d.%d", il, chunk);
|
||||
ggml_build_forward_expand(gf, selected_moe);
|
||||
}
|
||||
|
||||
ggml_tensor * up = build_lora_mm_id(up_exps, moe_inp, selected_moe);
|
||||
ggml_tensor * act = gate_exps
|
||||
? build_lora_mm_id(gate_exps, moe_inp, selected_moe)
|
||||
: up;
|
||||
|
||||
switch (type_op) {
|
||||
case LLM_FFN_SILU:
|
||||
act = gate_exps
|
||||
? ggml_swiglu_split(ctx0, act, up)
|
||||
: ggml_silu(ctx0, act);
|
||||
break;
|
||||
case LLM_FFN_GELU:
|
||||
act = gate_exps
|
||||
? ggml_geglu_split(ctx0, act, up)
|
||||
: ggml_gelu(ctx0, act);
|
||||
break;
|
||||
case LLM_FFN_RELU:
|
||||
act = gate_exps
|
||||
? ggml_reglu_split(ctx0, act, up)
|
||||
: ggml_relu(ctx0, act);
|
||||
break;
|
||||
default:
|
||||
GGML_ABORT("unsupported routed expert activation");
|
||||
}
|
||||
|
||||
ggml_tensor * experts = build_lora_mm_id(
|
||||
down_exps, act, selected_moe);
|
||||
experts = ggml_mul(ctx0, experts, weights_chunk);
|
||||
ggml_build_forward_expand(gf, experts);
|
||||
|
||||
ggml_tensor * chunk_out = nullptr;
|
||||
for (int64_t i = 0; i < count; ++i) {
|
||||
ggml_tensor * expert = ggml_view_2d(
|
||||
ctx0, experts, n_embd, n_tokens,
|
||||
experts->nb[2], i*experts->nb[1]);
|
||||
ggml_build_forward_expand(gf, expert);
|
||||
chunk_out = chunk_out ? ggml_add(ctx0, chunk_out, expert) : expert;
|
||||
ggml_build_forward_expand(gf, chunk_out);
|
||||
}
|
||||
if (count == 1) {
|
||||
chunk_out = ggml_cont(ctx0, chunk_out);
|
||||
}
|
||||
|
||||
ggml_tensor * contribution = chunk_out;
|
||||
if (longhaul) {
|
||||
ggml_tensor * release = ggml_cont(ctx0, chunk_out);
|
||||
ggml_format_name(release, "longhaul.release.%d.%d", il, chunk);
|
||||
ggml_build_forward_expand(gf, release);
|
||||
if (chunk_size < n_expert_used) {
|
||||
contribution = release;
|
||||
}
|
||||
}
|
||||
|
||||
moe_out = moe_out ? ggml_add(ctx0, moe_out, contribution) : contribution;
|
||||
ggml_build_forward_expand(gf, moe_out);
|
||||
}
|
||||
|
||||
cb(moe_out, "ffn_moe_out", il);
|
||||
return moe_out;
|
||||
}
|
||||
|
||||
ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||
ggml_tensor * cur,
|
||||
ggml_tensor * gate_inp,
|
||||
|
||||
@@ -1000,6 +1000,20 @@ struct llm_graph_context {
|
||||
llm_ffn_gate_type type_gate,
|
||||
int il) const;
|
||||
|
||||
// Execute experts for an architecture that supplies its own selected IDs
|
||||
// and weights. In longhaul mode this transparently stages those experts
|
||||
// through the bounded cache.
|
||||
ggml_tensor * build_moe_experts(
|
||||
ggml_tensor * cur,
|
||||
ggml_tensor * selected_experts,
|
||||
ggml_tensor * weights,
|
||||
ggml_tensor * up_exps,
|
||||
ggml_tensor * gate_exps,
|
||||
ggml_tensor * down_exps,
|
||||
int64_t n_expert_used,
|
||||
llm_ffn_op_type type_op,
|
||||
int il) const;
|
||||
|
||||
// build MoE FFN without bias tensors
|
||||
ggml_tensor * build_moe_ffn(
|
||||
ggml_tensor * cur,
|
||||
|
||||
@@ -191,6 +191,10 @@ uint32_t llama_hparams::n_embd_k_idx(uint32_t il) const {
|
||||
}
|
||||
|
||||
uint32_t llama_hparams::n_embd_r() const {
|
||||
if (n_embd_r_impl != 0) {
|
||||
return n_embd_r_impl;
|
||||
}
|
||||
|
||||
if (wkv_head_size != 0) {
|
||||
// for RWKV models
|
||||
return token_shift_count * n_embd;
|
||||
|
||||
@@ -80,6 +80,18 @@ struct llama_hparams {
|
||||
|
||||
uint32_t n_shortconv_l_cache = 0;
|
||||
|
||||
// Explicit recurrent-state size override. Inkling packs four independent
|
||||
// short-convolution histories into the recurrent cache for every layer.
|
||||
uint32_t n_embd_r_impl = 0;
|
||||
|
||||
// Inkling relative-attention and output metadata.
|
||||
uint32_t inkling_d_rel = 0;
|
||||
uint32_t inkling_rel_extent = 0;
|
||||
uint32_t inkling_rel_extent_swa = 0;
|
||||
uint32_t inkling_log_n_floor = 0;
|
||||
float inkling_log_alpha = 0.0f;
|
||||
uint32_t inkling_unpadded_n_vocab = 0;
|
||||
|
||||
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_arr;
|
||||
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_kv_arr;
|
||||
std::array<uint32_t, LLAMA_MAX_LAYERS> n_ff_arr;
|
||||
|
||||
@@ -1931,6 +1931,37 @@ void llama_kv_cache::set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch
|
||||
}
|
||||
}
|
||||
|
||||
void llama_kv_cache::set_input_pos_rel_flat(
|
||||
ggml_tensor * dst,
|
||||
const llama_ubatch * ubatch,
|
||||
uint32_t extent) const {
|
||||
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
|
||||
GGML_ASSERT(dst->ne[1] == (int64_t) ubatch->n_tokens);
|
||||
|
||||
int32_t * data = (int32_t *) dst->data;
|
||||
const int64_t n_kv = dst->ne[0];
|
||||
|
||||
// Each query owns an extent+1-row block in the flattened relative-logit
|
||||
// table. The final row is the zero-bias sentinel for empty/out-of-band KV
|
||||
// cells.
|
||||
for (int64_t i = 0; i < (int64_t) ubatch->n_tokens; ++i) {
|
||||
const llama_seq_id seq_id = ubatch->seq_id[i][0];
|
||||
const auto & cells = v_cells[seq_to_stream[seq_id]];
|
||||
const llama_pos p1 = ubatch->pos[i];
|
||||
|
||||
for (int64_t j = 0; j < n_kv; ++j) {
|
||||
int32_t rel = (int32_t) extent;
|
||||
if (!cells.is_empty(j)) {
|
||||
const llama_pos d = p1 - cells.pos_get(j);
|
||||
if (d >= 0 && d < (llama_pos) extent) {
|
||||
rel = (int32_t) d;
|
||||
}
|
||||
}
|
||||
data[i*n_kv + j] = (int32_t) (i*(extent + 1)) + rel;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void llama_kv_cache::set_input_k_rot(ggml_tensor * dst) const {
|
||||
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
|
||||
|
||||
@@ -2896,6 +2927,13 @@ void llama_kv_cache_context::set_input_pos_bucket(ggml_tensor * dst, const llama
|
||||
kv->set_input_pos_bucket(dst, ubatch);
|
||||
}
|
||||
|
||||
void llama_kv_cache_context::set_input_pos_rel_flat(
|
||||
ggml_tensor * dst,
|
||||
const llama_ubatch * ubatch,
|
||||
uint32_t extent) const {
|
||||
kv->set_input_pos_rel_flat(dst, ubatch, extent);
|
||||
}
|
||||
|
||||
void llama_kv_cache_context::set_input_k_rot(ggml_tensor * dst) const {
|
||||
kv->set_input_k_rot(dst);
|
||||
}
|
||||
|
||||
@@ -215,6 +215,8 @@ public:
|
||||
|
||||
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
|
||||
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
|
||||
void set_input_pos_rel_flat(
|
||||
ggml_tensor * dst, const llama_ubatch * ubatch, uint32_t extent) const;
|
||||
|
||||
void set_input_k_rot(ggml_tensor * dst) const;
|
||||
void set_input_v_rot(ggml_tensor * dst) const;
|
||||
@@ -405,6 +407,8 @@ public:
|
||||
void set_input_k_shift (ggml_tensor * dst) const;
|
||||
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
|
||||
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
|
||||
void set_input_pos_rel_flat(
|
||||
ggml_tensor * dst, const llama_ubatch * ubatch, uint32_t extent) const;
|
||||
|
||||
void set_input_k_rot(ggml_tensor * dst) const;
|
||||
void set_input_v_rot(ggml_tensor * dst) const;
|
||||
|
||||
@@ -29,6 +29,7 @@ bool llama_model_saver_supports_arch(llm_arch arch) {
|
||||
case LLM_ARCH_STEP35:
|
||||
case LLM_ARCH_MELLUM:
|
||||
case LLM_ARCH_LAGUNA:
|
||||
case LLM_ARCH_INKLING:
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
|
||||
+6
-3
@@ -312,6 +312,8 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
|
||||
return new llama_model_kimi_linear(params);
|
||||
case LLM_ARCH_STEP35:
|
||||
return new llama_model_step35(params);
|
||||
case LLM_ARCH_INKLING:
|
||||
return new llama_model_inkling(params);
|
||||
default:
|
||||
throw std::runtime_error(std::string("unsupported model architecture: '") + llm_arch_name(arch) + "'");
|
||||
}
|
||||
@@ -1356,8 +1358,8 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
||||
}
|
||||
|
||||
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
||||
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA) {
|
||||
throw std::runtime_error("longhaul currently requires a qwen35moe or laguna model");
|
||||
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA && arch != LLM_ARCH_INKLING) {
|
||||
throw std::runtime_error("longhaul currently requires a qwen35moe, laguna, or inkling model");
|
||||
}
|
||||
if (hparams.n_layer_nextn != 0) {
|
||||
throw std::runtime_error("longhaul does not support MTP tensors");
|
||||
@@ -2165,7 +2167,7 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,
|
||||
// layer filters, so pick the right one here
|
||||
llama_memory_hybrid::layer_filter_cb filter_attn = nullptr;
|
||||
llama_memory_hybrid::layer_filter_cb filter_recr = nullptr;
|
||||
if (arch == LLM_ARCH_FALCON_H1) {
|
||||
if (arch == LLM_ARCH_FALCON_H1 || arch == LLM_ARCH_INKLING) {
|
||||
filter_attn = [&](uint32_t) { return true; };
|
||||
filter_recr = [&](uint32_t) { return true; };
|
||||
} else if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
|
||||
@@ -2506,6 +2508,7 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
|
||||
case LLM_ARCH_NEMOTRON_H:
|
||||
case LLM_ARCH_NEMOTRON_H_MOE:
|
||||
case LLM_ARCH_KIMI_LINEAR:
|
||||
case LLM_ARCH_INKLING:
|
||||
return LLAMA_ROPE_TYPE_NONE;
|
||||
|
||||
// use what we call a normal RoPE, operating on pairs of consecutive head values
|
||||
|
||||
@@ -534,6 +534,15 @@ struct llama_layer {
|
||||
struct llama_layer_shortconv shortconv;
|
||||
|
||||
struct llama_layer_nextn nextn;
|
||||
|
||||
// Inkling relative attention and short-convolution weights.
|
||||
struct ggml_tensor * wr = nullptr;
|
||||
struct ggml_tensor * attn_rel_proj = nullptr;
|
||||
struct ggml_tensor * shortconv_k = nullptr;
|
||||
struct ggml_tensor * shortconv_v = nullptr;
|
||||
struct ggml_tensor * shortconv_attn = nullptr;
|
||||
struct ggml_tensor * shortconv_mlp = nullptr;
|
||||
struct ggml_tensor * ffn_gscale = nullptr;
|
||||
};
|
||||
|
||||
struct llama_device {
|
||||
|
||||
+17
-1
@@ -434,6 +434,13 @@ struct llm_tokenizer_bpe : llm_tokenizer {
|
||||
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))*((?=[\\p{L}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))+((?=[\\p{L}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
||||
};
|
||||
break;
|
||||
case LLAMA_VOCAB_PRE_TYPE_INKLING:
|
||||
// o200k-family expression with combining marks included in
|
||||
// letter lookaheads, as used by the Inkling tokenizer.
|
||||
regex_exprs = {
|
||||
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}\\p{M}])([^a-z]))*((?=[\\p{L}\\p{M}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}\\p{M}])([^a-z]))+((?=[\\p{L}\\p{M}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
||||
};
|
||||
break;
|
||||
case LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI:
|
||||
// Same lookaheads as GPT4O but with \p{M} added so combining marks
|
||||
// (diacritics) attach to their base letters. Avoids excessive
|
||||
@@ -2324,6 +2331,10 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
||||
tokenizer_pre == "talkie") {
|
||||
pre_type = LLAMA_VOCAB_PRE_TYPE_GPT4O;
|
||||
clean_spaces = false;
|
||||
} else if (
|
||||
tokenizer_pre == "inkling") {
|
||||
pre_type = LLAMA_VOCAB_PRE_TYPE_INKLING;
|
||||
clean_spaces = false;
|
||||
} else if (
|
||||
tokenizer_pre == "granite-embed-multi-97m") {
|
||||
pre_type = LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI;
|
||||
@@ -2607,7 +2618,12 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
||||
if (suppress_idx != -1) {
|
||||
const int n = gguf_get_arr_n(ctx, suppress_idx);
|
||||
const int32_t * data = (const int32_t *) gguf_get_arr_data(ctx, suppress_idx);
|
||||
suppress_tokens.assign(data, data + n);
|
||||
suppress_tokens.reserve(n);
|
||||
for (int i = 0; i < n; ++i) {
|
||||
if (data[i] >= 0 && data[i] < (int32_t) id_to_token.size()) {
|
||||
suppress_tokens.push_back(data[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -65,6 +65,7 @@ enum llama_vocab_pre_type {
|
||||
LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI = 54,
|
||||
LLAMA_VOCAB_PRE_TYPE_MELLUM2 = 55,
|
||||
LLAMA_VOCAB_PRE_TYPE_LAGUNA = 56,
|
||||
LLAMA_VOCAB_PRE_TYPE_INKLING = 57,
|
||||
};
|
||||
|
||||
struct LLM_KV;
|
||||
|
||||
@@ -0,0 +1,489 @@
|
||||
// Inkling: relative-position attention with per-layer packed short-convolution
|
||||
// state and a mixture of routed and shared experts.
|
||||
|
||||
#include "models.h"
|
||||
|
||||
#include "../llama-memory-hybrid-iswa.h"
|
||||
#include "../llama-memory-recurrent.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
void llama_model_inkling::load_arch_hparams(llama_model_loader & ml) {
|
||||
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
|
||||
|
||||
ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp);
|
||||
ml.get_key(LLM_KV_EXPERT_SHARED_COUNT, hparams.n_expert_shared);
|
||||
ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE, hparams.expert_weights_scale);
|
||||
ml.get_key(LLM_KV_EXPERT_GATING_FUNC, hparams.expert_gating_func, false);
|
||||
|
||||
ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa);
|
||||
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
|
||||
ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, hparams.is_swa_impl, hparams.n_layer());
|
||||
|
||||
for (uint32_t il = 0; il < hparams.n_layer(); ++il) {
|
||||
hparams.is_recr_impl[il] = 1;
|
||||
}
|
||||
|
||||
ml.get_key(LLM_KV_INKLING_D_REL, hparams.inkling_d_rel);
|
||||
ml.get_key(LLM_KV_INKLING_REL_EXTENT, hparams.inkling_rel_extent);
|
||||
ml.get_key(LLM_KV_INKLING_REL_EXTENT_SWA, hparams.inkling_rel_extent_swa);
|
||||
ml.get_key(LLM_KV_INKLING_SHORTCONV_KERNEL, hparams.n_shortconv_l_cache);
|
||||
ml.get_key(LLM_KV_INKLING_DENSE_BLOCK_COUNT, hparams.n_layer_dense_lead);
|
||||
|
||||
float logit_scale_denom = 0.0f;
|
||||
ml.get_key(LLM_KV_INKLING_LOGIT_SCALE_DENOM, logit_scale_denom);
|
||||
GGML_ASSERT(logit_scale_denom != 0.0f);
|
||||
hparams.f_logit_scale = 1.0f/logit_scale_denom;
|
||||
|
||||
ml.get_key(LLM_KV_INKLING_LOG_SCALING_N_FLOOR, hparams.inkling_log_n_floor, false);
|
||||
ml.get_key(LLM_KV_INKLING_LOG_SCALING_ALPHA, hparams.inkling_log_alpha, false);
|
||||
ml.get_key(LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, hparams.inkling_unpadded_n_vocab, false);
|
||||
|
||||
GGML_ASSERT(hparams.n_shortconv_l_cache > 1);
|
||||
GGML_ASSERT(hparams.inkling_d_rel > 0);
|
||||
GGML_ASSERT(hparams.inkling_rel_extent > 0 && hparams.inkling_rel_extent_swa > 0);
|
||||
|
||||
// Four streams per layer: K, V, attention output, and MLP output.
|
||||
const uint32_t d_conv = hparams.n_shortconv_l_cache - 1;
|
||||
hparams.n_embd_r_impl = d_conv *
|
||||
(hparams.n_embd_k_gqa_max() + hparams.n_embd_v_gqa_max() + 2*hparams.n_embd);
|
||||
|
||||
type = LLM_TYPE_UNKNOWN;
|
||||
}
|
||||
|
||||
void llama_model_inkling::load_arch_tensors(llama_model_loader &) {
|
||||
LLAMA_LOAD_LOCALS;
|
||||
|
||||
const int64_t head_dim = hparams.n_embd_head_k();
|
||||
const int64_t d_rel = hparams.inkling_d_rel;
|
||||
const int64_t K = hparams.n_shortconv_l_cache;
|
||||
const int64_t n_ff_exp = hparams.n_ff_exp;
|
||||
const int64_t n_shexp = hparams.n_expert_shared;
|
||||
const int expert_flags = params.load_mode == LLAMA_LOAD_MODE_LONGHAUL ? TENSOR_LONGHAUL : 0;
|
||||
|
||||
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
|
||||
tok_norm = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD_NORM, "weight", 0), {n_embd}, 0);
|
||||
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
|
||||
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, 0);
|
||||
|
||||
for (int i = 0; i < n_layer; ++i) {
|
||||
auto & layer = layers[i];
|
||||
|
||||
const int64_t n_head_kv_i = hparams.n_head_kv(i);
|
||||
const int64_t kvw = n_head_kv_i*head_dim;
|
||||
const int64_t rel_extent = hparams.is_swa(i)
|
||||
? hparams.inkling_rel_extent_swa
|
||||
: hparams.inkling_rel_extent;
|
||||
|
||||
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
|
||||
|
||||
layer.wq = create_tensor(tn(LLM_TENSOR_ATTN_Q, "weight", i), {n_embd, n_head*head_dim}, 0);
|
||||
layer.wk = create_tensor(tn(LLM_TENSOR_ATTN_K, "weight", i), {n_embd, kvw}, 0);
|
||||
layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V, "weight", i), {n_embd, kvw}, 0);
|
||||
layer.wr = create_tensor(tn(LLM_TENSOR_ATTN_R, "weight", i), {n_embd, n_head*d_rel}, 0);
|
||||
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_head*head_dim, n_embd}, 0);
|
||||
|
||||
layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {head_dim}, 0);
|
||||
layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {head_dim}, 0);
|
||||
|
||||
layer.attn_rel_proj = create_tensor(
|
||||
tn(LLM_TENSOR_ATTN_REL_PROJ, "weight", i), {rel_extent, d_rel}, 0);
|
||||
|
||||
layer.shortconv_k = create_tensor(tn(LLM_TENSOR_SHORTCONV_K, "weight", i), {K, kvw}, 0);
|
||||
layer.shortconv_v = create_tensor(tn(LLM_TENSOR_SHORTCONV_V, "weight", i), {K, kvw}, 0);
|
||||
layer.shortconv_attn = create_tensor(tn(LLM_TENSOR_SHORTCONV_ATTN, "weight", i), {K, n_embd}, 0);
|
||||
layer.shortconv_mlp = create_tensor(tn(LLM_TENSOR_SHORTCONV_MLP, "weight", i), {K, n_embd}, 0);
|
||||
|
||||
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
|
||||
layer.ffn_gscale = create_tensor(tn(LLM_TENSOR_FFN_GSCALE, "weight", i), {1}, 0);
|
||||
|
||||
if (i < (int) hparams.n_layer_dense_lead) {
|
||||
const int64_t n_ff_i = hparams.n_ff(i);
|
||||
layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff_i}, 0);
|
||||
layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff_i}, 0);
|
||||
layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff_i, n_embd}, 0);
|
||||
} else {
|
||||
GGML_ASSERT(n_expert > 0 && n_expert_used > 0 && n_shexp > 0);
|
||||
|
||||
// The final rows are shared-expert sink logits.
|
||||
layer.ffn_gate_inp = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_GATE_INP, "weight", i), {n_embd, n_expert + n_shexp}, 0);
|
||||
layer.ffn_exp_probs_b = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias", i), {n_expert}, 0);
|
||||
|
||||
layer.ffn_gate_exps = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i),
|
||||
{n_embd, n_ff_exp, n_expert}, expert_flags);
|
||||
layer.ffn_up_exps = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i),
|
||||
{n_embd, n_ff_exp, n_expert}, expert_flags);
|
||||
layer.ffn_down_exps = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i),
|
||||
{n_ff_exp, n_embd, n_expert}, expert_flags);
|
||||
|
||||
// Shared experts stay resident and are never part of the longhaul cache.
|
||||
layer.ffn_gate_shexp = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_GATE_SHEXPS, "weight", i),
|
||||
{n_embd, n_ff_exp, n_shexp}, 0);
|
||||
layer.ffn_up_shexp = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_UP_SHEXPS, "weight", i),
|
||||
{n_embd, n_ff_exp, n_shexp}, 0);
|
||||
layer.ffn_down_shexp = create_tensor(
|
||||
tn(LLM_TENSOR_FFN_DOWN_SHEXPS, "weight", i),
|
||||
{n_ff_exp, n_embd, n_shexp}, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
class llm_graph_input_inkling : public llm_graph_input_i {
|
||||
public:
|
||||
llm_graph_input_inkling(
|
||||
const llama_hparams & hparams,
|
||||
const llama_memory_hybrid_iswa_context * mctx) :
|
||||
hparams(hparams), mctx(mctx) {
|
||||
}
|
||||
|
||||
void set_input(const llama_ubatch * ubatch) override {
|
||||
if (tau) {
|
||||
GGML_ASSERT(ggml_backend_buffer_is_host(tau->buffer));
|
||||
float * data = (float *) tau->data;
|
||||
const float n_floor = (float) hparams.inkling_log_n_floor;
|
||||
const float alpha = hparams.inkling_log_alpha;
|
||||
for (int64_t i = 0; i < (int64_t) ubatch->n_tokens; ++i) {
|
||||
const float eff = (float) (ubatch->pos[i] + 1)/n_floor;
|
||||
data[i] = 1.0f + alpha*logf(std::max(eff, 1.0f));
|
||||
}
|
||||
}
|
||||
|
||||
if (rel_idx) {
|
||||
mctx->get_attn()->get_base()->set_input_pos_rel_flat(
|
||||
rel_idx, ubatch, hparams.inkling_rel_extent);
|
||||
}
|
||||
if (rel_idx_swa) {
|
||||
mctx->get_attn()->get_swa()->set_input_pos_rel_flat(
|
||||
rel_idx_swa, ubatch, hparams.inkling_rel_extent_swa);
|
||||
}
|
||||
|
||||
if (vocab_mask) {
|
||||
GGML_ASSERT(ggml_backend_buffer_is_host(vocab_mask->buffer));
|
||||
float * data = (float *) vocab_mask->data;
|
||||
for (int64_t id = 0; id < vocab_mask->ne[0]; ++id) {
|
||||
data[id] = id < (int64_t) hparams.inkling_unpadded_n_vocab ? 0.0f : -INFINITY;
|
||||
}
|
||||
}
|
||||
|
||||
if (shexp_idx) {
|
||||
GGML_ASSERT(ggml_backend_buffer_is_host(shexp_idx->buffer));
|
||||
int32_t * data = (int32_t *) shexp_idx->data;
|
||||
for (int64_t j = 0; j < shexp_idx->ne[1]; ++j) {
|
||||
for (int64_t s = 0; s < shexp_idx->ne[0]; ++s) {
|
||||
data[j*shexp_idx->ne[0] + s] = (int32_t) s;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
ggml_tensor * tau = nullptr;
|
||||
ggml_tensor * rel_idx = nullptr;
|
||||
ggml_tensor * rel_idx_swa = nullptr;
|
||||
ggml_tensor * vocab_mask = nullptr;
|
||||
ggml_tensor * shexp_idx = nullptr;
|
||||
|
||||
private:
|
||||
const llama_hparams hparams;
|
||||
const llama_memory_hybrid_iswa_context * mctx;
|
||||
};
|
||||
|
||||
std::unique_ptr<llm_graph_context> llama_model_inkling::build_arch_graph(
|
||||
const llm_graph_params & params) const {
|
||||
return std::make_unique<graph>(*this, params);
|
||||
}
|
||||
|
||||
llama_model_inkling::graph::graph(
|
||||
const llama_model & model,
|
||||
const llm_graph_params & params) :
|
||||
llm_graph_context(params) {
|
||||
const int64_t head_dim = hparams.n_embd_head_k();
|
||||
const int64_t d_rel = hparams.inkling_d_rel;
|
||||
const int64_t d_conv = hparams.n_shortconv_l_cache - 1;
|
||||
const int64_t n_embd_r = hparams.n_embd_r();
|
||||
const int64_t kw_max = hparams.n_embd_k_gqa_max();
|
||||
const int64_t vw_max = hparams.n_embd_v_gqa_max();
|
||||
|
||||
const int64_t off_k = 0;
|
||||
const int64_t off_v = d_conv*kw_max;
|
||||
const int64_t off_attn = d_conv*(kw_max + vw_max);
|
||||
const int64_t off_mlp = d_conv*(kw_max + vw_max + n_embd);
|
||||
|
||||
const auto * mctx_hyb = static_cast<const llama_memory_hybrid_iswa_context *>(mctx);
|
||||
const auto * mctx_recr = mctx_hyb->get_recr();
|
||||
const auto * mctx_attn = mctx_hyb->get_attn();
|
||||
const uint32_t kv_head = mctx_recr->get_head();
|
||||
|
||||
const int64_t n_seq_tokens = ubatch.n_seq_tokens;
|
||||
const int64_t n_seqs = ubatch.n_seqs;
|
||||
GGML_ASSERT(n_seqs != 0);
|
||||
GGML_ASSERT(ubatch.equal_seqs());
|
||||
GGML_ASSERT(ubatch.n_tokens == n_seq_tokens*n_seqs);
|
||||
|
||||
bool has_global = false;
|
||||
bool has_local = false;
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
has_local |= hparams.is_swa(il);
|
||||
has_global |= !hparams.is_swa(il);
|
||||
}
|
||||
|
||||
auto inp = std::make_unique<llm_graph_input_inkling>(hparams, mctx_hyb);
|
||||
if (hparams.inkling_log_n_floor > 0 && has_global) {
|
||||
inp->tau = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 1, 1, n_tokens);
|
||||
ggml_set_input(inp->tau);
|
||||
ggml_set_name(inp->tau, "inkling_tau");
|
||||
}
|
||||
if (has_global) {
|
||||
inp->rel_idx = ggml_new_tensor_2d(
|
||||
ctx0, GGML_TYPE_I32, mctx_attn->get_base()->get_n_kv(), n_tokens);
|
||||
ggml_set_input(inp->rel_idx);
|
||||
ggml_set_name(inp->rel_idx, "inkling_rel_idx");
|
||||
}
|
||||
if (has_local) {
|
||||
inp->rel_idx_swa = ggml_new_tensor_2d(
|
||||
ctx0, GGML_TYPE_I32, mctx_attn->get_swa()->get_n_kv(), n_tokens);
|
||||
ggml_set_input(inp->rel_idx_swa);
|
||||
ggml_set_name(inp->rel_idx_swa, "inkling_rel_idx_swa");
|
||||
}
|
||||
|
||||
const int64_t n_vocab = model.vocab.n_tokens();
|
||||
if (!cparams.embeddings &&
|
||||
hparams.inkling_unpadded_n_vocab > 0 &&
|
||||
(int64_t) hparams.inkling_unpadded_n_vocab < n_vocab) {
|
||||
inp->vocab_mask = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_vocab);
|
||||
ggml_set_input(inp->vocab_mask);
|
||||
ggml_set_name(inp->vocab_mask, "inkling_vocab_mask");
|
||||
}
|
||||
if (hparams.n_expert_shared > 0 && (uint32_t) n_layer > hparams.n_layer_dense_lead) {
|
||||
inp->shexp_idx = ggml_new_tensor_2d(
|
||||
ctx0, GGML_TYPE_I32, hparams.n_expert_shared, n_tokens);
|
||||
ggml_set_input(inp->shexp_idx);
|
||||
ggml_set_name(inp->shexp_idx, "inkling_shexp_idx");
|
||||
}
|
||||
|
||||
ggml_tensor * tau = inp->tau;
|
||||
ggml_tensor * rel_idx = inp->rel_idx;
|
||||
ggml_tensor * rel_idx_swa = inp->rel_idx_swa;
|
||||
ggml_tensor * vocab_mask = inp->vocab_mask;
|
||||
ggml_tensor * shexp_idx = inp->shexp_idx;
|
||||
res->add_input(std::move(inp));
|
||||
|
||||
auto * inp_hybrid = build_inp_mem_hybrid_iswa();
|
||||
ggml_tensor * conv_rs_cur = nullptr;
|
||||
|
||||
auto build_sconv = [&](ggml_tensor * x2d, ggml_tensor * kernel, int64_t off, int il) {
|
||||
const int64_t w = x2d->ne[0];
|
||||
ggml_tensor * x3 = ggml_reshape_3d(ctx0, x2d, w, n_seq_tokens, n_seqs);
|
||||
ggml_tensor * xt = ggml_transpose(ctx0, x3);
|
||||
|
||||
ggml_tensor * conv_state = mctx_recr->get_r_l(il);
|
||||
GGML_ASSERT(conv_rs_cur != nullptr);
|
||||
const size_t sz = ggml_element_size(conv_rs_cur);
|
||||
|
||||
ggml_tensor * state = ggml_view_3d(
|
||||
ctx0, conv_rs_cur, d_conv, w, n_seqs,
|
||||
d_conv*sz, conv_rs_cur->nb[1], off*sz);
|
||||
ggml_tensor * sx = ggml_concat(ctx0, state, xt, 0);
|
||||
|
||||
ggml_tensor * new_state = ggml_view_3d(
|
||||
ctx0, sx, d_conv, w, n_seqs,
|
||||
sx->nb[1], sx->nb[2], (sx->ne[0] - d_conv)*sx->nb[0]);
|
||||
ggml_tensor * state_dst = ggml_view_3d(
|
||||
ctx0, conv_state, d_conv, w, n_seqs,
|
||||
d_conv*sz, n_embd_r*sz, (kv_head*n_embd_r + off)*sz);
|
||||
ggml_build_forward_expand(gf, ggml_cpy(ctx0, new_state, state_dst));
|
||||
|
||||
ggml_tensor * conv_out = ggml_ssm_conv(ctx0, sx, kernel);
|
||||
return ggml_reshape_2d(ctx0, ggml_add(ctx0, x3, conv_out), w, n_seq_tokens*n_seqs);
|
||||
};
|
||||
|
||||
auto build_attn_block = [&](ggml_tensor * cur, int il) {
|
||||
const auto & layer = model.layers[il];
|
||||
const bool is_swa = hparams.is_swa(il);
|
||||
const int64_t n_head_kv = hparams.n_head_kv(il);
|
||||
const int64_t rel_extent = is_swa
|
||||
? hparams.inkling_rel_extent_swa
|
||||
: hparams.inkling_rel_extent;
|
||||
|
||||
ggml_tensor * q = build_lora_mm(layer.wq, cur);
|
||||
ggml_tensor * k = build_lora_mm(layer.wk, cur);
|
||||
ggml_tensor * v = build_lora_mm(layer.wv, cur);
|
||||
ggml_tensor * r = build_lora_mm(layer.wr, cur);
|
||||
|
||||
k = build_sconv(k, layer.shortconv_k, off_k, il);
|
||||
v = build_sconv(v, layer.shortconv_v, off_v, il);
|
||||
|
||||
q = ggml_reshape_3d(ctx0, q, head_dim, n_head, n_tokens);
|
||||
k = ggml_reshape_3d(ctx0, k, head_dim, n_head_kv, n_tokens);
|
||||
v = ggml_reshape_3d(ctx0, v, head_dim, n_head_kv, n_tokens);
|
||||
q = build_norm(q, layer.attn_q_norm, nullptr, LLM_NORM_RMS, il);
|
||||
k = build_norm(k, layer.attn_k_norm, nullptr, LLM_NORM_RMS, il);
|
||||
|
||||
if (tau && !is_swa) {
|
||||
q = ggml_mul(ctx0, q, tau);
|
||||
}
|
||||
|
||||
ggml_tensor * r2 = ggml_reshape_2d(ctx0, r, d_rel, n_head*n_tokens);
|
||||
ggml_tensor * proj = ggml_cont(ctx0, ggml_transpose(ctx0, layer.attn_rel_proj));
|
||||
ggml_tensor * rel = ggml_mul_mat(ctx0, proj, r2);
|
||||
ggml_mul_mat_set_prec(rel, GGML_PREC_F32);
|
||||
rel = ggml_reshape_3d(ctx0, rel, rel_extent, n_head, n_tokens);
|
||||
if (tau && !is_swa) {
|
||||
rel = ggml_mul(ctx0, rel, tau);
|
||||
}
|
||||
|
||||
// Scale Q rather than build_attn's joint logits so the relative term
|
||||
// remains unscaled, matching the reference implementation.
|
||||
q = ggml_scale(ctx0, q, 1.0f/(float) head_dim);
|
||||
|
||||
rel = ggml_pad(ctx0, rel, 1, 0, 0, 0);
|
||||
rel = ggml_cont(ctx0, ggml_permute(ctx0, rel, 1, 0, 2, 3));
|
||||
rel = ggml_reshape_2d(ctx0, rel, n_head, (rel_extent + 1)*n_tokens);
|
||||
|
||||
ggml_tensor * idx = is_swa ? rel_idx_swa : rel_idx;
|
||||
GGML_ASSERT(idx != nullptr);
|
||||
const int64_t n_kv = idx->ne[0];
|
||||
ggml_tensor * idx1 = ggml_reshape_1d(ctx0, idx, n_kv*n_tokens);
|
||||
ggml_tensor * kq_b = ggml_get_rows(ctx0, rel, idx1);
|
||||
kq_b = ggml_reshape_3d(ctx0, kq_b, n_head, n_kv, n_tokens);
|
||||
kq_b = ggml_cont(ctx0, ggml_permute(ctx0, kq_b, 2, 0, 1, 3));
|
||||
|
||||
auto * inp_attn = inp_hybrid->get_attn();
|
||||
const int64_t n_stream =
|
||||
(is_swa ? inp_attn->get_kq_mask_swa() : inp_attn->get_kq_mask())->ne[3];
|
||||
if (n_stream > 1) {
|
||||
kq_b = ggml_view_4d(
|
||||
ctx0, kq_b, n_kv, n_tokens/n_stream, n_head, n_stream,
|
||||
kq_b->nb[1], kq_b->nb[2], (n_tokens/n_stream)*kq_b->nb[1], 0);
|
||||
}
|
||||
|
||||
return build_attn(
|
||||
inp_attn, layer.wo, nullptr, nullptr,
|
||||
q, k, v, kq_b, nullptr, nullptr, 1.0f, il);
|
||||
};
|
||||
|
||||
auto build_dense_ffn = [&](ggml_tensor * cur, int il) {
|
||||
cur = build_ffn(
|
||||
cur,
|
||||
model.layers[il].ffn_up, nullptr, nullptr,
|
||||
model.layers[il].ffn_gate, nullptr, nullptr,
|
||||
model.layers[il].ffn_down, nullptr, nullptr,
|
||||
nullptr, LLM_FFN_SILU, LLM_FFN_PAR, il);
|
||||
return ggml_mul(ctx0, cur, model.layers[il].ffn_gscale);
|
||||
};
|
||||
|
||||
auto build_moe = [&](ggml_tensor * cur, int il) {
|
||||
const auto & layer = model.layers[il];
|
||||
const int64_t n_shexp = hparams.n_expert_shared;
|
||||
|
||||
ggml_tensor * logits = build_lora_mm(layer.ffn_gate_inp, cur);
|
||||
ggml_mul_mat_set_prec(logits, GGML_PREC_F32);
|
||||
const size_t lsz = ggml_element_size(logits);
|
||||
ggml_tensor * routed = ggml_cont(ctx0, ggml_view_2d(
|
||||
ctx0, logits, n_expert, n_tokens, logits->nb[1], 0));
|
||||
ggml_tensor * shared_logits = ggml_view_2d(
|
||||
ctx0, logits, n_shexp, n_tokens, logits->nb[1], n_expert*lsz);
|
||||
|
||||
ggml_tensor * scores = ggml_add(
|
||||
ctx0, ggml_sigmoid(ctx0, routed), layer.ffn_exp_probs_b);
|
||||
ggml_tensor * selected = ggml_argsort_top_k(
|
||||
ctx0, scores, n_expert_used);
|
||||
|
||||
ggml_tensor * routed3 = ggml_reshape_3d(ctx0, routed, 1, n_expert, n_tokens);
|
||||
ggml_tensor * topk_logits = ggml_reshape_2d(
|
||||
ctx0, ggml_get_rows(ctx0, routed3, selected), n_expert_used, n_tokens);
|
||||
ggml_tensor * all_logits = ggml_concat(ctx0, topk_logits, shared_logits, 0);
|
||||
|
||||
ggml_tensor * w = ggml_neg(
|
||||
ctx0, ggml_softplus(ctx0, ggml_neg(ctx0, all_logits)));
|
||||
w = ggml_soft_max(ctx0, w);
|
||||
w = ggml_scale(ctx0, w, hparams.expert_weights_scale);
|
||||
w = ggml_mul(ctx0, w, layer.ffn_gscale);
|
||||
|
||||
const size_t wsz = ggml_element_size(w);
|
||||
ggml_tensor * weights = ggml_cont(ctx0, ggml_view_2d(
|
||||
ctx0, w, n_expert_used, n_tokens, w->nb[1], 0));
|
||||
weights = ggml_reshape_3d(ctx0, weights, 1, n_expert_used, n_tokens);
|
||||
|
||||
ggml_tensor * moe_out = build_moe_experts(
|
||||
cur, selected, weights,
|
||||
layer.ffn_up_exps, layer.ffn_gate_exps, layer.ffn_down_exps,
|
||||
n_expert_used, LLM_FFN_SILU, il);
|
||||
|
||||
GGML_ASSERT(shexp_idx != nullptr);
|
||||
ggml_tensor * xr = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
|
||||
ggml_tensor * gs = build_lora_mm_id(layer.ffn_gate_shexp, xr, shexp_idx);
|
||||
ggml_tensor * us = build_lora_mm_id(layer.ffn_up_shexp, xr, shexp_idx);
|
||||
ggml_tensor * hs = ggml_swiglu_split(ctx0, gs, us);
|
||||
|
||||
ggml_tensor * gammas = ggml_cont(ctx0, ggml_view_2d(
|
||||
ctx0, w, n_shexp, n_tokens, w->nb[1], n_expert_used*wsz));
|
||||
hs = ggml_mul(ctx0, hs, ggml_reshape_3d(ctx0, gammas, 1, n_shexp, n_tokens));
|
||||
ggml_tensor * ds = build_lora_mm_id(layer.ffn_down_shexp, hs, shexp_idx);
|
||||
for (int64_t s = 0; s < n_shexp; ++s) {
|
||||
ggml_tensor * e = ggml_view_2d(
|
||||
ctx0, ds, n_embd, n_tokens, ds->nb[2], s*ds->nb[1]);
|
||||
moe_out = ggml_add(ctx0, moe_out, e);
|
||||
}
|
||||
return moe_out;
|
||||
};
|
||||
|
||||
ggml_tensor * cur = build_inp_embd(model.tok_embd);
|
||||
if (ubatch.token) {
|
||||
cur = build_norm(cur, model.tok_norm, nullptr, LLM_NORM_RMS, -1);
|
||||
}
|
||||
ggml_build_forward_expand(gf, cur);
|
||||
|
||||
for (int il = 0; il < n_layer; ++il) {
|
||||
conv_rs_cur = build_rs(
|
||||
inp_hybrid->get_recr(), mctx_recr->get_r_l(il), n_embd_r, n_seqs);
|
||||
|
||||
ggml_tensor * attn_in = build_norm(
|
||||
cur, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il);
|
||||
ggml_tensor * attn_out = build_attn_block(attn_in, il);
|
||||
attn_out = build_sconv(
|
||||
attn_out, model.layers[il].shortconv_attn, off_attn, il);
|
||||
cur = ggml_add(ctx0, cur, attn_out);
|
||||
|
||||
ggml_tensor * ffn_in = build_norm(
|
||||
cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);
|
||||
ggml_tensor * ffn_out = il < (int) hparams.n_layer_dense_lead
|
||||
? build_dense_ffn(ffn_in, il)
|
||||
: build_moe(ffn_in, il);
|
||||
ffn_out = build_sconv(
|
||||
ffn_out, model.layers[il].shortconv_mlp, off_mlp, il);
|
||||
cur = ggml_add(ctx0, cur, ffn_out);
|
||||
|
||||
cur = build_cvec(cur, il);
|
||||
cb(cur, "l_out", il);
|
||||
}
|
||||
|
||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
||||
if (inp_out_ids) {
|
||||
cur = ggml_get_rows(ctx0, cur, inp_out_ids);
|
||||
}
|
||||
|
||||
cur = build_norm(cur, model.output_norm, nullptr, LLM_NORM_RMS, -1);
|
||||
res->t_embd = cur;
|
||||
|
||||
if (!cparams.embeddings) {
|
||||
cur = ggml_scale(ctx0, cur, hparams.f_logit_scale);
|
||||
cur = build_lora_mm(model.output, cur);
|
||||
if (model.output->type == GGML_TYPE_F32) {
|
||||
ggml_mul_mat_set_prec(cur, GGML_PREC_F32);
|
||||
}
|
||||
if (vocab_mask) {
|
||||
cur = ggml_add(ctx0, cur, vocab_mask);
|
||||
}
|
||||
res->t_logits = cur;
|
||||
}
|
||||
|
||||
ggml_build_forward_expand(gf, cur);
|
||||
}
|
||||
@@ -1865,6 +1865,18 @@ struct llama_model_lfm2moe : public llama_model_base {
|
||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
||||
};
|
||||
|
||||
struct llama_model_inkling : public llama_model_base {
|
||||
llama_model_inkling(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
void load_arch_hparams(llama_model_loader & ml) override;
|
||||
void load_arch_tensors(llama_model_loader & ml) override;
|
||||
|
||||
struct graph : public llm_graph_context {
|
||||
graph(const llama_model & model, const llm_graph_params & params);
|
||||
};
|
||||
|
||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
||||
};
|
||||
|
||||
|
||||
struct llama_model_smallthinker : public llama_model_base {
|
||||
llama_model_smallthinker(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||
|
||||
@@ -275,6 +275,24 @@ llama_test(
|
||||
set_tests_properties(test-longhaul-cpu-laguna PROPERTIES
|
||||
FIXTURES_REQUIRED generate-models
|
||||
)
|
||||
llama_test(
|
||||
test-longhaul
|
||||
NAME test-longhaul-cpu-inkling
|
||||
ARGS --cpu-model "${MODEL_DIR}/inkling-moe.gguf" 73728
|
||||
)
|
||||
set_tests_properties(test-longhaul-cpu-inkling PROPERTIES
|
||||
FIXTURES_REQUIRED generate-models
|
||||
)
|
||||
if (APPLE AND GGML_METAL)
|
||||
llama_test(
|
||||
test-longhaul
|
||||
NAME test-longhaul-metal-inkling
|
||||
ARGS --metal-model "${MODEL_DIR}/inkling-moe.gguf" 73728
|
||||
)
|
||||
set_tests_properties(test-longhaul-metal-inkling PROPERTIES
|
||||
FIXTURES_REQUIRED generate-models
|
||||
)
|
||||
endif()
|
||||
llama_build_and_test(test-token-cache.cpp)
|
||||
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
||||
|
||||
|
||||
@@ -91,6 +91,7 @@ static void test_normalize_quotes_with_embedded_quotes(testing & t);
|
||||
static void test_tagged_args_with_embedded_quotes(testing & t);
|
||||
|
||||
static void test_role_markers_all_templates(testing & t);
|
||||
static void test_inkling_tml_parser(testing & t);
|
||||
|
||||
int main(int argc, char * argv[]) {
|
||||
testing t(std::cout);
|
||||
@@ -117,6 +118,7 @@ int main(int argc, char * argv[]) {
|
||||
t.test("standard_json_tools", test_standard_json_tools_formats);
|
||||
t.test("normalize_quotes_to_json", test_normalize_quotes_to_json);
|
||||
t.test("tagged_args_embedded_quotes", test_tagged_args_with_embedded_quotes);
|
||||
t.test("inkling_tml", test_inkling_tml_parser);
|
||||
t.test("role_markers_all_templates", test_role_markers_all_templates);
|
||||
|
||||
return t.summary();
|
||||
@@ -2075,6 +2077,53 @@ static void test_role_markers_all_templates(testing & t) {
|
||||
}
|
||||
}
|
||||
|
||||
static void test_inkling_tml_parser(testing & t) {
|
||||
const std::string tmpl_src =
|
||||
"{# <|content_thinking|><|content_text|> #}"
|
||||
"{%- for message in messages -%}"
|
||||
"<|message_user|><|content_text|>{{ message.content }}<|end_message|>"
|
||||
"{%- endfor -%}"
|
||||
"{%- if add_generation_prompt -%}<|message_model|>{%- endif -%}";
|
||||
|
||||
common_chat_templates_ptr tmpls =
|
||||
common_chat_templates_init(nullptr, tmpl_src);
|
||||
common_chat_templates_inputs inputs;
|
||||
common_chat_msg user_message;
|
||||
user_message.role = "user";
|
||||
user_message.content = "Hello";
|
||||
inputs.messages = { user_message };
|
||||
inputs.add_generation_prompt = true;
|
||||
inputs.reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
|
||||
inputs.use_jinja = true;
|
||||
|
||||
const common_chat_params params =
|
||||
common_chat_templates_apply(tmpls.get(), inputs);
|
||||
t.assert_equal(
|
||||
"Inkling uses the native PEG parser",
|
||||
COMMON_CHAT_FORMAT_PEG_NATIVE,
|
||||
params.format);
|
||||
t.assert_true("Inkling exposes thinking blocks", params.supports_thinking);
|
||||
t.assert_equal(
|
||||
"Inkling generation marker",
|
||||
"<|message_model|>",
|
||||
params.generation_prompt);
|
||||
|
||||
common_peg_arena arena;
|
||||
arena.load(params.parser);
|
||||
common_chat_parser_params parser_params(params);
|
||||
parser_params.reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
|
||||
|
||||
const common_chat_msg parsed = common_chat_peg_parse(
|
||||
arena,
|
||||
"<|content_thinking|>plan<|end_message|>"
|
||||
"<|message_model|><|content_text|>answer<|end_message|>"
|
||||
"<|content_model_end_sampling|>",
|
||||
false,
|
||||
parser_params);
|
||||
t.assert_equal("reasoning block", "plan", parsed.reasoning_content);
|
||||
t.assert_equal("text block", "answer", parsed.content);
|
||||
}
|
||||
|
||||
// Test that reproduces the Seed-OSS template issue with embedded quotes
|
||||
static void test_tagged_args_with_embedded_quotes(testing & t) {
|
||||
json tools = build_edit_tool();
|
||||
@@ -2192,4 +2241,3 @@ static void test_tagged_args_with_embedded_quotes(testing & t) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -114,6 +114,11 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
||||
n_layer = 3;
|
||||
} else if (arch == LLM_ARCH_CHAMELEON) {
|
||||
n_vocab = 10240;
|
||||
} else if (arch == LLM_ARCH_INKLING) {
|
||||
n_embd = 64;
|
||||
n_head = 2;
|
||||
n_ff = 96;
|
||||
n_layer = 2;
|
||||
}
|
||||
|
||||
const uint32_t n_embd_head = n_embd / n_head;
|
||||
@@ -198,6 +203,10 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
||||
pattern.push_back(il % 2);
|
||||
}
|
||||
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, pattern);
|
||||
} else if (arch == LLM_ARCH_INKLING) {
|
||||
ms.add_kv(
|
||||
LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN,
|
||||
std::vector<uint32_t>({1, 0}));
|
||||
} else {
|
||||
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, uint32_t(2));
|
||||
}
|
||||
@@ -217,12 +226,27 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
||||
if (moe) {
|
||||
ms.add_kv(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, n_ff);
|
||||
ms.add_kv(LLM_KV_INTERLEAVE_MOE_LAYER_STEP, uint32_t(2));
|
||||
ms.add_kv(LLM_KV_EXPERT_COUNT, uint32_t(2));
|
||||
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, uint32_t(1));
|
||||
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, uint32_t(1));
|
||||
ms.add_kv(LLM_KV_EXPERT_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(8) : uint32_t(2));
|
||||
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(6) : uint32_t(1));
|
||||
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(2) : uint32_t(1));
|
||||
ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, uint32_t(2)); // sigmoid
|
||||
ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f);
|
||||
ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1));
|
||||
if (arch == LLM_ARCH_INKLING) {
|
||||
ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE, 1.0f);
|
||||
}
|
||||
}
|
||||
|
||||
if (arch == LLM_ARCH_INKLING) {
|
||||
ms.add_kv(LLM_KV_INKLING_D_REL, uint32_t(8));
|
||||
ms.add_kv(LLM_KV_INKLING_REL_EXTENT, uint32_t(32));
|
||||
ms.add_kv(LLM_KV_INKLING_REL_EXTENT_SWA, uint32_t(16));
|
||||
ms.add_kv(LLM_KV_INKLING_SHORTCONV_KERNEL, uint32_t(4));
|
||||
ms.add_kv(LLM_KV_INKLING_DENSE_BLOCK_COUNT, uint32_t(1));
|
||||
ms.add_kv(LLM_KV_INKLING_LOGIT_SCALE_DENOM, 1.0f);
|
||||
ms.add_kv(LLM_KV_INKLING_LOG_SCALING_N_FLOOR, uint32_t(16));
|
||||
ms.add_kv(LLM_KV_INKLING_LOG_SCALING_ALPHA, 0.1f);
|
||||
ms.add_kv(LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, n_vocab);
|
||||
}
|
||||
|
||||
ms.add_kv(LLM_KV_POSNET_EMBEDDING_LENGTH, n_embd);
|
||||
@@ -372,6 +396,7 @@ static bool moe_mandatory(const llm_arch arch) {
|
||||
case LLM_ARCH_MISTRAL4:
|
||||
case LLM_ARCH_MELLUM:
|
||||
case LLM_ARCH_LAGUNA:
|
||||
case LLM_ARCH_INKLING:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
@@ -478,7 +503,9 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
|
||||
continue;
|
||||
}
|
||||
const bool can_save_fixture =
|
||||
llama_model_saver_supports_arch(arch) || arch == LLM_ARCH_LAGUNA;
|
||||
llama_model_saver_supports_arch(arch) ||
|
||||
arch == LLM_ARCH_LAGUNA ||
|
||||
arch == LLM_ARCH_INKLING;
|
||||
if (!can_save_fixture || !arch_supported(arch)) {
|
||||
LOG_INF("%s: %s model (%s) is unsupported, skipping\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense");
|
||||
continue;
|
||||
@@ -487,15 +514,21 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
|
||||
auto model_and_ctx = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {});
|
||||
const std::string path = dir + "/" + llm_arch_name(arch) + (moe ? "-moe.gguf" : "-dense.gguf");
|
||||
LOG_INF("%s: Saving %s model (%s) to %s...\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense", path.c_str());
|
||||
if (llama_model_saver_supports_arch(arch)) {
|
||||
const bool private_fixture =
|
||||
arch == LLM_ARCH_LAGUNA || arch == LLM_ARCH_INKLING;
|
||||
if (llama_model_saver_supports_arch(arch) && !private_fixture) {
|
||||
llama_model_save_to_file(model_and_ctx.first.get(), path.c_str());
|
||||
} else {
|
||||
// Laguna is not supported by the production model saver yet.
|
||||
// Laguna and Inkling are not supported by the production
|
||||
// model saver yet.
|
||||
// Preserve the exact synthetic fixture metadata and attach the
|
||||
// initialized model tensors so longhaul can be tested from a
|
||||
// real file without broadening the saver API.
|
||||
gguf_context_ptr fixture(gguf_init_empty());
|
||||
gguf_set_kv(fixture.get(), gguf_ctx.get());
|
||||
// Model initialization normalizes the caller's GGUF context,
|
||||
// so regenerate the exact private metadata for the file.
|
||||
gguf_context_ptr fixture_metadata = get_gguf_ctx(arch, moe);
|
||||
gguf_set_kv(fixture.get(), fixture_metadata.get());
|
||||
for (const auto & item : llama_internal_get_tensor_map(model_and_ctx.first.get())) {
|
||||
gguf_add_tensor(fixture.get(), item.second);
|
||||
}
|
||||
|
||||
+20
-11
@@ -191,12 +191,13 @@ struct cpu_decode_result {
|
||||
bool expert_buffer_is_host = false;
|
||||
};
|
||||
|
||||
static cpu_decode_result decode_cpu_model(
|
||||
static cpu_decode_result decode_model(
|
||||
const std::string & path,
|
||||
llama_load_mode load_mode,
|
||||
uint64_t cache_bytes) {
|
||||
uint64_t cache_bytes,
|
||||
int32_t n_gpu_layers) {
|
||||
llama_model_params model_params = llama_model_default_params();
|
||||
model_params.n_gpu_layers = 0;
|
||||
model_params.n_gpu_layers = n_gpu_layers;
|
||||
model_params.load_mode = load_mode;
|
||||
model_params.longhaul_cache_bytes = cache_bytes;
|
||||
model_params.use_extra_bufts = true;
|
||||
@@ -256,19 +257,24 @@ static cpu_decode_result decode_cpu_model(
|
||||
return result;
|
||||
}
|
||||
|
||||
static void test_cpu_model(
|
||||
static void test_model(
|
||||
testing & t,
|
||||
const std::string & path,
|
||||
uint64_t cache_bytes) {
|
||||
uint64_t cache_bytes,
|
||||
int32_t n_gpu_layers) {
|
||||
const cpu_decode_result regular =
|
||||
decode_cpu_model(path, LLAMA_LOAD_MODE_NONE, 0);
|
||||
decode_model(path, LLAMA_LOAD_MODE_NONE, 0, n_gpu_layers);
|
||||
const cpu_decode_result longhaul =
|
||||
decode_cpu_model(path, LLAMA_LOAD_MODE_LONGHAUL, cache_bytes);
|
||||
decode_model(path, LLAMA_LOAD_MODE_LONGHAUL, cache_bytes, n_gpu_layers);
|
||||
|
||||
t.assert_equal(1u, longhaul.capacity);
|
||||
t.assert_true("longhaul loaded at least one expert", longhaul.misses > 0);
|
||||
t.assert_true("longhaul read expert bytes", longhaul.bytes_read > 0);
|
||||
t.assert_true("streamed experts use a directly writable host buffer", longhaul.expert_buffer_is_host);
|
||||
if (n_gpu_layers == 0) {
|
||||
t.assert_true(
|
||||
"streamed CPU experts use a directly writable host buffer",
|
||||
longhaul.expert_buffer_is_host);
|
||||
}
|
||||
t.assert_equal(regular.logits.size(), longhaul.logits.size());
|
||||
|
||||
double squared_error = 0.0;
|
||||
@@ -296,11 +302,14 @@ int main(int argc, char ** argv) {
|
||||
llama_log_set([](ggml_log_level, const char *, void *) {}, nullptr);
|
||||
}
|
||||
|
||||
if (argc == 4 && std::string(argv[1]) == "--cpu-model") {
|
||||
if (argc == 4 &&
|
||||
(std::string(argv[1]) == "--cpu-model" ||
|
||||
std::string(argv[1]) == "--metal-model")) {
|
||||
const std::string path = argv[2];
|
||||
const uint64_t cache_bytes = std::stoull(argv[3]);
|
||||
t.test("cpu_model", [&](testing & current) {
|
||||
test_cpu_model(current, path, cache_bytes);
|
||||
const bool metal = std::string(argv[1]) == "--metal-model";
|
||||
t.test(metal ? "metal_model" : "cpu_model", [&](testing & current) {
|
||||
test_model(current, path, cache_bytes, metal ? 99 : 0);
|
||||
});
|
||||
const int result = t.summary();
|
||||
llama_backend_free();
|
||||
|
||||
+1
-1
@@ -61,7 +61,7 @@
|
||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||
| `--longhaul` | stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||
|
||||
@@ -144,7 +144,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
|
||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||
| `--longhaul` | stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||
|
||||
@@ -78,7 +78,7 @@ For the full list of features, please refer to [server's changelog](https://gith
|
||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||
| `--longhaul` | stream routed Qwen3.5 MoE or Laguna experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||
|
||||
Reference in New Issue
Block a user