Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d080ba3d65 |
+1
-1
@@ -2631,7 +2631,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
|||||||
).set_env("LLAMA_ARG_LOAD_MODE"));
|
).set_env("LLAMA_ARG_LOAD_MODE"));
|
||||||
add_opt(common_arg(
|
add_opt(common_arg(
|
||||||
{"--longhaul"},
|
{"--longhaul"},
|
||||||
"stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache",
|
"stream routed Qwen3.5 MoE experts through a bounded Metal cache",
|
||||||
[](common_params & params) {
|
[](common_params & params) {
|
||||||
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
|
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
|
||||||
}
|
}
|
||||||
|
|||||||
-119
@@ -2530,118 +2530,6 @@ static common_chat_params common_chat_params_init_minimax_m3(const common_chat_t
|
|||||||
return data;
|
return data;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Inkling's TML format can emit multiple typed content blocks in one turn.
|
|
||||||
// <|end_message|> closes a block while content_model_end_sampling closes the
|
|
||||||
// generation, so the generic single-block parser is not sufficient.
|
|
||||||
static common_chat_params common_chat_params_init_inkling(
|
|
||||||
const common_chat_template & tmpl,
|
|
||||||
const autoparser::generation_params & inputs) {
|
|
||||||
common_chat_params data;
|
|
||||||
|
|
||||||
const std::string MSG_MODEL = "<|message_model|>";
|
|
||||||
const std::string MSG_USER = "<|message_user|>";
|
|
||||||
const std::string MSG_SYSTEM = "<|message_system|>";
|
|
||||||
const std::string MSG_TOOL = "<|message_tool|>";
|
|
||||||
const std::string THINK = "<|content_thinking|>";
|
|
||||||
const std::string TEXT = "<|content_text|>";
|
|
||||||
const std::string END_MESSAGE = "<|end_message|>";
|
|
||||||
const std::string END_SAMPLING = "<|content_model_end_sampling|>";
|
|
||||||
const std::string INVOKE_TOOL = "<|content_invoke_tool_json|>";
|
|
||||||
|
|
||||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);
|
|
||||||
data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs);
|
|
||||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
|
||||||
data.supports_thinking = true;
|
|
||||||
data.thinking_start_tag = THINK;
|
|
||||||
data.thinking_end_tags = {END_MESSAGE};
|
|
||||||
data.preserved_tokens = {
|
|
||||||
MSG_MODEL, MSG_USER, MSG_SYSTEM, MSG_TOOL,
|
|
||||||
THINK, TEXT, END_MESSAGE, END_SAMPLING, INVOKE_TOOL,
|
|
||||||
};
|
|
||||||
data.message_delimiters = {
|
|
||||||
{ COMMON_CHAT_ROLE_ASSISTANT, MSG_MODEL },
|
|
||||||
{ COMMON_CHAT_ROLE_USER, MSG_USER },
|
|
||||||
{ COMMON_CHAT_ROLE_SYSTEM, MSG_SYSTEM },
|
|
||||||
{ COMMON_CHAT_ROLE_TOOL, MSG_TOOL },
|
|
||||||
};
|
|
||||||
|
|
||||||
const bool has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
|
||||||
const bool extract_reasoning =
|
|
||||||
inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
|
||||||
|
|
||||||
if (inputs.has_continuation()) {
|
|
||||||
const auto & msg = inputs.continue_msg;
|
|
||||||
data.generation_prompt = MSG_MODEL + THINK + msg.reasoning_content;
|
|
||||||
if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
|
|
||||||
data.generation_prompt += END_MESSAGE + TEXT + msg.render_content();
|
|
||||||
}
|
|
||||||
data.prompt += data.generation_prompt;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
|
||||||
auto generation_prompt = p.literal(MSG_MODEL);
|
|
||||||
auto end = p.end();
|
|
||||||
|
|
||||||
common_peg_parser reasoning_block = p.eps();
|
|
||||||
if (extract_reasoning) {
|
|
||||||
reasoning_block =
|
|
||||||
p.literal(THINK) +
|
|
||||||
p.reasoning(p.until_one_of({ END_MESSAGE, TEXT, END_SAMPLING })) +
|
|
||||||
p.optional(p.literal(END_MESSAGE));
|
|
||||||
} else {
|
|
||||||
reasoning_block = p.content(
|
|
||||||
p.literal(THINK) +
|
|
||||||
p.until_one_of({ END_MESSAGE, TEXT, END_SAMPLING }) +
|
|
||||||
p.optional(p.literal(END_MESSAGE)));
|
|
||||||
}
|
|
||||||
auto reasoning = p.optional(reasoning_block);
|
|
||||||
|
|
||||||
auto text_block =
|
|
||||||
p.optional(p.literal(MSG_MODEL)) +
|
|
||||||
p.optional(p.literal(TEXT)) +
|
|
||||||
p.content(p.until_one_of({ THINK, END_MESSAGE, END_SAMPLING })) +
|
|
||||||
p.optional(p.literal(END_MESSAGE));
|
|
||||||
auto text_content =
|
|
||||||
p.one_or_more(p.choice({ reasoning_block, text_block }));
|
|
||||||
|
|
||||||
if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
|
|
||||||
return generation_prompt + reasoning + text_content +
|
|
||||||
p.optional(p.literal(END_SAMPLING)) + end;
|
|
||||||
}
|
|
||||||
|
|
||||||
auto tool_section = p.standard_json_tools(
|
|
||||||
INVOKE_TOOL, END_MESSAGE, inputs.tools,
|
|
||||||
/* parallel_tool_calls = */ false,
|
|
||||||
/* force_tool_calls = */ true,
|
|
||||||
/* name_key = */ "name",
|
|
||||||
/* args_key = */ "args",
|
|
||||||
/* array_wrapped = */ false,
|
|
||||||
/* function_is_key = */ false,
|
|
||||||
/* call_id_key = */ "",
|
|
||||||
/* gen_call_id_key = */ "",
|
|
||||||
/* parameters_order = */ {},
|
|
||||||
/* accept_openai_wrapper = */ false);
|
|
||||||
auto tool_block =
|
|
||||||
p.optional(p.literal(MSG_MODEL)) +
|
|
||||||
p.until_one_of({ INVOKE_TOOL, TEXT, THINK, END_MESSAGE, END_SAMPLING }) +
|
|
||||||
tool_section;
|
|
||||||
auto tool_calls = inputs.parallel_tool_calls
|
|
||||||
? p.one_or_more(tool_block)
|
|
||||||
: tool_block;
|
|
||||||
auto mixed_body =
|
|
||||||
p.one_or_more(p.choice({ tool_block, reasoning_block, text_block }));
|
|
||||||
auto body = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED
|
|
||||||
? tool_calls
|
|
||||||
: mixed_body;
|
|
||||||
|
|
||||||
return generation_prompt + reasoning + body +
|
|
||||||
p.optional(p.literal(END_SAMPLING)) + end;
|
|
||||||
});
|
|
||||||
|
|
||||||
data.parser = parser.save();
|
|
||||||
return data;
|
|
||||||
}
|
|
||||||
|
|
||||||
namespace workaround {
|
namespace workaround {
|
||||||
|
|
||||||
static void map_developer_role_to_system(json & messages) {
|
static void map_developer_role_to_system(json & messages) {
|
||||||
@@ -3059,13 +2947,6 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
|||||||
return common_chat_params_init_cohere2moe(tmpl, params);
|
return common_chat_params_init_cohere2moe(tmpl, params);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (src.find("<|content_thinking|>") != std::string::npos &&
|
|
||||||
src.find("<|content_text|>") != std::string::npos &&
|
|
||||||
src.find("<|message_model|>") != std::string::npos) {
|
|
||||||
LOG_DBG("Using specialized template: Inkling\n");
|
|
||||||
return common_chat_params_init_inkling(tmpl, params);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (is_lfm2_template(src)) {
|
if (is_lfm2_template(src)) {
|
||||||
LOG_DBG("Using specialized template: LFM2\n");
|
LOG_DBG("Using specialized template: LFM2\n");
|
||||||
return common_chat_params_init_lfm2(tmpl, params, /* tool_list_tokens = */ true);
|
return common_chat_params_init_lfm2(tmpl, params, /* tool_list_tokens = */ true);
|
||||||
|
|||||||
+44
-45
@@ -11,16 +11,45 @@ llama-cli \
|
|||||||
--model /path/to/model.gguf \
|
--model /path/to/model.gguf \
|
||||||
--longhaul \
|
--longhaul \
|
||||||
--longhaul-cache 2 \
|
--longhaul-cache 2 \
|
||||||
--n-gpu-layers 0
|
--n-gpu-layers 99
|
||||||
```
|
```
|
||||||
|
|
||||||
`--longhaul-cache` is the expert cache budget in GiB. It is required when `--longhaul` is used. The same options are accepted by `llama-server`.
|
`--longhaul-cache` is the expert cache budget in GiB. It is required when `--longhaul` is used. The same options are accepted by `llama-server`.
|
||||||
|
|
||||||
The command above runs every repeating layer on CPU and is supported on macOS,
|
## RPC mode
|
||||||
Linux, and Windows. On macOS, the existing all-Metal mode remains available by
|
|
||||||
using `--n-gpu-layers 99` instead. Longhaul does not silently change device
|
RPC longhaul keeps the paging loop on the accelerator host so expert payloads do
|
||||||
placement: an unsupported GPU or mixed CPU/GPU repeating-layer placement fails
|
not cross the network during generation. Build both hosts with `-DGGML_RPC=ON`.
|
||||||
with guidance to use `--n-gpu-layers 0`.
|
Place the same GGUF shard files on both hosts, preserving their basenames, then
|
||||||
|
start the Metal host with the directory containing those shards:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
ggml-rpc-server \
|
||||||
|
--device MTL0 \
|
||||||
|
--longhaul-root /path/to/model-directory \
|
||||||
|
--host 192.168.1.20
|
||||||
|
```
|
||||||
|
|
||||||
|
Start the client normally:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
llama-server \
|
||||||
|
--model /local/path/model.gguf \
|
||||||
|
--rpc 192.168.1.20:50052 \
|
||||||
|
--longhaul \
|
||||||
|
--longhaul-cache 2 \
|
||||||
|
--n-gpu-layers 99
|
||||||
|
```
|
||||||
|
|
||||||
|
RPC longhaul currently requires all repeating layers on one remote Metal device.
|
||||||
|
The server validates each shard basename, size, GGUF metadata layout digest, and
|
||||||
|
every registered expert range before decoding. An older server, a server without
|
||||||
|
`--longhaul-root`, or a shard mismatch is a fatal model-load error; the client
|
||||||
|
does not fall back to sending expert slices over RPC.
|
||||||
|
|
||||||
|
`--longhaul-root` grants connected RPC clients read access to registered byte
|
||||||
|
ranges in files in that directory. The RPC server remains experimental and
|
||||||
|
insecure and must not be exposed to an untrusted network.
|
||||||
|
|
||||||
Longhaul may reduce `--ubatch-size` so that every expert selected by one graph segment can be present in the cache at the same time. The effective value is logged during context creation. If one token selects more experts than the cache has slots, the routed MoE computation is split into multiple stages and the partial results are summed. This permits smaller caches at the cost of additional graph work.
|
Longhaul may reduce `--ubatch-size` so that every expert selected by one graph segment can be present in the cache at the same time. The effective value is logged during context creation. If one token selects more experts than the cache has slots, the routed MoE computation is split into multiple stages and the partial results are summed. This permits smaller caches at the cost of additional graph work.
|
||||||
|
|
||||||
@@ -30,53 +59,23 @@ The normal startup warmup is skipped automatically in longhaul mode. Routed expe
|
|||||||
|
|
||||||
Longhaul currently requires:
|
Longhaul currently requires:
|
||||||
|
|
||||||
- all repeating layers on CPU, or all repeating layers on Metal
|
- the Metal backend, either local on macOS or on one RPC host
|
||||||
- Qwen3.5 MoE, Laguna, or Inkling architecture
|
- Qwen3.5 MoE or Laguna architecture
|
||||||
|
- all repeating model layers assigned to local Metal or one RPC Metal device
|
||||||
- text generation without embeddings or LoRA adapters
|
- text generation without embeddings or LoRA adapters
|
||||||
|
|
||||||
CPU mode is available wherever the CPU backend is supported. Metal mode requires
|
Longhaul does not restrict the GGUF quantization type. Individual tensor types must still be supported by Metal.
|
||||||
macOS. Longhaul does not restrict the GGUF quantization type; individual tensor
|
|
||||||
types must still be supported by the selected compute backend.
|
|
||||||
|
|
||||||
MTP/speculative decoding, tensor validation during loading, vocabulary-only
|
MTP/speculative decoding, tensor validation during loading, vocabulary-only loading, and CPU or mixed CPU/Metal layer placement are not supported. Both single-file and split GGUF models are supported; routed expert tensors are read from the shard that owns each tensor.
|
||||||
loading, and mixed CPU/GPU repeating-layer placement are not supported. Both
|
|
||||||
single-file and split GGUF models are supported; routed expert tensors are read
|
|
||||||
from the shard that owns each tensor.
|
|
||||||
|
|
||||||
The cache budget covers the compact routed-expert tensors. It does not include
|
The cache budget covers the compact routed-expert tensors. It does not include dense weights, attention weights, the KV cache, graph allocations, or the temporary buffer used for one expert read. Disk reads bypass the macOS unified file cache where supported, avoiding a second long-lived copy of streamed weights in system RAM.
|
||||||
dense weights, attention weights, the KV cache, graph allocations, or temporary
|
|
||||||
staging buffers. CPU expert caches use the standard directly writable tensor
|
|
||||||
layout so individual slots can be replaced; optimized CPU buffer selection
|
|
||||||
continues to apply to all non-streamed weights.
|
|
||||||
|
|
||||||
File-cache control is best effort and is outside the explicit cache budget.
|
|
||||||
macOS disables caching for streamed files where supported, and Linux advises the
|
|
||||||
kernel to discard completed reads. Windows may retain file data in its system
|
|
||||||
cache. Windows reads from one GGUF shard are serialized to preserve positional
|
|
||||||
read correctness, while POSIX systems retain concurrent `pread` operations.
|
|
||||||
|
|
||||||
This implementation synchronizes at each routed MoE layer to discover the selected experts, populate missing cache slots, and continue execution with cache-local expert IDs. Storage speed and expert reuse therefore have a large effect on generation speed.
|
This implementation synchronizes at each routed MoE layer to discover the selected experts, populate missing cache slots, and continue execution with cache-local expert IDs. Storage speed and expert reuse therefore have a large effect on generation speed.
|
||||||
|
|
||||||
Expert IDs are planned as a batch at each synchronization point. Experts already
|
Expert IDs are planned as a batch at each synchronization point. Experts already
|
||||||
needed by that batch are protected from eviction, duplicate IDs are loaded only
|
needed by that batch are protected from eviction, duplicate IDs are loaded only
|
||||||
once, and independent expert slices are read concurrently where the platform
|
once, and independent expert slices are read concurrently on shared-memory Metal
|
||||||
supports positional reads. CPU and shared Metal buffers are populated directly;
|
devices. Private Metal buffers use a staged fallback.
|
||||||
private Metal buffers use a staged fallback.
|
|
||||||
|
|
||||||
Inkling keeps its two shared experts resident and streams only the routed
|
|
||||||
256-expert banks. Inkling-Small selects six routed experts per token. For the
|
|
||||||
current three-shard UD-IQ1_S model, a 2 GiB cache provides seven routed-expert
|
|
||||||
slots and therefore executes each MoE layer in one stage:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
llama-cli \
|
|
||||||
--model /path/to/Inkling-Small-UD-IQ1_S-00001-of-00003.gguf \
|
|
||||||
--longhaul \
|
|
||||||
--longhaul-cache 2
|
|
||||||
```
|
|
||||||
|
|
||||||
Add `--n-gpu-layers 0` to run entirely on CPU; on macOS the default all-Metal
|
|
||||||
placement is supported.
|
|
||||||
|
|
||||||
## Benchmarking prompt processing
|
## Benchmarking prompt processing
|
||||||
|
|
||||||
@@ -87,7 +86,7 @@ llama-bench \
|
|||||||
--model /path/to/model.gguf \
|
--model /path/to/model.gguf \
|
||||||
--load-mode longhaul \
|
--load-mode longhaul \
|
||||||
--longhaul-cache 2 \
|
--longhaul-cache 2 \
|
||||||
--n-gpu-layers 0 \
|
--n-gpu-layers 99 \
|
||||||
--n-prompt 2048 \
|
--n-prompt 2048 \
|
||||||
--n-gen 0
|
--n-gen 0
|
||||||
```
|
```
|
||||||
|
|||||||
+44
-1
@@ -8,7 +8,7 @@ extern "C" {
|
|||||||
|
|
||||||
#define RPC_PROTO_MAJOR_VERSION 4
|
#define RPC_PROTO_MAJOR_VERSION 4
|
||||||
#define RPC_PROTO_MINOR_VERSION 0
|
#define RPC_PROTO_MINOR_VERSION 0
|
||||||
#define RPC_PROTO_PATCH_VERSION 3
|
#define RPC_PROTO_PATCH_VERSION 4
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT has changed - update RPC_PROTO_PATCH_VERSION");
|
static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT has changed - update RPC_PROTO_PATCH_VERSION");
|
||||||
@@ -27,6 +27,49 @@ GGML_BACKEND_API void ggml_backend_rpc_get_device_memory(const char * endpoint,
|
|||||||
GGML_BACKEND_API void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir,
|
GGML_BACKEND_API void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir,
|
||||||
size_t n_threads, size_t n_devices, ggml_backend_dev_t * devices);
|
size_t n_threads, size_t n_devices, ggml_backend_dev_t * devices);
|
||||||
|
|
||||||
|
struct ggml_backend_rpc_longhaul_shard {
|
||||||
|
const char * name;
|
||||||
|
uint64_t size;
|
||||||
|
uint64_t metadata_size;
|
||||||
|
uint8_t metadata_sha256[32];
|
||||||
|
};
|
||||||
|
|
||||||
|
struct ggml_backend_rpc_longhaul_source {
|
||||||
|
struct ggml_tensor * tensor;
|
||||||
|
uint32_t shard;
|
||||||
|
uint64_t offset;
|
||||||
|
uint64_t expert_size;
|
||||||
|
int32_t layer;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct ggml_backend_rpc_longhaul_params {
|
||||||
|
uint32_t n_layers;
|
||||||
|
uint32_t n_experts;
|
||||||
|
uint32_t n_slots;
|
||||||
|
size_t n_shards;
|
||||||
|
const struct ggml_backend_rpc_longhaul_shard * shards;
|
||||||
|
size_t n_sources;
|
||||||
|
const struct ggml_backend_rpc_longhaul_source * sources;
|
||||||
|
};
|
||||||
|
|
||||||
|
GGML_BACKEND_API bool ggml_backend_rpc_longhaul_register(
|
||||||
|
const struct ggml_backend_rpc_longhaul_params * params,
|
||||||
|
uint64_t * registration_id,
|
||||||
|
char * error,
|
||||||
|
size_t error_size);
|
||||||
|
|
||||||
|
GGML_BACKEND_API void ggml_backend_rpc_longhaul_unregister(
|
||||||
|
struct ggml_tensor * tensor,
|
||||||
|
uint64_t registration_id);
|
||||||
|
|
||||||
|
GGML_BACKEND_API void ggml_backend_rpc_start_server_with_options(
|
||||||
|
const char * endpoint,
|
||||||
|
const char * cache_dir,
|
||||||
|
const char * longhaul_root,
|
||||||
|
size_t n_threads,
|
||||||
|
size_t n_devices,
|
||||||
|
ggml_backend_dev_t * devices);
|
||||||
|
|
||||||
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_reg(void);
|
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_reg(void);
|
||||||
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint);
|
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint);
|
||||||
|
|
||||||
|
|||||||
+998
-14
File diff suppressed because it is too large
Load Diff
@@ -144,7 +144,6 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
|
|||||||
{ LLM_ARCH_TALKIE, "talkie" },
|
{ LLM_ARCH_TALKIE, "talkie" },
|
||||||
{ LLM_ARCH_MELLUM, "mellum" },
|
{ LLM_ARCH_MELLUM, "mellum" },
|
||||||
{ LLM_ARCH_NANBEIGE, "nanbeige" },
|
{ LLM_ARCH_NANBEIGE, "nanbeige" },
|
||||||
{ LLM_ARCH_INKLING, "inkling" },
|
|
||||||
{ LLM_ARCH_UNKNOWN, "(unknown)" },
|
{ LLM_ARCH_UNKNOWN, "(unknown)" },
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -321,15 +320,6 @@ static const std::map<llm_kv, const char *> LLM_KV_NAMES = {
|
|||||||
{ LLM_KV_NORM_BEFORE_FC, "%s.norm_before_fc" },
|
{ LLM_KV_NORM_BEFORE_FC, "%s.norm_before_fc" },
|
||||||
|
|
||||||
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
|
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
|
||||||
{ LLM_KV_INKLING_D_REL, "%s.d_rel" },
|
|
||||||
{ LLM_KV_INKLING_REL_EXTENT, "%s.rel_extent" },
|
|
||||||
{ LLM_KV_INKLING_REL_EXTENT_SWA, "%s.rel_extent_swa" },
|
|
||||||
{ LLM_KV_INKLING_SHORTCONV_KERNEL, "%s.shortconv_kernel" },
|
|
||||||
{ LLM_KV_INKLING_DENSE_BLOCK_COUNT, "%s.dense_block_count" },
|
|
||||||
{ LLM_KV_INKLING_LOGIT_SCALE_DENOM, "%s.logit_scale_denom" },
|
|
||||||
{ LLM_KV_INKLING_LOG_SCALING_N_FLOOR, "%s.log_scaling_n_floor" },
|
|
||||||
{ LLM_KV_INKLING_LOG_SCALING_ALPHA, "%s.log_scaling_alpha" },
|
|
||||||
{ LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, "%s.unpadded_vocab_size" },
|
|
||||||
// sentence-transformers dense modules feature dims
|
// sentence-transformers dense modules feature dims
|
||||||
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
|
{ LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
|
||||||
{ LLM_KV_DENSE_2_FEAT_OUT, "%s.dense_2_feat_out" },
|
{ LLM_KV_DENSE_2_FEAT_OUT, "%s.dense_2_feat_out" },
|
||||||
@@ -602,19 +592,9 @@ static const std::map<llm_tensor, const char *> LLM_TENSOR_NAMES = {
|
|||||||
{ LLM_TENSOR_SHORTCONV_CONV, "blk.%d.shortconv.conv" },
|
{ LLM_TENSOR_SHORTCONV_CONV, "blk.%d.shortconv.conv" },
|
||||||
{ LLM_TENSOR_SHORTCONV_INPROJ, "blk.%d.shortconv.in_proj" },
|
{ LLM_TENSOR_SHORTCONV_INPROJ, "blk.%d.shortconv.in_proj" },
|
||||||
{ LLM_TENSOR_SHORTCONV_OUTPROJ, "blk.%d.shortconv.out_proj" },
|
{ LLM_TENSOR_SHORTCONV_OUTPROJ, "blk.%d.shortconv.out_proj" },
|
||||||
{ LLM_TENSOR_ATTN_R, "blk.%d.attn_r" },
|
|
||||||
{ LLM_TENSOR_ATTN_REL_PROJ, "blk.%d.attn_rel_proj" },
|
|
||||||
{ LLM_TENSOR_SHORTCONV_K, "blk.%d.shortconv_k" },
|
|
||||||
{ LLM_TENSOR_SHORTCONV_V, "blk.%d.shortconv_v" },
|
|
||||||
{ LLM_TENSOR_SHORTCONV_ATTN, "blk.%d.shortconv_attn" },
|
|
||||||
{ LLM_TENSOR_SHORTCONV_MLP, "blk.%d.shortconv_mlp" },
|
|
||||||
{ LLM_TENSOR_FFN_GSCALE, "blk.%d.ffn_gscale" },
|
|
||||||
{ LLM_TENSOR_FFN_GATE_CHEXPS, "blk.%d.ffn_gate_chexps" },
|
{ LLM_TENSOR_FFN_GATE_CHEXPS, "blk.%d.ffn_gate_chexps" },
|
||||||
{ LLM_TENSOR_FFN_DOWN_CHEXPS, "blk.%d.ffn_down_chexps" },
|
{ LLM_TENSOR_FFN_DOWN_CHEXPS, "blk.%d.ffn_down_chexps" },
|
||||||
{ LLM_TENSOR_FFN_UP_CHEXPS, "blk.%d.ffn_up_chexps" },
|
{ LLM_TENSOR_FFN_UP_CHEXPS, "blk.%d.ffn_up_chexps" },
|
||||||
{ LLM_TENSOR_FFN_GATE_SHEXPS, "blk.%d.ffn_gate_shexp" },
|
|
||||||
{ LLM_TENSOR_FFN_DOWN_SHEXPS, "blk.%d.ffn_down_shexp" },
|
|
||||||
{ LLM_TENSOR_FFN_UP_SHEXPS, "blk.%d.ffn_up_shexp" },
|
|
||||||
{ LLM_TENSOR_VISEXP_ATTN_QKV, "blk.%d.vis_attn_qkv" },
|
{ LLM_TENSOR_VISEXP_ATTN_QKV, "blk.%d.vis_attn_qkv" },
|
||||||
{ LLM_TENSOR_VISEXP_ATTN_OUT, "blk.%d.vis_attn_output" },
|
{ LLM_TENSOR_VISEXP_ATTN_OUT, "blk.%d.vis_attn_output" },
|
||||||
{ LLM_TENSOR_VISEXP_FFN_GATE, "blk.%d.vis_gate" },
|
{ LLM_TENSOR_VISEXP_FFN_GATE, "blk.%d.vis_gate" },
|
||||||
@@ -815,9 +795,6 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
|||||||
{LLM_TENSOR_FFN_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
{LLM_TENSOR_FFN_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||||
{LLM_TENSOR_FFN_GATE_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
{LLM_TENSOR_FFN_GATE_UP_EXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||||
{LLM_TENSOR_FFN_DOWN_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
{LLM_TENSOR_FFN_DOWN_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||||
{LLM_TENSOR_FFN_DOWN_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
|
||||||
{LLM_TENSOR_FFN_GATE_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
|
||||||
{LLM_TENSOR_FFN_UP_SHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
|
||||||
{LLM_TENSOR_FFN_GATE_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
{LLM_TENSOR_FFN_GATE_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||||
{LLM_TENSOR_FFN_UP_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
{LLM_TENSOR_FFN_UP_CHEXPS, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT_ID}},
|
||||||
{LLM_TENSOR_FFN_EXP_PROBS_B, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ADD}},
|
{LLM_TENSOR_FFN_EXP_PROBS_B, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_ADD}},
|
||||||
@@ -859,13 +836,6 @@ static const std::map<llm_tensor, llm_tensor_info> LLM_TENSOR_INFOS = {
|
|||||||
{LLM_TENSOR_SHORTCONV_CONV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
{LLM_TENSOR_SHORTCONV_CONV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
||||||
{LLM_TENSOR_SHORTCONV_INPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
{LLM_TENSOR_SHORTCONV_INPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||||
{LLM_TENSOR_SHORTCONV_OUTPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
{LLM_TENSOR_SHORTCONV_OUTPROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||||
{LLM_TENSOR_ATTN_R, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
|
||||||
{LLM_TENSOR_ATTN_REL_PROJ, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
|
||||||
{LLM_TENSOR_SHORTCONV_K, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
|
||||||
{LLM_TENSOR_SHORTCONV_V, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
|
||||||
{LLM_TENSOR_SHORTCONV_ATTN, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
|
||||||
{LLM_TENSOR_SHORTCONV_MLP, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_SSM_CONV}},
|
|
||||||
{LLM_TENSOR_FFN_GSCALE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL}},
|
|
||||||
{LLM_TENSOR_VISEXP_ATTN_QKV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
{LLM_TENSOR_VISEXP_ATTN_QKV, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||||
{LLM_TENSOR_VISEXP_ATTN_OUT, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
{LLM_TENSOR_VISEXP_ATTN_OUT, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||||
{LLM_TENSOR_VISEXP_FFN_GATE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
{LLM_TENSOR_VISEXP_FFN_GATE, {LLM_TENSOR_LAYER_REPEATING, GGML_OP_MUL_MAT}},
|
||||||
@@ -998,7 +968,6 @@ bool llm_arch_is_hybrid(const llm_arch & arch) {
|
|||||||
case LLM_ARCH_KIMI_LINEAR:
|
case LLM_ARCH_KIMI_LINEAR:
|
||||||
case LLM_ARCH_QWEN35:
|
case LLM_ARCH_QWEN35:
|
||||||
case LLM_ARCH_QWEN35MOE:
|
case LLM_ARCH_QWEN35MOE:
|
||||||
case LLM_ARCH_INKLING:
|
|
||||||
return true;
|
return true;
|
||||||
default:
|
default:
|
||||||
return false;
|
return false;
|
||||||
@@ -1055,7 +1024,6 @@ bool llm_arch_supports_sm_tensor(const llm_arch & arch) {
|
|||||||
case LLM_ARCH_MINIMAX_M3:
|
case LLM_ARCH_MINIMAX_M3:
|
||||||
case LLM_ARCH_MISTRAL4:
|
case LLM_ARCH_MISTRAL4:
|
||||||
case LLM_ARCH_KIMI_LINEAR:
|
case LLM_ARCH_KIMI_LINEAR:
|
||||||
case LLM_ARCH_INKLING:
|
|
||||||
return false;
|
return false;
|
||||||
default:
|
default:
|
||||||
return true;
|
return true;
|
||||||
|
|||||||
@@ -149,7 +149,6 @@ enum llm_arch {
|
|||||||
LLM_ARCH_MINIMAX_M3,
|
LLM_ARCH_MINIMAX_M3,
|
||||||
LLM_ARCH_DFLASH,
|
LLM_ARCH_DFLASH,
|
||||||
LLM_ARCH_NANBEIGE,
|
LLM_ARCH_NANBEIGE,
|
||||||
LLM_ARCH_INKLING,
|
|
||||||
LLM_ARCH_UNKNOWN,
|
LLM_ARCH_UNKNOWN,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -367,15 +366,6 @@ enum llm_kv {
|
|||||||
LLM_KV_NORM_BEFORE_FC,
|
LLM_KV_NORM_BEFORE_FC,
|
||||||
|
|
||||||
LLM_KV_SHORTCONV_L_CACHE,
|
LLM_KV_SHORTCONV_L_CACHE,
|
||||||
LLM_KV_INKLING_D_REL,
|
|
||||||
LLM_KV_INKLING_REL_EXTENT,
|
|
||||||
LLM_KV_INKLING_REL_EXTENT_SWA,
|
|
||||||
LLM_KV_INKLING_SHORTCONV_KERNEL,
|
|
||||||
LLM_KV_INKLING_DENSE_BLOCK_COUNT,
|
|
||||||
LLM_KV_INKLING_LOGIT_SCALE_DENOM,
|
|
||||||
LLM_KV_INKLING_LOG_SCALING_N_FLOOR,
|
|
||||||
LLM_KV_INKLING_LOG_SCALING_ALPHA,
|
|
||||||
LLM_KV_INKLING_UNPADDED_VOCAB_SIZE,
|
|
||||||
|
|
||||||
LLM_KV_XIELU_ALPHA_N,
|
LLM_KV_XIELU_ALPHA_N,
|
||||||
LLM_KV_XIELU_ALPHA_P,
|
LLM_KV_XIELU_ALPHA_P,
|
||||||
@@ -444,9 +434,6 @@ enum llm_tensor {
|
|||||||
LLM_TENSOR_FFN_DOWN_CHEXPS,
|
LLM_TENSOR_FFN_DOWN_CHEXPS,
|
||||||
LLM_TENSOR_FFN_GATE_CHEXPS,
|
LLM_TENSOR_FFN_GATE_CHEXPS,
|
||||||
LLM_TENSOR_FFN_UP_CHEXPS,
|
LLM_TENSOR_FFN_UP_CHEXPS,
|
||||||
LLM_TENSOR_FFN_DOWN_SHEXPS,
|
|
||||||
LLM_TENSOR_FFN_GATE_SHEXPS,
|
|
||||||
LLM_TENSOR_FFN_UP_SHEXPS,
|
|
||||||
LLM_TENSOR_FFN_EXP_PROBS_B,
|
LLM_TENSOR_FFN_EXP_PROBS_B,
|
||||||
LLM_TENSOR_FFN_LATENT_DOWN,
|
LLM_TENSOR_FFN_LATENT_DOWN,
|
||||||
LLM_TENSOR_FFN_LATENT_UP,
|
LLM_TENSOR_FFN_LATENT_UP,
|
||||||
@@ -608,13 +595,6 @@ enum llm_tensor {
|
|||||||
LLM_TENSOR_SHORTCONV_CONV,
|
LLM_TENSOR_SHORTCONV_CONV,
|
||||||
LLM_TENSOR_SHORTCONV_INPROJ,
|
LLM_TENSOR_SHORTCONV_INPROJ,
|
||||||
LLM_TENSOR_SHORTCONV_OUTPROJ,
|
LLM_TENSOR_SHORTCONV_OUTPROJ,
|
||||||
LLM_TENSOR_ATTN_R,
|
|
||||||
LLM_TENSOR_ATTN_REL_PROJ,
|
|
||||||
LLM_TENSOR_SHORTCONV_K,
|
|
||||||
LLM_TENSOR_SHORTCONV_V,
|
|
||||||
LLM_TENSOR_SHORTCONV_ATTN,
|
|
||||||
LLM_TENSOR_SHORTCONV_MLP,
|
|
||||||
LLM_TENSOR_FFN_GSCALE,
|
|
||||||
LLM_TENSOR_VISEXP_ATTN_QKV,
|
LLM_TENSOR_VISEXP_ATTN_QKV,
|
||||||
LLM_TENSOR_VISEXP_ATTN_OUT,
|
LLM_TENSOR_VISEXP_ATTN_OUT,
|
||||||
LLM_TENSOR_VISEXP_FFN_GATE,
|
LLM_TENSOR_VISEXP_FFN_GATE,
|
||||||
|
|||||||
@@ -1359,10 +1359,12 @@ llm_graph_result * llama_context::process_ubatch(const llama_ubatch & ubatch, ll
|
|||||||
res->reset();
|
res->reset();
|
||||||
|
|
||||||
ggml_backend_sched_reset(sched.get());
|
ggml_backend_sched_reset(sched.get());
|
||||||
|
auto * longhaul = model.longhaul_cache();
|
||||||
|
const bool local_longhaul = longhaul && !longhaul->is_remote();
|
||||||
ggml_backend_sched_set_eval_callback(
|
ggml_backend_sched_set_eval_callback(
|
||||||
sched.get(),
|
sched.get(),
|
||||||
model.longhaul_cache() ? longhaul_eval_callback : cparams.cb_eval,
|
local_longhaul ? longhaul_eval_callback : cparams.cb_eval,
|
||||||
model.longhaul_cache() ? this : cparams.cb_eval_user_data);
|
local_longhaul ? this : cparams.cb_eval_user_data);
|
||||||
|
|
||||||
//const auto t_start_us = ggml_time_us();
|
//const auto t_start_us = ggml_time_us();
|
||||||
|
|
||||||
@@ -2461,7 +2463,7 @@ llm_graph_params llama_context::graph_params(
|
|||||||
bool llama_context::longhaul_eval_callback(ggml_tensor * tensor, bool ask, void * user_data) {
|
bool llama_context::longhaul_eval_callback(ggml_tensor * tensor, bool ask, void * user_data) {
|
||||||
auto * ctx = static_cast<llama_context *>(user_data);
|
auto * ctx = static_cast<llama_context *>(user_data);
|
||||||
auto * cache = ctx->model.longhaul_cache();
|
auto * cache = ctx->model.longhaul_cache();
|
||||||
GGML_ASSERT(cache != nullptr);
|
GGML_ASSERT(cache != nullptr && !cache->is_remote());
|
||||||
|
|
||||||
int layer = -1;
|
int layer = -1;
|
||||||
const bool remap = sscanf(tensor->name, "longhaul.remap.%d", &layer) == 1;
|
const bool remap = sscanf(tensor->name, "longhaul.remap.%d", &layer) == 1;
|
||||||
|
|||||||
@@ -1765,119 +1765,6 @@ ggml_tensor * llm_graph_context::build_ffn(
|
|||||||
return cur;
|
return cur;
|
||||||
}
|
}
|
||||||
|
|
||||||
ggml_tensor * llm_graph_context::build_moe_experts(
|
|
||||||
ggml_tensor * cur,
|
|
||||||
ggml_tensor * selected_experts,
|
|
||||||
ggml_tensor * weights,
|
|
||||||
ggml_tensor * up_exps,
|
|
||||||
ggml_tensor * gate_exps,
|
|
||||||
ggml_tensor * down_exps,
|
|
||||||
int64_t n_expert_used,
|
|
||||||
llm_ffn_op_type type_op,
|
|
||||||
int il) const {
|
|
||||||
GGML_ASSERT(cur != nullptr);
|
|
||||||
GGML_ASSERT(selected_experts != nullptr);
|
|
||||||
GGML_ASSERT(weights != nullptr);
|
|
||||||
GGML_ASSERT(up_exps != nullptr && down_exps != nullptr);
|
|
||||||
|
|
||||||
const int64_t n_embd = cur->ne[0];
|
|
||||||
const int64_t n_tokens = cur->ne[1];
|
|
||||||
|
|
||||||
// Ensure routing and weights are available before a longhaul cache
|
|
||||||
// callback resolves and loads the first expert chunk.
|
|
||||||
ggml_build_forward_expand(gf, weights);
|
|
||||||
|
|
||||||
ggml_tensor * moe_inp = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
|
|
||||||
const int64_t chunk_size = longhaul
|
|
||||||
? std::min<int64_t>(n_expert_used, longhaul->capacity())
|
|
||||||
: n_expert_used;
|
|
||||||
GGML_ASSERT(chunk_size > 0);
|
|
||||||
|
|
||||||
ggml_tensor * moe_out = nullptr;
|
|
||||||
int chunk = 0;
|
|
||||||
|
|
||||||
for (int64_t first = 0; first < n_expert_used; first += chunk_size, ++chunk) {
|
|
||||||
const int64_t count = std::min<int64_t>(chunk_size, n_expert_used - first);
|
|
||||||
|
|
||||||
ggml_tensor * selected_chunk = selected_experts;
|
|
||||||
ggml_tensor * weights_chunk = weights;
|
|
||||||
if (count != n_expert_used) {
|
|
||||||
selected_chunk = ggml_view_2d(
|
|
||||||
ctx0, selected_experts, count, n_tokens,
|
|
||||||
selected_experts->nb[1], first*selected_experts->nb[0]);
|
|
||||||
weights_chunk = ggml_view_3d(
|
|
||||||
ctx0, weights, 1, count, n_tokens,
|
|
||||||
weights->nb[1], weights->nb[2], first*weights->nb[1]);
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * selected_moe = selected_chunk;
|
|
||||||
if (longhaul) {
|
|
||||||
selected_moe = ggml_dup(ctx0, selected_chunk);
|
|
||||||
ggml_format_name(selected_moe, "longhaul.remap.%d.%d", il, chunk);
|
|
||||||
ggml_build_forward_expand(gf, selected_moe);
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * up = build_lora_mm_id(up_exps, moe_inp, selected_moe);
|
|
||||||
ggml_tensor * act = gate_exps
|
|
||||||
? build_lora_mm_id(gate_exps, moe_inp, selected_moe)
|
|
||||||
: up;
|
|
||||||
|
|
||||||
switch (type_op) {
|
|
||||||
case LLM_FFN_SILU:
|
|
||||||
act = gate_exps
|
|
||||||
? ggml_swiglu_split(ctx0, act, up)
|
|
||||||
: ggml_silu(ctx0, act);
|
|
||||||
break;
|
|
||||||
case LLM_FFN_GELU:
|
|
||||||
act = gate_exps
|
|
||||||
? ggml_geglu_split(ctx0, act, up)
|
|
||||||
: ggml_gelu(ctx0, act);
|
|
||||||
break;
|
|
||||||
case LLM_FFN_RELU:
|
|
||||||
act = gate_exps
|
|
||||||
? ggml_reglu_split(ctx0, act, up)
|
|
||||||
: ggml_relu(ctx0, act);
|
|
||||||
break;
|
|
||||||
default:
|
|
||||||
GGML_ABORT("unsupported routed expert activation");
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * experts = build_lora_mm_id(
|
|
||||||
down_exps, act, selected_moe);
|
|
||||||
experts = ggml_mul(ctx0, experts, weights_chunk);
|
|
||||||
ggml_build_forward_expand(gf, experts);
|
|
||||||
|
|
||||||
ggml_tensor * chunk_out = nullptr;
|
|
||||||
for (int64_t i = 0; i < count; ++i) {
|
|
||||||
ggml_tensor * expert = ggml_view_2d(
|
|
||||||
ctx0, experts, n_embd, n_tokens,
|
|
||||||
experts->nb[2], i*experts->nb[1]);
|
|
||||||
ggml_build_forward_expand(gf, expert);
|
|
||||||
chunk_out = chunk_out ? ggml_add(ctx0, chunk_out, expert) : expert;
|
|
||||||
ggml_build_forward_expand(gf, chunk_out);
|
|
||||||
}
|
|
||||||
if (count == 1) {
|
|
||||||
chunk_out = ggml_cont(ctx0, chunk_out);
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * contribution = chunk_out;
|
|
||||||
if (longhaul) {
|
|
||||||
ggml_tensor * release = ggml_cont(ctx0, chunk_out);
|
|
||||||
ggml_format_name(release, "longhaul.release.%d.%d", il, chunk);
|
|
||||||
ggml_build_forward_expand(gf, release);
|
|
||||||
if (chunk_size < n_expert_used) {
|
|
||||||
contribution = release;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
moe_out = moe_out ? ggml_add(ctx0, moe_out, contribution) : contribution;
|
|
||||||
ggml_build_forward_expand(gf, moe_out);
|
|
||||||
}
|
|
||||||
|
|
||||||
cb(moe_out, "ffn_moe_out", il);
|
|
||||||
return moe_out;
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * llm_graph_context::build_moe_ffn(
|
ggml_tensor * llm_graph_context::build_moe_ffn(
|
||||||
ggml_tensor * cur,
|
ggml_tensor * cur,
|
||||||
ggml_tensor * gate_inp,
|
ggml_tensor * gate_inp,
|
||||||
|
|||||||
@@ -1000,20 +1000,6 @@ struct llm_graph_context {
|
|||||||
llm_ffn_gate_type type_gate,
|
llm_ffn_gate_type type_gate,
|
||||||
int il) const;
|
int il) const;
|
||||||
|
|
||||||
// Execute experts for an architecture that supplies its own selected IDs
|
|
||||||
// and weights. In longhaul mode this transparently stages those experts
|
|
||||||
// through the bounded cache.
|
|
||||||
ggml_tensor * build_moe_experts(
|
|
||||||
ggml_tensor * cur,
|
|
||||||
ggml_tensor * selected_experts,
|
|
||||||
ggml_tensor * weights,
|
|
||||||
ggml_tensor * up_exps,
|
|
||||||
ggml_tensor * gate_exps,
|
|
||||||
ggml_tensor * down_exps,
|
|
||||||
int64_t n_expert_used,
|
|
||||||
llm_ffn_op_type type_op,
|
|
||||||
int il) const;
|
|
||||||
|
|
||||||
// build MoE FFN without bias tensors
|
// build MoE FFN without bias tensors
|
||||||
ggml_tensor * build_moe_ffn(
|
ggml_tensor * build_moe_ffn(
|
||||||
ggml_tensor * cur,
|
ggml_tensor * cur,
|
||||||
|
|||||||
@@ -191,10 +191,6 @@ uint32_t llama_hparams::n_embd_k_idx(uint32_t il) const {
|
|||||||
}
|
}
|
||||||
|
|
||||||
uint32_t llama_hparams::n_embd_r() const {
|
uint32_t llama_hparams::n_embd_r() const {
|
||||||
if (n_embd_r_impl != 0) {
|
|
||||||
return n_embd_r_impl;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (wkv_head_size != 0) {
|
if (wkv_head_size != 0) {
|
||||||
// for RWKV models
|
// for RWKV models
|
||||||
return token_shift_count * n_embd;
|
return token_shift_count * n_embd;
|
||||||
|
|||||||
@@ -80,18 +80,6 @@ struct llama_hparams {
|
|||||||
|
|
||||||
uint32_t n_shortconv_l_cache = 0;
|
uint32_t n_shortconv_l_cache = 0;
|
||||||
|
|
||||||
// Explicit recurrent-state size override. Inkling packs four independent
|
|
||||||
// short-convolution histories into the recurrent cache for every layer.
|
|
||||||
uint32_t n_embd_r_impl = 0;
|
|
||||||
|
|
||||||
// Inkling relative-attention and output metadata.
|
|
||||||
uint32_t inkling_d_rel = 0;
|
|
||||||
uint32_t inkling_rel_extent = 0;
|
|
||||||
uint32_t inkling_rel_extent_swa = 0;
|
|
||||||
uint32_t inkling_log_n_floor = 0;
|
|
||||||
float inkling_log_alpha = 0.0f;
|
|
||||||
uint32_t inkling_unpadded_n_vocab = 0;
|
|
||||||
|
|
||||||
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_arr;
|
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_arr;
|
||||||
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_kv_arr;
|
std::array<uint32_t, LLAMA_MAX_LAYERS> n_head_kv_arr;
|
||||||
std::array<uint32_t, LLAMA_MAX_LAYERS> n_ff_arr;
|
std::array<uint32_t, LLAMA_MAX_LAYERS> n_ff_arr;
|
||||||
|
|||||||
@@ -1931,37 +1931,6 @@ void llama_kv_cache::set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void llama_kv_cache::set_input_pos_rel_flat(
|
|
||||||
ggml_tensor * dst,
|
|
||||||
const llama_ubatch * ubatch,
|
|
||||||
uint32_t extent) const {
|
|
||||||
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
|
|
||||||
GGML_ASSERT(dst->ne[1] == (int64_t) ubatch->n_tokens);
|
|
||||||
|
|
||||||
int32_t * data = (int32_t *) dst->data;
|
|
||||||
const int64_t n_kv = dst->ne[0];
|
|
||||||
|
|
||||||
// Each query owns an extent+1-row block in the flattened relative-logit
|
|
||||||
// table. The final row is the zero-bias sentinel for empty/out-of-band KV
|
|
||||||
// cells.
|
|
||||||
for (int64_t i = 0; i < (int64_t) ubatch->n_tokens; ++i) {
|
|
||||||
const llama_seq_id seq_id = ubatch->seq_id[i][0];
|
|
||||||
const auto & cells = v_cells[seq_to_stream[seq_id]];
|
|
||||||
const llama_pos p1 = ubatch->pos[i];
|
|
||||||
|
|
||||||
for (int64_t j = 0; j < n_kv; ++j) {
|
|
||||||
int32_t rel = (int32_t) extent;
|
|
||||||
if (!cells.is_empty(j)) {
|
|
||||||
const llama_pos d = p1 - cells.pos_get(j);
|
|
||||||
if (d >= 0 && d < (llama_pos) extent) {
|
|
||||||
rel = (int32_t) d;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
data[i*n_kv + j] = (int32_t) (i*(extent + 1)) + rel;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void llama_kv_cache::set_input_k_rot(ggml_tensor * dst) const {
|
void llama_kv_cache::set_input_k_rot(ggml_tensor * dst) const {
|
||||||
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
|
GGML_ASSERT(ggml_backend_buffer_is_host(dst->buffer));
|
||||||
|
|
||||||
@@ -2927,13 +2896,6 @@ void llama_kv_cache_context::set_input_pos_bucket(ggml_tensor * dst, const llama
|
|||||||
kv->set_input_pos_bucket(dst, ubatch);
|
kv->set_input_pos_bucket(dst, ubatch);
|
||||||
}
|
}
|
||||||
|
|
||||||
void llama_kv_cache_context::set_input_pos_rel_flat(
|
|
||||||
ggml_tensor * dst,
|
|
||||||
const llama_ubatch * ubatch,
|
|
||||||
uint32_t extent) const {
|
|
||||||
kv->set_input_pos_rel_flat(dst, ubatch, extent);
|
|
||||||
}
|
|
||||||
|
|
||||||
void llama_kv_cache_context::set_input_k_rot(ggml_tensor * dst) const {
|
void llama_kv_cache_context::set_input_k_rot(ggml_tensor * dst) const {
|
||||||
kv->set_input_k_rot(dst);
|
kv->set_input_k_rot(dst);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -215,8 +215,6 @@ public:
|
|||||||
|
|
||||||
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
|
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
|
||||||
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
|
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
|
||||||
void set_input_pos_rel_flat(
|
|
||||||
ggml_tensor * dst, const llama_ubatch * ubatch, uint32_t extent) const;
|
|
||||||
|
|
||||||
void set_input_k_rot(ggml_tensor * dst) const;
|
void set_input_k_rot(ggml_tensor * dst) const;
|
||||||
void set_input_v_rot(ggml_tensor * dst) const;
|
void set_input_v_rot(ggml_tensor * dst) const;
|
||||||
@@ -407,8 +405,6 @@ public:
|
|||||||
void set_input_k_shift (ggml_tensor * dst) const;
|
void set_input_k_shift (ggml_tensor * dst) const;
|
||||||
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
|
void set_input_kq_mask (ggml_tensor * dst, const llama_ubatch * ubatch, bool causal_attn) const;
|
||||||
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
|
void set_input_pos_bucket(ggml_tensor * dst, const llama_ubatch * ubatch) const;
|
||||||
void set_input_pos_rel_flat(
|
|
||||||
ggml_tensor * dst, const llama_ubatch * ubatch, uint32_t extent) const;
|
|
||||||
|
|
||||||
void set_input_k_rot(ggml_tensor * dst) const;
|
void set_input_k_rot(ggml_tensor * dst) const;
|
||||||
void set_input_v_rot(ggml_tensor * dst) const;
|
void set_input_v_rot(ggml_tensor * dst) const;
|
||||||
|
|||||||
+102
-2
@@ -1,23 +1,108 @@
|
|||||||
#include "llama-longhaul.h"
|
#include "llama-longhaul.h"
|
||||||
|
|
||||||
#include "llama-impl.h"
|
#include "llama-impl.h"
|
||||||
|
#include "llama-token-cache.h"
|
||||||
|
|
||||||
|
#include "ggml-rpc.h"
|
||||||
|
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
|
#include <array>
|
||||||
#include <cstring>
|
#include <cstring>
|
||||||
#include <inttypes.h>
|
#include <inttypes.h>
|
||||||
|
#include <stdexcept>
|
||||||
|
|
||||||
llama_longhaul_cache::llama_longhaul_cache(
|
llama_longhaul_cache::llama_longhaul_cache(
|
||||||
llama_files files,
|
llama_files files,
|
||||||
std::vector<llama_model_loader::longhaul_source> sources,
|
std::vector<llama_model_loader::longhaul_source> sources,
|
||||||
|
std::vector<llama_model_loader::longhaul_shard> shards,
|
||||||
size_t n_slots,
|
size_t n_slots,
|
||||||
uint32_t n_experts,
|
uint32_t n_experts,
|
||||||
uint32_t n_layers) :
|
uint32_t n_layers,
|
||||||
|
bool remote) :
|
||||||
files(std::move(files)),
|
files(std::move(files)),
|
||||||
sources(std::move(sources)),
|
sources(std::move(sources)),
|
||||||
|
shards(std::move(shards)),
|
||||||
sources_by_layer(n_layers),
|
sources_by_layer(n_layers),
|
||||||
layers(n_layers),
|
layers(n_layers),
|
||||||
n_slots(n_slots),
|
n_slots(n_slots),
|
||||||
n_experts(n_experts) {
|
n_experts(n_experts),
|
||||||
|
remote(remote) {
|
||||||
|
if (remote) {
|
||||||
|
if (this->shards.size() != this->files.size() || this->sources.empty()) {
|
||||||
|
throw std::runtime_error("RPC longhaul requires named GGUF shards");
|
||||||
|
}
|
||||||
|
std::vector<std::array<uint8_t, 32>> digests(this->shards.size());
|
||||||
|
std::vector<ggml_backend_rpc_longhaul_shard> rpc_shards(this->shards.size());
|
||||||
|
for (size_t i = 0; i < this->shards.size(); ++i) {
|
||||||
|
const auto & shard = this->shards[i];
|
||||||
|
if (shard.metadata_size > shard.size) {
|
||||||
|
throw std::runtime_error("invalid RPC longhaul shard metadata range");
|
||||||
|
}
|
||||||
|
std::vector<uint8_t> metadata(shard.metadata_size);
|
||||||
|
if (!metadata.empty()) {
|
||||||
|
this->files[i]->read_at(metadata.data(), metadata.size(), 0);
|
||||||
|
}
|
||||||
|
const std::string hex = llama_token_cache_hash(metadata.data(), metadata.size());
|
||||||
|
if (hex.size() != 64) {
|
||||||
|
throw std::runtime_error("failed to fingerprint RPC longhaul shard");
|
||||||
|
}
|
||||||
|
for (size_t j = 0; j < digests[i].size(); ++j) {
|
||||||
|
auto nibble = [](char c) -> uint8_t {
|
||||||
|
if (c >= '0' && c <= '9') return c - '0';
|
||||||
|
if (c >= 'a' && c <= 'f') return c - 'a' + 10;
|
||||||
|
if (c >= 'A' && c <= 'F') return c - 'A' + 10;
|
||||||
|
return 0xff;
|
||||||
|
};
|
||||||
|
const uint8_t hi = nibble(hex[2*j]);
|
||||||
|
const uint8_t lo = nibble(hex[2*j + 1]);
|
||||||
|
if (hi > 0x0f || lo > 0x0f) {
|
||||||
|
throw std::runtime_error("invalid RPC longhaul shard fingerprint");
|
||||||
|
}
|
||||||
|
digests[i][j] = (hi << 4) | lo;
|
||||||
|
}
|
||||||
|
rpc_shards[i] = {
|
||||||
|
shard.name.c_str(),
|
||||||
|
shard.size,
|
||||||
|
shard.metadata_size,
|
||||||
|
{},
|
||||||
|
};
|
||||||
|
memcpy(rpc_shards[i].metadata_sha256, digests[i].data(), digests[i].size());
|
||||||
|
}
|
||||||
|
|
||||||
|
std::vector<ggml_backend_rpc_longhaul_source> rpc_sources;
|
||||||
|
rpc_sources.reserve(this->sources.size());
|
||||||
|
for (const auto & source : this->sources) {
|
||||||
|
rpc_sources.push_back({
|
||||||
|
source.tensor,
|
||||||
|
source.file_idx,
|
||||||
|
source.offset,
|
||||||
|
source.expert_size,
|
||||||
|
source.layer,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
ggml_backend_rpc_longhaul_params params = {
|
||||||
|
n_layers,
|
||||||
|
n_experts,
|
||||||
|
(uint32_t) n_slots,
|
||||||
|
rpc_shards.size(),
|
||||||
|
rpc_shards.data(),
|
||||||
|
rpc_sources.size(),
|
||||||
|
rpc_sources.data(),
|
||||||
|
};
|
||||||
|
ggml_backend_reg_t reg = ggml_backend_reg_by_name("RPC");
|
||||||
|
auto register_fn = reg ? (decltype(ggml_backend_rpc_longhaul_register) *)
|
||||||
|
ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_longhaul_register") : nullptr;
|
||||||
|
if (!register_fn) {
|
||||||
|
throw std::runtime_error("RPC backend does not expose longhaul registration");
|
||||||
|
}
|
||||||
|
char error[256] = {};
|
||||||
|
if (!register_fn(¶ms, &remote_registration_id, error, sizeof(error))) {
|
||||||
|
throw std::runtime_error(error[0] ? error : "RPC longhaul registration failed");
|
||||||
|
}
|
||||||
|
this->files.clear();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
for (auto & file : this->files) {
|
for (auto & file : this->files) {
|
||||||
file->set_no_cache();
|
file->set_no_cache();
|
||||||
}
|
}
|
||||||
@@ -45,6 +130,17 @@ llama_longhaul_cache::llama_longhaul_cache(
|
|||||||
}
|
}
|
||||||
|
|
||||||
llama_longhaul_cache::~llama_longhaul_cache() {
|
llama_longhaul_cache::~llama_longhaul_cache() {
|
||||||
|
if (remote) {
|
||||||
|
if (!sources.empty() && remote_registration_id != 0) {
|
||||||
|
ggml_backend_reg_t reg = ggml_backend_reg_by_name("RPC");
|
||||||
|
auto unregister_fn = reg ? (decltype(ggml_backend_rpc_longhaul_unregister) *)
|
||||||
|
ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_longhaul_unregister") : nullptr;
|
||||||
|
if (unregister_fn) {
|
||||||
|
unregister_fn(sources.front().tensor, remote_registration_id);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
{
|
{
|
||||||
std::lock_guard<std::mutex> lock(io_mutex);
|
std::lock_guard<std::mutex> lock(io_mutex);
|
||||||
io_stopping = true;
|
io_stopping = true;
|
||||||
@@ -321,3 +417,7 @@ uint64_t llama_longhaul_cache::misses() const {
|
|||||||
uint64_t llama_longhaul_cache::bytes_read_count() const {
|
uint64_t llama_longhaul_cache::bytes_read_count() const {
|
||||||
return bytes_read;
|
return bytes_read;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool llama_longhaul_cache::is_remote() const {
|
||||||
|
return remote;
|
||||||
|
}
|
||||||
|
|||||||
@@ -17,9 +17,11 @@ struct llama_longhaul_cache {
|
|||||||
llama_longhaul_cache(
|
llama_longhaul_cache(
|
||||||
llama_files files,
|
llama_files files,
|
||||||
std::vector<llama_model_loader::longhaul_source> sources,
|
std::vector<llama_model_loader::longhaul_source> sources,
|
||||||
|
std::vector<llama_model_loader::longhaul_shard> shards,
|
||||||
size_t n_slots,
|
size_t n_slots,
|
||||||
uint32_t n_experts,
|
uint32_t n_experts,
|
||||||
uint32_t n_layers);
|
uint32_t n_layers,
|
||||||
|
bool remote);
|
||||||
~llama_longhaul_cache();
|
~llama_longhaul_cache();
|
||||||
|
|
||||||
bool remap(int layer, ggml_tensor * ids);
|
bool remap(int layer, ggml_tensor * ids);
|
||||||
@@ -31,6 +33,7 @@ struct llama_longhaul_cache {
|
|||||||
bool failed() const;
|
bool failed() const;
|
||||||
uint64_t misses() const;
|
uint64_t misses() const;
|
||||||
uint64_t bytes_read_count() const;
|
uint64_t bytes_read_count() const;
|
||||||
|
bool is_remote() const;
|
||||||
|
|
||||||
private:
|
private:
|
||||||
struct layer_state {
|
struct layer_state {
|
||||||
@@ -49,6 +52,7 @@ private:
|
|||||||
|
|
||||||
llama_files files;
|
llama_files files;
|
||||||
std::vector<llama_model_loader::longhaul_source> sources;
|
std::vector<llama_model_loader::longhaul_source> sources;
|
||||||
|
std::vector<llama_model_loader::longhaul_shard> shards;
|
||||||
std::vector<std::vector<const llama_model_loader::longhaul_source *>> sources_by_layer;
|
std::vector<std::vector<const llama_model_loader::longhaul_source *>> sources_by_layer;
|
||||||
std::vector<layer_state> layers;
|
std::vector<layer_state> layers;
|
||||||
size_t n_slots;
|
size_t n_slots;
|
||||||
@@ -84,6 +88,8 @@ private:
|
|||||||
std::condition_variable io_done;
|
std::condition_variable io_done;
|
||||||
size_t io_pending = 0;
|
size_t io_pending = 0;
|
||||||
bool io_stopping = false;
|
bool io_stopping = false;
|
||||||
|
bool remote = false;
|
||||||
|
uint64_t remote_registration_id = 0;
|
||||||
|
|
||||||
bool source_is_direct(const llama_model_loader::longhaul_source & source) const;
|
bool source_is_direct(const llama_model_loader::longhaul_source & source) const;
|
||||||
bool load_plan_sources();
|
bool load_plan_sources();
|
||||||
|
|||||||
@@ -9,7 +9,6 @@
|
|||||||
#include <stdexcept>
|
#include <stdexcept>
|
||||||
#include <cerrno>
|
#include <cerrno>
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <mutex>
|
|
||||||
|
|
||||||
#ifdef __has_include
|
#ifdef __has_include
|
||||||
#if __has_include(<unistd.h>)
|
#if __has_include(<unistd.h>)
|
||||||
@@ -67,12 +66,8 @@ static std::string llama_format_win_err(DWORD err) {
|
|||||||
// llama_file
|
// llama_file
|
||||||
|
|
||||||
struct llama_file::impl {
|
struct llama_file::impl {
|
||||||
bool no_cache = false;
|
|
||||||
|
|
||||||
#if defined(_WIN32)
|
#if defined(_WIN32)
|
||||||
HANDLE fp_win32;
|
HANDLE fp_win32;
|
||||||
std::mutex read_at_mutex;
|
|
||||||
|
|
||||||
std::string GetErrorMessageWin32(DWORD error_code) const {
|
std::string GetErrorMessageWin32(DWORD error_code) const {
|
||||||
std::string ret;
|
std::string ret;
|
||||||
LPSTR lpMsgBuf = NULL;
|
LPSTR lpMsgBuf = NULL;
|
||||||
@@ -438,10 +433,6 @@ void llama_file::read_raw_unsafe(void * ptr, size_t len) { pimpl->read_raw_unsaf
|
|||||||
|
|
||||||
void llama_file::read_at(void * ptr, size_t len, size_t offset) {
|
void llama_file::read_at(void * ptr, size_t len, size_t offset) {
|
||||||
#ifdef _WIN32
|
#ifdef _WIN32
|
||||||
// The Windows FILE pointer is shared by all longhaul I/O workers. Keep the
|
|
||||||
// seek and read together so concurrent jobs cannot change each other's
|
|
||||||
// offsets. POSIX uses pread below and does not need this serialization.
|
|
||||||
std::lock_guard<std::mutex> lock(pimpl->read_at_mutex);
|
|
||||||
seek(offset, SEEK_SET);
|
seek(offset, SEEK_SET);
|
||||||
read_raw(ptr, len);
|
read_raw(ptr, len);
|
||||||
#else
|
#else
|
||||||
@@ -457,13 +448,6 @@ void llama_file::read_at(void * ptr, size_t len, size_t offset) {
|
|||||||
}
|
}
|
||||||
n_read += result;
|
n_read += result;
|
||||||
}
|
}
|
||||||
#if defined(__linux__) && defined(POSIX_FADV_DONTNEED)
|
|
||||||
if (pimpl->no_cache) {
|
|
||||||
// Longhaul owns a bounded expert cache, so avoid retaining a second
|
|
||||||
// long-lived copy of completed reads in the kernel page cache.
|
|
||||||
(void) posix_fadvise(file_id(), (off_t) offset, (off_t) len, POSIX_FADV_DONTNEED);
|
|
||||||
}
|
|
||||||
#endif
|
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -472,12 +456,6 @@ void llama_file::set_no_cache() const {
|
|||||||
if (fcntl(file_id(), F_NOCACHE, 1) != 0) {
|
if (fcntl(file_id(), F_NOCACHE, 1) != 0) {
|
||||||
throw std::runtime_error(format("failed to disable file cache: %s", strerror(errno)));
|
throw std::runtime_error(format("failed to disable file cache: %s", strerror(errno)));
|
||||||
}
|
}
|
||||||
#elif defined(__linux__) && defined(POSIX_FADV_RANDOM)
|
|
||||||
pimpl->no_cache = true;
|
|
||||||
const int err = posix_fadvise(file_id(), 0, 0, POSIX_FADV_RANDOM);
|
|
||||||
if (err != 0) {
|
|
||||||
LLAMA_LOG_WARN("%s: failed to set random-access file advice: %s\n", __func__, strerror(err));
|
|
||||||
}
|
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+10
-21
@@ -594,6 +594,11 @@ llama_model_loader::llama_model_loader(
|
|||||||
|
|
||||||
files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io));
|
files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io));
|
||||||
contexts.emplace_back(ctx);
|
contexts.emplace_back(ctx);
|
||||||
|
longhaul_shards.push_back({
|
||||||
|
std::filesystem::path(fname).filename().string(),
|
||||||
|
files.back()->size(),
|
||||||
|
gguf_get_data_offset(metadata),
|
||||||
|
});
|
||||||
|
|
||||||
// Save tensors data offset of the main file.
|
// Save tensors data offset of the main file.
|
||||||
// For subsidiary files, `meta` tensor data offset must not be used,
|
// For subsidiary files, `meta` tensor data offset must not be used,
|
||||||
@@ -662,6 +667,11 @@ llama_model_loader::llama_model_loader(
|
|||||||
|
|
||||||
files.emplace_back(new llama_file(fname_split, "rb", use_direct_io));
|
files.emplace_back(new llama_file(fname_split, "rb", use_direct_io));
|
||||||
contexts.emplace_back(ctx);
|
contexts.emplace_back(ctx);
|
||||||
|
longhaul_shards.push_back({
|
||||||
|
std::filesystem::path(fname_split).filename().string(),
|
||||||
|
files.back()->size(),
|
||||||
|
gguf_get_data_offset(ctx_gguf.get()),
|
||||||
|
});
|
||||||
|
|
||||||
// Save tensors data offset info of the shard.
|
// Save tensors data offset info of the shard.
|
||||||
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
||||||
@@ -1220,7 +1230,6 @@ struct ggml_tensor * llama_model_loader::create_tensor(
|
|||||||
}
|
}
|
||||||
|
|
||||||
ggml_backend_buffer_type_t buft = nullptr;
|
ggml_backend_buffer_type_t buft = nullptr;
|
||||||
ggml_backend_buffer_type_t requested_override = nullptr;
|
|
||||||
|
|
||||||
// check overrides
|
// check overrides
|
||||||
if (tensor_buft_overrides) {
|
if (tensor_buft_overrides) {
|
||||||
@@ -1228,7 +1237,6 @@ struct ggml_tensor * llama_model_loader::create_tensor(
|
|||||||
for (const auto * overrides = tensor_buft_overrides; overrides->pattern != nullptr; ++overrides) {
|
for (const auto * overrides = tensor_buft_overrides; overrides->pattern != nullptr; ++overrides) {
|
||||||
std::regex pattern(overrides->pattern);
|
std::regex pattern(overrides->pattern);
|
||||||
if (std::regex_search(tensor_name, pattern)) {
|
if (std::regex_search(tensor_name, pattern)) {
|
||||||
requested_override = overrides->buft;
|
|
||||||
if (overrides->buft == ggml_backend_cpu_buffer_type()) {
|
if (overrides->buft == ggml_backend_cpu_buffer_type()) {
|
||||||
// when overriding to a CPU buffer, consider the extra buffer types
|
// when overriding to a CPU buffer, consider the extra buffer types
|
||||||
buft = select_weight_buft(hparams, t_meta, op, buft_list_cpu);
|
buft = select_weight_buft(hparams, t_meta, op, buft_list_cpu);
|
||||||
@@ -1258,25 +1266,6 @@ struct ggml_tensor * llama_model_loader::create_tensor(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Streamed CPU experts are updated one expert slice at a time. CPU
|
|
||||||
// repack and accelerator buffers only accept whole-tensor updates, so
|
|
||||||
// keep these cache tensors in the standard, directly writable layout.
|
|
||||||
if ((flags & TENSOR_LONGHAUL) && buft_list == buft_list_cpu) {
|
|
||||||
auto * cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);
|
|
||||||
if (!cpu_dev) {
|
|
||||||
throw std::runtime_error("no CPU backend found");
|
|
||||||
}
|
|
||||||
ggml_backend_buffer_type_t cpu_buft = ggml_backend_dev_buffer_type(cpu_dev);
|
|
||||||
if (requested_override != nullptr &&
|
|
||||||
requested_override != ggml_backend_cpu_buffer_type() &&
|
|
||||||
requested_override != cpu_buft) {
|
|
||||||
throw std::runtime_error(format(
|
|
||||||
"longhaul CPU expert tensor '%s' has an incompatible buffer override",
|
|
||||||
tn.str().c_str()));
|
|
||||||
}
|
|
||||||
buft = cpu_buft;
|
|
||||||
}
|
|
||||||
|
|
||||||
// avoid using a host buffer when using mmap
|
// avoid using a host buffer when using mmap
|
||||||
auto * buft_dev = ggml_backend_buft_get_device(buft);
|
auto * buft_dev = ggml_backend_buft_get_device(buft);
|
||||||
if (use_mmap && buft_dev && buft == ggml_backend_dev_host_buffer_type(buft_dev)) {
|
if (use_mmap && buft_dev && buft == ggml_backend_dev_host_buffer_type(buft_dev)) {
|
||||||
|
|||||||
@@ -9,6 +9,7 @@
|
|||||||
|
|
||||||
#include "ggml-cpp.h"
|
#include "ggml-cpp.h"
|
||||||
|
|
||||||
|
#include <array>
|
||||||
#include <cstddef>
|
#include <cstddef>
|
||||||
#include <cstring>
|
#include <cstring>
|
||||||
#include <map>
|
#include <map>
|
||||||
@@ -77,6 +78,12 @@ struct llama_model_loader {
|
|||||||
int layer;
|
int layer;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
struct longhaul_shard {
|
||||||
|
std::string name;
|
||||||
|
uint64_t size;
|
||||||
|
uint64_t metadata_size;
|
||||||
|
};
|
||||||
|
|
||||||
int n_kv = 0;
|
int n_kv = 0;
|
||||||
int n_tensors = 0;
|
int n_tensors = 0;
|
||||||
int n_created = 0;
|
int n_created = 0;
|
||||||
@@ -115,6 +122,7 @@ struct llama_model_loader {
|
|||||||
std::vector<std::pair<size_t, size_t>> mmaps_used;
|
std::vector<std::pair<size_t, size_t>> mmaps_used;
|
||||||
size_t longhaul_slots = 0;
|
size_t longhaul_slots = 0;
|
||||||
std::vector<longhaul_source> longhaul_sources;
|
std::vector<longhaul_source> longhaul_sources;
|
||||||
|
std::vector<longhaul_shard> longhaul_shards;
|
||||||
|
|
||||||
// define a comparator for the buft -> ctx map to ensure that the order is well-defined:
|
// define a comparator for the buft -> ctx map to ensure that the order is well-defined:
|
||||||
struct ggml_backend_buft_comparator {
|
struct ggml_backend_buft_comparator {
|
||||||
|
|||||||
@@ -29,7 +29,6 @@ bool llama_model_saver_supports_arch(llm_arch arch) {
|
|||||||
case LLM_ARCH_STEP35:
|
case LLM_ARCH_STEP35:
|
||||||
case LLM_ARCH_MELLUM:
|
case LLM_ARCH_MELLUM:
|
||||||
case LLM_ARCH_LAGUNA:
|
case LLM_ARCH_LAGUNA:
|
||||||
case LLM_ARCH_INKLING:
|
|
||||||
return false;
|
return false;
|
||||||
default:
|
default:
|
||||||
return true;
|
return true;
|
||||||
|
|||||||
+26
-18
@@ -312,8 +312,6 @@ static llama_model * llama_model_mapping(llm_arch arch, const llama_model_params
|
|||||||
return new llama_model_kimi_linear(params);
|
return new llama_model_kimi_linear(params);
|
||||||
case LLM_ARCH_STEP35:
|
case LLM_ARCH_STEP35:
|
||||||
return new llama_model_step35(params);
|
return new llama_model_step35(params);
|
||||||
case LLM_ARCH_INKLING:
|
|
||||||
return new llama_model_inkling(params);
|
|
||||||
default:
|
default:
|
||||||
throw std::runtime_error(std::string("unsupported model architecture: '") + llm_arch_name(arch) + "'");
|
throw std::runtime_error(std::string("unsupported model architecture: '") + llm_arch_name(arch) + "'");
|
||||||
}
|
}
|
||||||
@@ -1041,6 +1039,7 @@ struct llama_model::impl {
|
|||||||
|
|
||||||
std::vector<float> tensor_split_owned;
|
std::vector<float> tensor_split_owned;
|
||||||
std::unique_ptr<llama_longhaul_cache> longhaul;
|
std::unique_ptr<llama_longhaul_cache> longhaul;
|
||||||
|
bool longhaul_remote = false;
|
||||||
};
|
};
|
||||||
|
|
||||||
llama_model::llama_model(const llama_model_params & params) : params(params), pimpl(std::make_unique<impl>()) {
|
llama_model::llama_model(const llama_model_params & params) : params(params), pimpl(std::make_unique<impl>()) {
|
||||||
@@ -1358,8 +1357,8 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
||||||
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA && arch != LLM_ARCH_INKLING) {
|
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA) {
|
||||||
throw std::runtime_error("longhaul currently requires a qwen35moe, laguna, or inkling model");
|
throw std::runtime_error("longhaul currently requires a qwen35moe or laguna model");
|
||||||
}
|
}
|
||||||
if (hparams.n_layer_nextn != 0) {
|
if (hparams.n_layer_nextn != 0) {
|
||||||
throw std::runtime_error("longhaul does not support MTP tensors");
|
throw std::runtime_error("longhaul does not support MTP tensors");
|
||||||
@@ -1367,27 +1366,36 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
|||||||
if (params.check_tensors || params.no_alloc || params.vocab_only) {
|
if (params.check_tensors || params.no_alloc || params.vocab_only) {
|
||||||
throw std::runtime_error("longhaul does not support check-tensors, no-alloc, or vocab-only loading");
|
throw std::runtime_error("longhaul does not support check-tensors, no-alloc, or vocab-only loading");
|
||||||
}
|
}
|
||||||
bool all_cpu = true;
|
|
||||||
bool all_metal = true;
|
bool all_metal = true;
|
||||||
|
bool all_rpc = true;
|
||||||
for (int il = 0; il < n_layer_all; ++il) {
|
for (int il = 0; il < n_layer_all; ++il) {
|
||||||
ggml_backend_dev_t dev = pimpl->dev_layer[il].dev;
|
ggml_backend_dev_t dev = pimpl->dev_layer[il].dev;
|
||||||
const char * backend = ggml_backend_reg_name(ggml_backend_dev_backend_reg(dev));
|
const char * reg_name = ggml_backend_reg_name(ggml_backend_dev_backend_reg(dev));
|
||||||
all_cpu = all_cpu && ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_CPU;
|
|
||||||
all_metal = all_metal &&
|
all_metal = all_metal &&
|
||||||
ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_GPU &&
|
ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_GPU &&
|
||||||
strcmp(backend, "MTL") == 0;
|
strcmp(reg_name, "MTL") == 0;
|
||||||
|
all_rpc = all_rpc &&
|
||||||
|
ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_GPU &&
|
||||||
|
strcmp(reg_name, "RPC") == 0;
|
||||||
}
|
}
|
||||||
if (!all_cpu && !all_metal) {
|
if (!all_metal && !all_rpc) {
|
||||||
throw std::runtime_error(
|
throw std::runtime_error(
|
||||||
"longhaul requires every repeating model layer on CPU or Metal; "
|
"longhaul requires all model layers on local Metal or one RPC device");
|
||||||
"use --n-gpu-layers 0 for CPU mode");
|
|
||||||
}
|
}
|
||||||
#if !defined(__APPLE__)
|
|
||||||
if (all_metal) {
|
if (all_metal) {
|
||||||
throw std::runtime_error("longhaul Metal mode is only supported on macOS");
|
#if !defined(__APPLE__)
|
||||||
}
|
throw std::runtime_error("local longhaul is only supported on macOS");
|
||||||
#endif
|
#endif
|
||||||
LLAMA_LOG_INFO("%s: longhaul using %s expert cache\n", __func__, all_cpu ? "CPU" : "Metal");
|
} else {
|
||||||
|
ggml_backend_dev_t first = pimpl->dev_layer.front().dev;
|
||||||
|
for (int il = 1; il < n_layer_all; ++il) {
|
||||||
|
if (pimpl->dev_layer[il].dev != first) {
|
||||||
|
throw std::runtime_error(
|
||||||
|
"RPC longhaul v1 requires every model layer on one RPC device");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pimpl->longhaul_remote = true;
|
||||||
|
}
|
||||||
ml.configure_longhaul(params.longhaul_cache_bytes, n_expert, n_layer_all);
|
ml.configure_longhaul(params.longhaul_cache_bytes, n_expert, n_layer_all);
|
||||||
if (ml.longhaul_slots < (size_t) n_expert_used) {
|
if (ml.longhaul_slots < (size_t) n_expert_used) {
|
||||||
LLAMA_LOG_WARN("%s: longhaul cache has %zu slots for %lld selected experts; "
|
LLAMA_LOG_WARN("%s: longhaul cache has %zu slots for %lld selected experts; "
|
||||||
@@ -1707,7 +1715,8 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
|||||||
|
|
||||||
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
||||||
pimpl->longhaul = std::make_unique<llama_longhaul_cache>(
|
pimpl->longhaul = std::make_unique<llama_longhaul_cache>(
|
||||||
std::move(ml.files), std::move(ml.longhaul_sources), ml.longhaul_slots, hparams.n_expert, hparams.n_layer_all);
|
std::move(ml.files), std::move(ml.longhaul_sources), std::move(ml.longhaul_shards),
|
||||||
|
ml.longhaul_slots, hparams.n_expert, hparams.n_layer_all, pimpl->longhaul_remote);
|
||||||
}
|
}
|
||||||
|
|
||||||
return true;
|
return true;
|
||||||
@@ -2167,7 +2176,7 @@ llama_memory_i * llama_model::create_memory(const llama_memory_params & params,
|
|||||||
// layer filters, so pick the right one here
|
// layer filters, so pick the right one here
|
||||||
llama_memory_hybrid::layer_filter_cb filter_attn = nullptr;
|
llama_memory_hybrid::layer_filter_cb filter_attn = nullptr;
|
||||||
llama_memory_hybrid::layer_filter_cb filter_recr = nullptr;
|
llama_memory_hybrid::layer_filter_cb filter_recr = nullptr;
|
||||||
if (arch == LLM_ARCH_FALCON_H1 || arch == LLM_ARCH_INKLING) {
|
if (arch == LLM_ARCH_FALCON_H1) {
|
||||||
filter_attn = [&](uint32_t) { return true; };
|
filter_attn = [&](uint32_t) { return true; };
|
||||||
filter_recr = [&](uint32_t) { return true; };
|
filter_recr = [&](uint32_t) { return true; };
|
||||||
} else if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
|
} else if (arch == LLM_ARCH_NEMOTRON_H || arch == LLM_ARCH_NEMOTRON_H_MOE) {
|
||||||
@@ -2508,7 +2517,6 @@ llama_rope_type llama_model_rope_type(const llama_model * model) {
|
|||||||
case LLM_ARCH_NEMOTRON_H:
|
case LLM_ARCH_NEMOTRON_H:
|
||||||
case LLM_ARCH_NEMOTRON_H_MOE:
|
case LLM_ARCH_NEMOTRON_H_MOE:
|
||||||
case LLM_ARCH_KIMI_LINEAR:
|
case LLM_ARCH_KIMI_LINEAR:
|
||||||
case LLM_ARCH_INKLING:
|
|
||||||
return LLAMA_ROPE_TYPE_NONE;
|
return LLAMA_ROPE_TYPE_NONE;
|
||||||
|
|
||||||
// use what we call a normal RoPE, operating on pairs of consecutive head values
|
// use what we call a normal RoPE, operating on pairs of consecutive head values
|
||||||
|
|||||||
@@ -534,15 +534,6 @@ struct llama_layer {
|
|||||||
struct llama_layer_shortconv shortconv;
|
struct llama_layer_shortconv shortconv;
|
||||||
|
|
||||||
struct llama_layer_nextn nextn;
|
struct llama_layer_nextn nextn;
|
||||||
|
|
||||||
// Inkling relative attention and short-convolution weights.
|
|
||||||
struct ggml_tensor * wr = nullptr;
|
|
||||||
struct ggml_tensor * attn_rel_proj = nullptr;
|
|
||||||
struct ggml_tensor * shortconv_k = nullptr;
|
|
||||||
struct ggml_tensor * shortconv_v = nullptr;
|
|
||||||
struct ggml_tensor * shortconv_attn = nullptr;
|
|
||||||
struct ggml_tensor * shortconv_mlp = nullptr;
|
|
||||||
struct ggml_tensor * ffn_gscale = nullptr;
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct llama_device {
|
struct llama_device {
|
||||||
|
|||||||
+1
-17
@@ -434,13 +434,6 @@ struct llm_tokenizer_bpe : llm_tokenizer {
|
|||||||
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))*((?=[\\p{L}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))+((?=[\\p{L}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))*((?=[\\p{L}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}])([^a-z]))+((?=[\\p{L}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
||||||
};
|
};
|
||||||
break;
|
break;
|
||||||
case LLAMA_VOCAB_PRE_TYPE_INKLING:
|
|
||||||
// o200k-family expression with combining marks included in
|
|
||||||
// letter lookaheads, as used by the Inkling tokenizer.
|
|
||||||
regex_exprs = {
|
|
||||||
"[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}\\p{M}])([^a-z]))*((?=[\\p{L}\\p{M}])([^A-Z]))+(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|[^\\r\\n\\p{L}\\p{N}]?((?=[\\p{L}\\p{M}])([^a-z]))+((?=[\\p{L}\\p{M}])([^A-Z]))*(?:'[sS]|'[tT]|'[rR][eE]|'[vV][eE]|'[mM]|'[lL][lL]|'[dD])?|\\p{N}{1,3}| ?[^\\s\\p{L}\\p{N}]+[\\r\\n/]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
|
||||||
};
|
|
||||||
break;
|
|
||||||
case LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI:
|
case LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI:
|
||||||
// Same lookaheads as GPT4O but with \p{M} added so combining marks
|
// Same lookaheads as GPT4O but with \p{M} added so combining marks
|
||||||
// (diacritics) attach to their base letters. Avoids excessive
|
// (diacritics) attach to their base letters. Avoids excessive
|
||||||
@@ -2331,10 +2324,6 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
|||||||
tokenizer_pre == "talkie") {
|
tokenizer_pre == "talkie") {
|
||||||
pre_type = LLAMA_VOCAB_PRE_TYPE_GPT4O;
|
pre_type = LLAMA_VOCAB_PRE_TYPE_GPT4O;
|
||||||
clean_spaces = false;
|
clean_spaces = false;
|
||||||
} else if (
|
|
||||||
tokenizer_pre == "inkling") {
|
|
||||||
pre_type = LLAMA_VOCAB_PRE_TYPE_INKLING;
|
|
||||||
clean_spaces = false;
|
|
||||||
} else if (
|
} else if (
|
||||||
tokenizer_pre == "granite-embed-multi-97m") {
|
tokenizer_pre == "granite-embed-multi-97m") {
|
||||||
pre_type = LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI;
|
pre_type = LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI;
|
||||||
@@ -2618,12 +2607,7 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
|||||||
if (suppress_idx != -1) {
|
if (suppress_idx != -1) {
|
||||||
const int n = gguf_get_arr_n(ctx, suppress_idx);
|
const int n = gguf_get_arr_n(ctx, suppress_idx);
|
||||||
const int32_t * data = (const int32_t *) gguf_get_arr_data(ctx, suppress_idx);
|
const int32_t * data = (const int32_t *) gguf_get_arr_data(ctx, suppress_idx);
|
||||||
suppress_tokens.reserve(n);
|
suppress_tokens.assign(data, data + n);
|
||||||
for (int i = 0; i < n; ++i) {
|
|
||||||
if (data[i] >= 0 && data[i] < (int32_t) id_to_token.size()) {
|
|
||||||
suppress_tokens.push_back(data[i]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -65,7 +65,6 @@ enum llama_vocab_pre_type {
|
|||||||
LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI = 54,
|
LLAMA_VOCAB_PRE_TYPE_GRANITE_EMB_MULTI = 54,
|
||||||
LLAMA_VOCAB_PRE_TYPE_MELLUM2 = 55,
|
LLAMA_VOCAB_PRE_TYPE_MELLUM2 = 55,
|
||||||
LLAMA_VOCAB_PRE_TYPE_LAGUNA = 56,
|
LLAMA_VOCAB_PRE_TYPE_LAGUNA = 56,
|
||||||
LLAMA_VOCAB_PRE_TYPE_INKLING = 57,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
struct LLM_KV;
|
struct LLM_KV;
|
||||||
|
|||||||
+13
-8
@@ -263,16 +263,21 @@ static bool llama_prepare_model_devices(const llama_model_params & params, llama
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// add RPC servers at the front of the list to minimize network transfers
|
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL && !rpc_servers.empty()) {
|
||||||
model->devices.insert(model->devices.begin(), rpc_servers.begin(), rpc_servers.end());
|
// RPC longhaul v1 keeps the entire paging domain on one remote device.
|
||||||
|
model->devices.push_back(rpc_servers.front());
|
||||||
|
} else {
|
||||||
|
// add RPC servers at the front of the list to minimize network transfers
|
||||||
|
model->devices.insert(model->devices.begin(), rpc_servers.begin(), rpc_servers.end());
|
||||||
|
|
||||||
// add GPUs
|
// add GPUs
|
||||||
model->devices.insert(model->devices.end(), gpus.begin(), gpus.end());
|
model->devices.insert(model->devices.end(), gpus.begin(), gpus.end());
|
||||||
|
|
||||||
// add integrated GPUs only if no discrete GPUs were found
|
// add integrated GPUs only if no discrete GPUs were found
|
||||||
// (RPC servers do not count, otherwise the local iGPU would be dropped on iGPU+RPC setups)
|
// (RPC servers do not count, otherwise the local iGPU would be dropped on iGPU+RPC setups)
|
||||||
if (gpus.empty()) {
|
if (gpus.empty()) {
|
||||||
model->devices.insert(model->devices.end(), igpus.begin(), igpus.end());
|
model->devices.insert(model->devices.end(), igpus.begin(), igpus.end());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,489 +0,0 @@
|
|||||||
// Inkling: relative-position attention with per-layer packed short-convolution
|
|
||||||
// state and a mixture of routed and shared experts.
|
|
||||||
|
|
||||||
#include "models.h"
|
|
||||||
|
|
||||||
#include "../llama-memory-hybrid-iswa.h"
|
|
||||||
#include "../llama-memory-recurrent.h"
|
|
||||||
|
|
||||||
#include <algorithm>
|
|
||||||
#include <cmath>
|
|
||||||
|
|
||||||
void llama_model_inkling::load_arch_hparams(llama_model_loader & ml) {
|
|
||||||
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
|
|
||||||
|
|
||||||
ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, hparams.n_ff_exp);
|
|
||||||
ml.get_key(LLM_KV_EXPERT_SHARED_COUNT, hparams.n_expert_shared);
|
|
||||||
ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE, hparams.expert_weights_scale);
|
|
||||||
ml.get_key(LLM_KV_EXPERT_GATING_FUNC, hparams.expert_gating_func, false);
|
|
||||||
|
|
||||||
ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa);
|
|
||||||
hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;
|
|
||||||
ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, hparams.is_swa_impl, hparams.n_layer());
|
|
||||||
|
|
||||||
for (uint32_t il = 0; il < hparams.n_layer(); ++il) {
|
|
||||||
hparams.is_recr_impl[il] = 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
ml.get_key(LLM_KV_INKLING_D_REL, hparams.inkling_d_rel);
|
|
||||||
ml.get_key(LLM_KV_INKLING_REL_EXTENT, hparams.inkling_rel_extent);
|
|
||||||
ml.get_key(LLM_KV_INKLING_REL_EXTENT_SWA, hparams.inkling_rel_extent_swa);
|
|
||||||
ml.get_key(LLM_KV_INKLING_SHORTCONV_KERNEL, hparams.n_shortconv_l_cache);
|
|
||||||
ml.get_key(LLM_KV_INKLING_DENSE_BLOCK_COUNT, hparams.n_layer_dense_lead);
|
|
||||||
|
|
||||||
float logit_scale_denom = 0.0f;
|
|
||||||
ml.get_key(LLM_KV_INKLING_LOGIT_SCALE_DENOM, logit_scale_denom);
|
|
||||||
GGML_ASSERT(logit_scale_denom != 0.0f);
|
|
||||||
hparams.f_logit_scale = 1.0f/logit_scale_denom;
|
|
||||||
|
|
||||||
ml.get_key(LLM_KV_INKLING_LOG_SCALING_N_FLOOR, hparams.inkling_log_n_floor, false);
|
|
||||||
ml.get_key(LLM_KV_INKLING_LOG_SCALING_ALPHA, hparams.inkling_log_alpha, false);
|
|
||||||
ml.get_key(LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, hparams.inkling_unpadded_n_vocab, false);
|
|
||||||
|
|
||||||
GGML_ASSERT(hparams.n_shortconv_l_cache > 1);
|
|
||||||
GGML_ASSERT(hparams.inkling_d_rel > 0);
|
|
||||||
GGML_ASSERT(hparams.inkling_rel_extent > 0 && hparams.inkling_rel_extent_swa > 0);
|
|
||||||
|
|
||||||
// Four streams per layer: K, V, attention output, and MLP output.
|
|
||||||
const uint32_t d_conv = hparams.n_shortconv_l_cache - 1;
|
|
||||||
hparams.n_embd_r_impl = d_conv *
|
|
||||||
(hparams.n_embd_k_gqa_max() + hparams.n_embd_v_gqa_max() + 2*hparams.n_embd);
|
|
||||||
|
|
||||||
type = LLM_TYPE_UNKNOWN;
|
|
||||||
}
|
|
||||||
|
|
||||||
void llama_model_inkling::load_arch_tensors(llama_model_loader &) {
|
|
||||||
LLAMA_LOAD_LOCALS;
|
|
||||||
|
|
||||||
const int64_t head_dim = hparams.n_embd_head_k();
|
|
||||||
const int64_t d_rel = hparams.inkling_d_rel;
|
|
||||||
const int64_t K = hparams.n_shortconv_l_cache;
|
|
||||||
const int64_t n_ff_exp = hparams.n_ff_exp;
|
|
||||||
const int64_t n_shexp = hparams.n_expert_shared;
|
|
||||||
const int expert_flags = params.load_mode == LLAMA_LOAD_MODE_LONGHAUL ? TENSOR_LONGHAUL : 0;
|
|
||||||
|
|
||||||
tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);
|
|
||||||
tok_norm = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD_NORM, "weight", 0), {n_embd}, 0);
|
|
||||||
output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);
|
|
||||||
output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, 0);
|
|
||||||
|
|
||||||
for (int i = 0; i < n_layer; ++i) {
|
|
||||||
auto & layer = layers[i];
|
|
||||||
|
|
||||||
const int64_t n_head_kv_i = hparams.n_head_kv(i);
|
|
||||||
const int64_t kvw = n_head_kv_i*head_dim;
|
|
||||||
const int64_t rel_extent = hparams.is_swa(i)
|
|
||||||
? hparams.inkling_rel_extent_swa
|
|
||||||
: hparams.inkling_rel_extent;
|
|
||||||
|
|
||||||
layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);
|
|
||||||
|
|
||||||
layer.wq = create_tensor(tn(LLM_TENSOR_ATTN_Q, "weight", i), {n_embd, n_head*head_dim}, 0);
|
|
||||||
layer.wk = create_tensor(tn(LLM_TENSOR_ATTN_K, "weight", i), {n_embd, kvw}, 0);
|
|
||||||
layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V, "weight", i), {n_embd, kvw}, 0);
|
|
||||||
layer.wr = create_tensor(tn(LLM_TENSOR_ATTN_R, "weight", i), {n_embd, n_head*d_rel}, 0);
|
|
||||||
layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_head*head_dim, n_embd}, 0);
|
|
||||||
|
|
||||||
layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), {head_dim}, 0);
|
|
||||||
layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), {head_dim}, 0);
|
|
||||||
|
|
||||||
layer.attn_rel_proj = create_tensor(
|
|
||||||
tn(LLM_TENSOR_ATTN_REL_PROJ, "weight", i), {rel_extent, d_rel}, 0);
|
|
||||||
|
|
||||||
layer.shortconv_k = create_tensor(tn(LLM_TENSOR_SHORTCONV_K, "weight", i), {K, kvw}, 0);
|
|
||||||
layer.shortconv_v = create_tensor(tn(LLM_TENSOR_SHORTCONV_V, "weight", i), {K, kvw}, 0);
|
|
||||||
layer.shortconv_attn = create_tensor(tn(LLM_TENSOR_SHORTCONV_ATTN, "weight", i), {K, n_embd}, 0);
|
|
||||||
layer.shortconv_mlp = create_tensor(tn(LLM_TENSOR_SHORTCONV_MLP, "weight", i), {K, n_embd}, 0);
|
|
||||||
|
|
||||||
layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);
|
|
||||||
layer.ffn_gscale = create_tensor(tn(LLM_TENSOR_FFN_GSCALE, "weight", i), {1}, 0);
|
|
||||||
|
|
||||||
if (i < (int) hparams.n_layer_dense_lead) {
|
|
||||||
const int64_t n_ff_i = hparams.n_ff(i);
|
|
||||||
layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff_i}, 0);
|
|
||||||
layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff_i}, 0);
|
|
||||||
layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {n_ff_i, n_embd}, 0);
|
|
||||||
} else {
|
|
||||||
GGML_ASSERT(n_expert > 0 && n_expert_used > 0 && n_shexp > 0);
|
|
||||||
|
|
||||||
// The final rows are shared-expert sink logits.
|
|
||||||
layer.ffn_gate_inp = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_GATE_INP, "weight", i), {n_embd, n_expert + n_shexp}, 0);
|
|
||||||
layer.ffn_exp_probs_b = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias", i), {n_expert}, 0);
|
|
||||||
|
|
||||||
layer.ffn_gate_exps = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i),
|
|
||||||
{n_embd, n_ff_exp, n_expert}, expert_flags);
|
|
||||||
layer.ffn_up_exps = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_UP_EXPS, "weight", i),
|
|
||||||
{n_embd, n_ff_exp, n_expert}, expert_flags);
|
|
||||||
layer.ffn_down_exps = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i),
|
|
||||||
{n_ff_exp, n_embd, n_expert}, expert_flags);
|
|
||||||
|
|
||||||
// Shared experts stay resident and are never part of the longhaul cache.
|
|
||||||
layer.ffn_gate_shexp = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_GATE_SHEXPS, "weight", i),
|
|
||||||
{n_embd, n_ff_exp, n_shexp}, 0);
|
|
||||||
layer.ffn_up_shexp = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_UP_SHEXPS, "weight", i),
|
|
||||||
{n_embd, n_ff_exp, n_shexp}, 0);
|
|
||||||
layer.ffn_down_shexp = create_tensor(
|
|
||||||
tn(LLM_TENSOR_FFN_DOWN_SHEXPS, "weight", i),
|
|
||||||
{n_ff_exp, n_embd, n_shexp}, 0);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
class llm_graph_input_inkling : public llm_graph_input_i {
|
|
||||||
public:
|
|
||||||
llm_graph_input_inkling(
|
|
||||||
const llama_hparams & hparams,
|
|
||||||
const llama_memory_hybrid_iswa_context * mctx) :
|
|
||||||
hparams(hparams), mctx(mctx) {
|
|
||||||
}
|
|
||||||
|
|
||||||
void set_input(const llama_ubatch * ubatch) override {
|
|
||||||
if (tau) {
|
|
||||||
GGML_ASSERT(ggml_backend_buffer_is_host(tau->buffer));
|
|
||||||
float * data = (float *) tau->data;
|
|
||||||
const float n_floor = (float) hparams.inkling_log_n_floor;
|
|
||||||
const float alpha = hparams.inkling_log_alpha;
|
|
||||||
for (int64_t i = 0; i < (int64_t) ubatch->n_tokens; ++i) {
|
|
||||||
const float eff = (float) (ubatch->pos[i] + 1)/n_floor;
|
|
||||||
data[i] = 1.0f + alpha*logf(std::max(eff, 1.0f));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (rel_idx) {
|
|
||||||
mctx->get_attn()->get_base()->set_input_pos_rel_flat(
|
|
||||||
rel_idx, ubatch, hparams.inkling_rel_extent);
|
|
||||||
}
|
|
||||||
if (rel_idx_swa) {
|
|
||||||
mctx->get_attn()->get_swa()->set_input_pos_rel_flat(
|
|
||||||
rel_idx_swa, ubatch, hparams.inkling_rel_extent_swa);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (vocab_mask) {
|
|
||||||
GGML_ASSERT(ggml_backend_buffer_is_host(vocab_mask->buffer));
|
|
||||||
float * data = (float *) vocab_mask->data;
|
|
||||||
for (int64_t id = 0; id < vocab_mask->ne[0]; ++id) {
|
|
||||||
data[id] = id < (int64_t) hparams.inkling_unpadded_n_vocab ? 0.0f : -INFINITY;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (shexp_idx) {
|
|
||||||
GGML_ASSERT(ggml_backend_buffer_is_host(shexp_idx->buffer));
|
|
||||||
int32_t * data = (int32_t *) shexp_idx->data;
|
|
||||||
for (int64_t j = 0; j < shexp_idx->ne[1]; ++j) {
|
|
||||||
for (int64_t s = 0; s < shexp_idx->ne[0]; ++s) {
|
|
||||||
data[j*shexp_idx->ne[0] + s] = (int32_t) s;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * tau = nullptr;
|
|
||||||
ggml_tensor * rel_idx = nullptr;
|
|
||||||
ggml_tensor * rel_idx_swa = nullptr;
|
|
||||||
ggml_tensor * vocab_mask = nullptr;
|
|
||||||
ggml_tensor * shexp_idx = nullptr;
|
|
||||||
|
|
||||||
private:
|
|
||||||
const llama_hparams hparams;
|
|
||||||
const llama_memory_hybrid_iswa_context * mctx;
|
|
||||||
};
|
|
||||||
|
|
||||||
std::unique_ptr<llm_graph_context> llama_model_inkling::build_arch_graph(
|
|
||||||
const llm_graph_params & params) const {
|
|
||||||
return std::make_unique<graph>(*this, params);
|
|
||||||
}
|
|
||||||
|
|
||||||
llama_model_inkling::graph::graph(
|
|
||||||
const llama_model & model,
|
|
||||||
const llm_graph_params & params) :
|
|
||||||
llm_graph_context(params) {
|
|
||||||
const int64_t head_dim = hparams.n_embd_head_k();
|
|
||||||
const int64_t d_rel = hparams.inkling_d_rel;
|
|
||||||
const int64_t d_conv = hparams.n_shortconv_l_cache - 1;
|
|
||||||
const int64_t n_embd_r = hparams.n_embd_r();
|
|
||||||
const int64_t kw_max = hparams.n_embd_k_gqa_max();
|
|
||||||
const int64_t vw_max = hparams.n_embd_v_gqa_max();
|
|
||||||
|
|
||||||
const int64_t off_k = 0;
|
|
||||||
const int64_t off_v = d_conv*kw_max;
|
|
||||||
const int64_t off_attn = d_conv*(kw_max + vw_max);
|
|
||||||
const int64_t off_mlp = d_conv*(kw_max + vw_max + n_embd);
|
|
||||||
|
|
||||||
const auto * mctx_hyb = static_cast<const llama_memory_hybrid_iswa_context *>(mctx);
|
|
||||||
const auto * mctx_recr = mctx_hyb->get_recr();
|
|
||||||
const auto * mctx_attn = mctx_hyb->get_attn();
|
|
||||||
const uint32_t kv_head = mctx_recr->get_head();
|
|
||||||
|
|
||||||
const int64_t n_seq_tokens = ubatch.n_seq_tokens;
|
|
||||||
const int64_t n_seqs = ubatch.n_seqs;
|
|
||||||
GGML_ASSERT(n_seqs != 0);
|
|
||||||
GGML_ASSERT(ubatch.equal_seqs());
|
|
||||||
GGML_ASSERT(ubatch.n_tokens == n_seq_tokens*n_seqs);
|
|
||||||
|
|
||||||
bool has_global = false;
|
|
||||||
bool has_local = false;
|
|
||||||
for (int il = 0; il < n_layer; ++il) {
|
|
||||||
has_local |= hparams.is_swa(il);
|
|
||||||
has_global |= !hparams.is_swa(il);
|
|
||||||
}
|
|
||||||
|
|
||||||
auto inp = std::make_unique<llm_graph_input_inkling>(hparams, mctx_hyb);
|
|
||||||
if (hparams.inkling_log_n_floor > 0 && has_global) {
|
|
||||||
inp->tau = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 1, 1, n_tokens);
|
|
||||||
ggml_set_input(inp->tau);
|
|
||||||
ggml_set_name(inp->tau, "inkling_tau");
|
|
||||||
}
|
|
||||||
if (has_global) {
|
|
||||||
inp->rel_idx = ggml_new_tensor_2d(
|
|
||||||
ctx0, GGML_TYPE_I32, mctx_attn->get_base()->get_n_kv(), n_tokens);
|
|
||||||
ggml_set_input(inp->rel_idx);
|
|
||||||
ggml_set_name(inp->rel_idx, "inkling_rel_idx");
|
|
||||||
}
|
|
||||||
if (has_local) {
|
|
||||||
inp->rel_idx_swa = ggml_new_tensor_2d(
|
|
||||||
ctx0, GGML_TYPE_I32, mctx_attn->get_swa()->get_n_kv(), n_tokens);
|
|
||||||
ggml_set_input(inp->rel_idx_swa);
|
|
||||||
ggml_set_name(inp->rel_idx_swa, "inkling_rel_idx_swa");
|
|
||||||
}
|
|
||||||
|
|
||||||
const int64_t n_vocab = model.vocab.n_tokens();
|
|
||||||
if (!cparams.embeddings &&
|
|
||||||
hparams.inkling_unpadded_n_vocab > 0 &&
|
|
||||||
(int64_t) hparams.inkling_unpadded_n_vocab < n_vocab) {
|
|
||||||
inp->vocab_mask = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_vocab);
|
|
||||||
ggml_set_input(inp->vocab_mask);
|
|
||||||
ggml_set_name(inp->vocab_mask, "inkling_vocab_mask");
|
|
||||||
}
|
|
||||||
if (hparams.n_expert_shared > 0 && (uint32_t) n_layer > hparams.n_layer_dense_lead) {
|
|
||||||
inp->shexp_idx = ggml_new_tensor_2d(
|
|
||||||
ctx0, GGML_TYPE_I32, hparams.n_expert_shared, n_tokens);
|
|
||||||
ggml_set_input(inp->shexp_idx);
|
|
||||||
ggml_set_name(inp->shexp_idx, "inkling_shexp_idx");
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * tau = inp->tau;
|
|
||||||
ggml_tensor * rel_idx = inp->rel_idx;
|
|
||||||
ggml_tensor * rel_idx_swa = inp->rel_idx_swa;
|
|
||||||
ggml_tensor * vocab_mask = inp->vocab_mask;
|
|
||||||
ggml_tensor * shexp_idx = inp->shexp_idx;
|
|
||||||
res->add_input(std::move(inp));
|
|
||||||
|
|
||||||
auto * inp_hybrid = build_inp_mem_hybrid_iswa();
|
|
||||||
ggml_tensor * conv_rs_cur = nullptr;
|
|
||||||
|
|
||||||
auto build_sconv = [&](ggml_tensor * x2d, ggml_tensor * kernel, int64_t off, int il) {
|
|
||||||
const int64_t w = x2d->ne[0];
|
|
||||||
ggml_tensor * x3 = ggml_reshape_3d(ctx0, x2d, w, n_seq_tokens, n_seqs);
|
|
||||||
ggml_tensor * xt = ggml_transpose(ctx0, x3);
|
|
||||||
|
|
||||||
ggml_tensor * conv_state = mctx_recr->get_r_l(il);
|
|
||||||
GGML_ASSERT(conv_rs_cur != nullptr);
|
|
||||||
const size_t sz = ggml_element_size(conv_rs_cur);
|
|
||||||
|
|
||||||
ggml_tensor * state = ggml_view_3d(
|
|
||||||
ctx0, conv_rs_cur, d_conv, w, n_seqs,
|
|
||||||
d_conv*sz, conv_rs_cur->nb[1], off*sz);
|
|
||||||
ggml_tensor * sx = ggml_concat(ctx0, state, xt, 0);
|
|
||||||
|
|
||||||
ggml_tensor * new_state = ggml_view_3d(
|
|
||||||
ctx0, sx, d_conv, w, n_seqs,
|
|
||||||
sx->nb[1], sx->nb[2], (sx->ne[0] - d_conv)*sx->nb[0]);
|
|
||||||
ggml_tensor * state_dst = ggml_view_3d(
|
|
||||||
ctx0, conv_state, d_conv, w, n_seqs,
|
|
||||||
d_conv*sz, n_embd_r*sz, (kv_head*n_embd_r + off)*sz);
|
|
||||||
ggml_build_forward_expand(gf, ggml_cpy(ctx0, new_state, state_dst));
|
|
||||||
|
|
||||||
ggml_tensor * conv_out = ggml_ssm_conv(ctx0, sx, kernel);
|
|
||||||
return ggml_reshape_2d(ctx0, ggml_add(ctx0, x3, conv_out), w, n_seq_tokens*n_seqs);
|
|
||||||
};
|
|
||||||
|
|
||||||
auto build_attn_block = [&](ggml_tensor * cur, int il) {
|
|
||||||
const auto & layer = model.layers[il];
|
|
||||||
const bool is_swa = hparams.is_swa(il);
|
|
||||||
const int64_t n_head_kv = hparams.n_head_kv(il);
|
|
||||||
const int64_t rel_extent = is_swa
|
|
||||||
? hparams.inkling_rel_extent_swa
|
|
||||||
: hparams.inkling_rel_extent;
|
|
||||||
|
|
||||||
ggml_tensor * q = build_lora_mm(layer.wq, cur);
|
|
||||||
ggml_tensor * k = build_lora_mm(layer.wk, cur);
|
|
||||||
ggml_tensor * v = build_lora_mm(layer.wv, cur);
|
|
||||||
ggml_tensor * r = build_lora_mm(layer.wr, cur);
|
|
||||||
|
|
||||||
k = build_sconv(k, layer.shortconv_k, off_k, il);
|
|
||||||
v = build_sconv(v, layer.shortconv_v, off_v, il);
|
|
||||||
|
|
||||||
q = ggml_reshape_3d(ctx0, q, head_dim, n_head, n_tokens);
|
|
||||||
k = ggml_reshape_3d(ctx0, k, head_dim, n_head_kv, n_tokens);
|
|
||||||
v = ggml_reshape_3d(ctx0, v, head_dim, n_head_kv, n_tokens);
|
|
||||||
q = build_norm(q, layer.attn_q_norm, nullptr, LLM_NORM_RMS, il);
|
|
||||||
k = build_norm(k, layer.attn_k_norm, nullptr, LLM_NORM_RMS, il);
|
|
||||||
|
|
||||||
if (tau && !is_swa) {
|
|
||||||
q = ggml_mul(ctx0, q, tau);
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * r2 = ggml_reshape_2d(ctx0, r, d_rel, n_head*n_tokens);
|
|
||||||
ggml_tensor * proj = ggml_cont(ctx0, ggml_transpose(ctx0, layer.attn_rel_proj));
|
|
||||||
ggml_tensor * rel = ggml_mul_mat(ctx0, proj, r2);
|
|
||||||
ggml_mul_mat_set_prec(rel, GGML_PREC_F32);
|
|
||||||
rel = ggml_reshape_3d(ctx0, rel, rel_extent, n_head, n_tokens);
|
|
||||||
if (tau && !is_swa) {
|
|
||||||
rel = ggml_mul(ctx0, rel, tau);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Scale Q rather than build_attn's joint logits so the relative term
|
|
||||||
// remains unscaled, matching the reference implementation.
|
|
||||||
q = ggml_scale(ctx0, q, 1.0f/(float) head_dim);
|
|
||||||
|
|
||||||
rel = ggml_pad(ctx0, rel, 1, 0, 0, 0);
|
|
||||||
rel = ggml_cont(ctx0, ggml_permute(ctx0, rel, 1, 0, 2, 3));
|
|
||||||
rel = ggml_reshape_2d(ctx0, rel, n_head, (rel_extent + 1)*n_tokens);
|
|
||||||
|
|
||||||
ggml_tensor * idx = is_swa ? rel_idx_swa : rel_idx;
|
|
||||||
GGML_ASSERT(idx != nullptr);
|
|
||||||
const int64_t n_kv = idx->ne[0];
|
|
||||||
ggml_tensor * idx1 = ggml_reshape_1d(ctx0, idx, n_kv*n_tokens);
|
|
||||||
ggml_tensor * kq_b = ggml_get_rows(ctx0, rel, idx1);
|
|
||||||
kq_b = ggml_reshape_3d(ctx0, kq_b, n_head, n_kv, n_tokens);
|
|
||||||
kq_b = ggml_cont(ctx0, ggml_permute(ctx0, kq_b, 2, 0, 1, 3));
|
|
||||||
|
|
||||||
auto * inp_attn = inp_hybrid->get_attn();
|
|
||||||
const int64_t n_stream =
|
|
||||||
(is_swa ? inp_attn->get_kq_mask_swa() : inp_attn->get_kq_mask())->ne[3];
|
|
||||||
if (n_stream > 1) {
|
|
||||||
kq_b = ggml_view_4d(
|
|
||||||
ctx0, kq_b, n_kv, n_tokens/n_stream, n_head, n_stream,
|
|
||||||
kq_b->nb[1], kq_b->nb[2], (n_tokens/n_stream)*kq_b->nb[1], 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
return build_attn(
|
|
||||||
inp_attn, layer.wo, nullptr, nullptr,
|
|
||||||
q, k, v, kq_b, nullptr, nullptr, 1.0f, il);
|
|
||||||
};
|
|
||||||
|
|
||||||
auto build_dense_ffn = [&](ggml_tensor * cur, int il) {
|
|
||||||
cur = build_ffn(
|
|
||||||
cur,
|
|
||||||
model.layers[il].ffn_up, nullptr, nullptr,
|
|
||||||
model.layers[il].ffn_gate, nullptr, nullptr,
|
|
||||||
model.layers[il].ffn_down, nullptr, nullptr,
|
|
||||||
nullptr, LLM_FFN_SILU, LLM_FFN_PAR, il);
|
|
||||||
return ggml_mul(ctx0, cur, model.layers[il].ffn_gscale);
|
|
||||||
};
|
|
||||||
|
|
||||||
auto build_moe = [&](ggml_tensor * cur, int il) {
|
|
||||||
const auto & layer = model.layers[il];
|
|
||||||
const int64_t n_shexp = hparams.n_expert_shared;
|
|
||||||
|
|
||||||
ggml_tensor * logits = build_lora_mm(layer.ffn_gate_inp, cur);
|
|
||||||
ggml_mul_mat_set_prec(logits, GGML_PREC_F32);
|
|
||||||
const size_t lsz = ggml_element_size(logits);
|
|
||||||
ggml_tensor * routed = ggml_cont(ctx0, ggml_view_2d(
|
|
||||||
ctx0, logits, n_expert, n_tokens, logits->nb[1], 0));
|
|
||||||
ggml_tensor * shared_logits = ggml_view_2d(
|
|
||||||
ctx0, logits, n_shexp, n_tokens, logits->nb[1], n_expert*lsz);
|
|
||||||
|
|
||||||
ggml_tensor * scores = ggml_add(
|
|
||||||
ctx0, ggml_sigmoid(ctx0, routed), layer.ffn_exp_probs_b);
|
|
||||||
ggml_tensor * selected = ggml_argsort_top_k(
|
|
||||||
ctx0, scores, n_expert_used);
|
|
||||||
|
|
||||||
ggml_tensor * routed3 = ggml_reshape_3d(ctx0, routed, 1, n_expert, n_tokens);
|
|
||||||
ggml_tensor * topk_logits = ggml_reshape_2d(
|
|
||||||
ctx0, ggml_get_rows(ctx0, routed3, selected), n_expert_used, n_tokens);
|
|
||||||
ggml_tensor * all_logits = ggml_concat(ctx0, topk_logits, shared_logits, 0);
|
|
||||||
|
|
||||||
ggml_tensor * w = ggml_neg(
|
|
||||||
ctx0, ggml_softplus(ctx0, ggml_neg(ctx0, all_logits)));
|
|
||||||
w = ggml_soft_max(ctx0, w);
|
|
||||||
w = ggml_scale(ctx0, w, hparams.expert_weights_scale);
|
|
||||||
w = ggml_mul(ctx0, w, layer.ffn_gscale);
|
|
||||||
|
|
||||||
const size_t wsz = ggml_element_size(w);
|
|
||||||
ggml_tensor * weights = ggml_cont(ctx0, ggml_view_2d(
|
|
||||||
ctx0, w, n_expert_used, n_tokens, w->nb[1], 0));
|
|
||||||
weights = ggml_reshape_3d(ctx0, weights, 1, n_expert_used, n_tokens);
|
|
||||||
|
|
||||||
ggml_tensor * moe_out = build_moe_experts(
|
|
||||||
cur, selected, weights,
|
|
||||||
layer.ffn_up_exps, layer.ffn_gate_exps, layer.ffn_down_exps,
|
|
||||||
n_expert_used, LLM_FFN_SILU, il);
|
|
||||||
|
|
||||||
GGML_ASSERT(shexp_idx != nullptr);
|
|
||||||
ggml_tensor * xr = ggml_reshape_3d(ctx0, cur, n_embd, 1, n_tokens);
|
|
||||||
ggml_tensor * gs = build_lora_mm_id(layer.ffn_gate_shexp, xr, shexp_idx);
|
|
||||||
ggml_tensor * us = build_lora_mm_id(layer.ffn_up_shexp, xr, shexp_idx);
|
|
||||||
ggml_tensor * hs = ggml_swiglu_split(ctx0, gs, us);
|
|
||||||
|
|
||||||
ggml_tensor * gammas = ggml_cont(ctx0, ggml_view_2d(
|
|
||||||
ctx0, w, n_shexp, n_tokens, w->nb[1], n_expert_used*wsz));
|
|
||||||
hs = ggml_mul(ctx0, hs, ggml_reshape_3d(ctx0, gammas, 1, n_shexp, n_tokens));
|
|
||||||
ggml_tensor * ds = build_lora_mm_id(layer.ffn_down_shexp, hs, shexp_idx);
|
|
||||||
for (int64_t s = 0; s < n_shexp; ++s) {
|
|
||||||
ggml_tensor * e = ggml_view_2d(
|
|
||||||
ctx0, ds, n_embd, n_tokens, ds->nb[2], s*ds->nb[1]);
|
|
||||||
moe_out = ggml_add(ctx0, moe_out, e);
|
|
||||||
}
|
|
||||||
return moe_out;
|
|
||||||
};
|
|
||||||
|
|
||||||
ggml_tensor * cur = build_inp_embd(model.tok_embd);
|
|
||||||
if (ubatch.token) {
|
|
||||||
cur = build_norm(cur, model.tok_norm, nullptr, LLM_NORM_RMS, -1);
|
|
||||||
}
|
|
||||||
ggml_build_forward_expand(gf, cur);
|
|
||||||
|
|
||||||
for (int il = 0; il < n_layer; ++il) {
|
|
||||||
conv_rs_cur = build_rs(
|
|
||||||
inp_hybrid->get_recr(), mctx_recr->get_r_l(il), n_embd_r, n_seqs);
|
|
||||||
|
|
||||||
ggml_tensor * attn_in = build_norm(
|
|
||||||
cur, model.layers[il].attn_norm, nullptr, LLM_NORM_RMS, il);
|
|
||||||
ggml_tensor * attn_out = build_attn_block(attn_in, il);
|
|
||||||
attn_out = build_sconv(
|
|
||||||
attn_out, model.layers[il].shortconv_attn, off_attn, il);
|
|
||||||
cur = ggml_add(ctx0, cur, attn_out);
|
|
||||||
|
|
||||||
ggml_tensor * ffn_in = build_norm(
|
|
||||||
cur, model.layers[il].ffn_norm, nullptr, LLM_NORM_RMS, il);
|
|
||||||
ggml_tensor * ffn_out = il < (int) hparams.n_layer_dense_lead
|
|
||||||
? build_dense_ffn(ffn_in, il)
|
|
||||||
: build_moe(ffn_in, il);
|
|
||||||
ffn_out = build_sconv(
|
|
||||||
ffn_out, model.layers[il].shortconv_mlp, off_mlp, il);
|
|
||||||
cur = ggml_add(ctx0, cur, ffn_out);
|
|
||||||
|
|
||||||
cur = build_cvec(cur, il);
|
|
||||||
cb(cur, "l_out", il);
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_tensor * inp_out_ids = build_inp_out_ids();
|
|
||||||
if (inp_out_ids) {
|
|
||||||
cur = ggml_get_rows(ctx0, cur, inp_out_ids);
|
|
||||||
}
|
|
||||||
|
|
||||||
cur = build_norm(cur, model.output_norm, nullptr, LLM_NORM_RMS, -1);
|
|
||||||
res->t_embd = cur;
|
|
||||||
|
|
||||||
if (!cparams.embeddings) {
|
|
||||||
cur = ggml_scale(ctx0, cur, hparams.f_logit_scale);
|
|
||||||
cur = build_lora_mm(model.output, cur);
|
|
||||||
if (model.output->type == GGML_TYPE_F32) {
|
|
||||||
ggml_mul_mat_set_prec(cur, GGML_PREC_F32);
|
|
||||||
}
|
|
||||||
if (vocab_mask) {
|
|
||||||
cur = ggml_add(ctx0, cur, vocab_mask);
|
|
||||||
}
|
|
||||||
res->t_logits = cur;
|
|
||||||
}
|
|
||||||
|
|
||||||
ggml_build_forward_expand(gf, cur);
|
|
||||||
}
|
|
||||||
@@ -1865,18 +1865,6 @@ struct llama_model_lfm2moe : public llama_model_base {
|
|||||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
||||||
};
|
};
|
||||||
|
|
||||||
struct llama_model_inkling : public llama_model_base {
|
|
||||||
llama_model_inkling(const struct llama_model_params & params) : llama_model_base(params) {}
|
|
||||||
void load_arch_hparams(llama_model_loader & ml) override;
|
|
||||||
void load_arch_tensors(llama_model_loader & ml) override;
|
|
||||||
|
|
||||||
struct graph : public llm_graph_context {
|
|
||||||
graph(const llama_model & model, const llm_graph_params & params);
|
|
||||||
};
|
|
||||||
|
|
||||||
std::unique_ptr<llm_graph_context> build_arch_graph(const llm_graph_params & params) const override;
|
|
||||||
};
|
|
||||||
|
|
||||||
|
|
||||||
struct llama_model_smallthinker : public llama_model_base {
|
struct llama_model_smallthinker : public llama_model_base {
|
||||||
llama_model_smallthinker(const struct llama_model_params & params) : llama_model_base(params) {}
|
llama_model_smallthinker(const struct llama_model_params & params) : llama_model_base(params) {}
|
||||||
|
|||||||
@@ -259,40 +259,6 @@ set_tests_properties(test-thread-safety PROPERTIES FIXTURES_REQUIRED test-downlo
|
|||||||
|
|
||||||
llama_build_and_test(test-arg-parser.cpp)
|
llama_build_and_test(test-arg-parser.cpp)
|
||||||
llama_build_and_test(test-longhaul.cpp)
|
llama_build_and_test(test-longhaul.cpp)
|
||||||
llama_test(
|
|
||||||
test-longhaul
|
|
||||||
NAME test-longhaul-cpu-qwen35moe
|
|
||||||
ARGS --cpu-model "${MODEL_DIR}/qwen35moe-moe.gguf" 3145728
|
|
||||||
)
|
|
||||||
set_tests_properties(test-longhaul-cpu-qwen35moe PROPERTIES
|
|
||||||
FIXTURES_REQUIRED generate-models
|
|
||||||
)
|
|
||||||
llama_test(
|
|
||||||
test-longhaul
|
|
||||||
NAME test-longhaul-cpu-laguna
|
|
||||||
ARGS --cpu-model "${MODEL_DIR}/laguna-moe.gguf" 2097152
|
|
||||||
)
|
|
||||||
set_tests_properties(test-longhaul-cpu-laguna PROPERTIES
|
|
||||||
FIXTURES_REQUIRED generate-models
|
|
||||||
)
|
|
||||||
llama_test(
|
|
||||||
test-longhaul
|
|
||||||
NAME test-longhaul-cpu-inkling
|
|
||||||
ARGS --cpu-model "${MODEL_DIR}/inkling-moe.gguf" 73728
|
|
||||||
)
|
|
||||||
set_tests_properties(test-longhaul-cpu-inkling PROPERTIES
|
|
||||||
FIXTURES_REQUIRED generate-models
|
|
||||||
)
|
|
||||||
if (APPLE AND GGML_METAL)
|
|
||||||
llama_test(
|
|
||||||
test-longhaul
|
|
||||||
NAME test-longhaul-metal-inkling
|
|
||||||
ARGS --metal-model "${MODEL_DIR}/inkling-moe.gguf" 73728
|
|
||||||
)
|
|
||||||
set_tests_properties(test-longhaul-metal-inkling PROPERTIES
|
|
||||||
FIXTURES_REQUIRED generate-models
|
|
||||||
)
|
|
||||||
endif()
|
|
||||||
llama_build_and_test(test-token-cache.cpp)
|
llama_build_and_test(test-token-cache.cpp)
|
||||||
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
||||||
|
|
||||||
|
|||||||
@@ -91,7 +91,6 @@ static void test_normalize_quotes_with_embedded_quotes(testing & t);
|
|||||||
static void test_tagged_args_with_embedded_quotes(testing & t);
|
static void test_tagged_args_with_embedded_quotes(testing & t);
|
||||||
|
|
||||||
static void test_role_markers_all_templates(testing & t);
|
static void test_role_markers_all_templates(testing & t);
|
||||||
static void test_inkling_tml_parser(testing & t);
|
|
||||||
|
|
||||||
int main(int argc, char * argv[]) {
|
int main(int argc, char * argv[]) {
|
||||||
testing t(std::cout);
|
testing t(std::cout);
|
||||||
@@ -118,7 +117,6 @@ int main(int argc, char * argv[]) {
|
|||||||
t.test("standard_json_tools", test_standard_json_tools_formats);
|
t.test("standard_json_tools", test_standard_json_tools_formats);
|
||||||
t.test("normalize_quotes_to_json", test_normalize_quotes_to_json);
|
t.test("normalize_quotes_to_json", test_normalize_quotes_to_json);
|
||||||
t.test("tagged_args_embedded_quotes", test_tagged_args_with_embedded_quotes);
|
t.test("tagged_args_embedded_quotes", test_tagged_args_with_embedded_quotes);
|
||||||
t.test("inkling_tml", test_inkling_tml_parser);
|
|
||||||
t.test("role_markers_all_templates", test_role_markers_all_templates);
|
t.test("role_markers_all_templates", test_role_markers_all_templates);
|
||||||
|
|
||||||
return t.summary();
|
return t.summary();
|
||||||
@@ -2077,53 +2075,6 @@ static void test_role_markers_all_templates(testing & t) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
static void test_inkling_tml_parser(testing & t) {
|
|
||||||
const std::string tmpl_src =
|
|
||||||
"{# <|content_thinking|><|content_text|> #}"
|
|
||||||
"{%- for message in messages -%}"
|
|
||||||
"<|message_user|><|content_text|>{{ message.content }}<|end_message|>"
|
|
||||||
"{%- endfor -%}"
|
|
||||||
"{%- if add_generation_prompt -%}<|message_model|>{%- endif -%}";
|
|
||||||
|
|
||||||
common_chat_templates_ptr tmpls =
|
|
||||||
common_chat_templates_init(nullptr, tmpl_src);
|
|
||||||
common_chat_templates_inputs inputs;
|
|
||||||
common_chat_msg user_message;
|
|
||||||
user_message.role = "user";
|
|
||||||
user_message.content = "Hello";
|
|
||||||
inputs.messages = { user_message };
|
|
||||||
inputs.add_generation_prompt = true;
|
|
||||||
inputs.reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
|
|
||||||
inputs.use_jinja = true;
|
|
||||||
|
|
||||||
const common_chat_params params =
|
|
||||||
common_chat_templates_apply(tmpls.get(), inputs);
|
|
||||||
t.assert_equal(
|
|
||||||
"Inkling uses the native PEG parser",
|
|
||||||
COMMON_CHAT_FORMAT_PEG_NATIVE,
|
|
||||||
params.format);
|
|
||||||
t.assert_true("Inkling exposes thinking blocks", params.supports_thinking);
|
|
||||||
t.assert_equal(
|
|
||||||
"Inkling generation marker",
|
|
||||||
"<|message_model|>",
|
|
||||||
params.generation_prompt);
|
|
||||||
|
|
||||||
common_peg_arena arena;
|
|
||||||
arena.load(params.parser);
|
|
||||||
common_chat_parser_params parser_params(params);
|
|
||||||
parser_params.reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
|
|
||||||
|
|
||||||
const common_chat_msg parsed = common_chat_peg_parse(
|
|
||||||
arena,
|
|
||||||
"<|content_thinking|>plan<|end_message|>"
|
|
||||||
"<|message_model|><|content_text|>answer<|end_message|>"
|
|
||||||
"<|content_model_end_sampling|>",
|
|
||||||
false,
|
|
||||||
parser_params);
|
|
||||||
t.assert_equal("reasoning block", "plan", parsed.reasoning_content);
|
|
||||||
t.assert_equal("text block", "answer", parsed.content);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Test that reproduces the Seed-OSS template issue with embedded quotes
|
// Test that reproduces the Seed-OSS template issue with embedded quotes
|
||||||
static void test_tagged_args_with_embedded_quotes(testing & t) {
|
static void test_tagged_args_with_embedded_quotes(testing & t) {
|
||||||
json tools = build_edit_tool();
|
json tools = build_edit_tool();
|
||||||
@@ -2241,3 +2192,4 @@ static void test_tagged_args_with_embedded_quotes(testing & t) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,6 @@
|
|||||||
|
|
||||||
// TODO: replace with #include "llama-ext.h" in the future
|
// TODO: replace with #include "llama-ext.h" in the future
|
||||||
#include "../src/llama-arch.h"
|
#include "../src/llama-arch.h"
|
||||||
#include "../src/llama-model.h"
|
|
||||||
#include "../src/llama-model-saver.h"
|
#include "../src/llama-model-saver.h"
|
||||||
|
|
||||||
#include <cinttypes>
|
#include <cinttypes>
|
||||||
@@ -114,11 +113,6 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
n_layer = 3;
|
n_layer = 3;
|
||||||
} else if (arch == LLM_ARCH_CHAMELEON) {
|
} else if (arch == LLM_ARCH_CHAMELEON) {
|
||||||
n_vocab = 10240;
|
n_vocab = 10240;
|
||||||
} else if (arch == LLM_ARCH_INKLING) {
|
|
||||||
n_embd = 64;
|
|
||||||
n_head = 2;
|
|
||||||
n_ff = 96;
|
|
||||||
n_layer = 2;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const uint32_t n_embd_head = n_embd / n_head;
|
const uint32_t n_embd_head = n_embd / n_head;
|
||||||
@@ -203,10 +197,6 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
pattern.push_back(il % 2);
|
pattern.push_back(il % 2);
|
||||||
}
|
}
|
||||||
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, pattern);
|
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, pattern);
|
||||||
} else if (arch == LLM_ARCH_INKLING) {
|
|
||||||
ms.add_kv(
|
|
||||||
LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN,
|
|
||||||
std::vector<uint32_t>({1, 0}));
|
|
||||||
} else {
|
} else {
|
||||||
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, uint32_t(2));
|
ms.add_kv(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, uint32_t(2));
|
||||||
}
|
}
|
||||||
@@ -226,27 +216,12 @@ static gguf_context_ptr get_gguf_ctx(const llm_arch arch, const bool moe) {
|
|||||||
if (moe) {
|
if (moe) {
|
||||||
ms.add_kv(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, n_ff);
|
ms.add_kv(LLM_KV_EXPERT_FEED_FORWARD_LENGTH, n_ff);
|
||||||
ms.add_kv(LLM_KV_INTERLEAVE_MOE_LAYER_STEP, uint32_t(2));
|
ms.add_kv(LLM_KV_INTERLEAVE_MOE_LAYER_STEP, uint32_t(2));
|
||||||
ms.add_kv(LLM_KV_EXPERT_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(8) : uint32_t(2));
|
ms.add_kv(LLM_KV_EXPERT_COUNT, uint32_t(2));
|
||||||
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(6) : uint32_t(1));
|
ms.add_kv(LLM_KV_EXPERT_USED_COUNT, uint32_t(1));
|
||||||
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, arch == LLM_ARCH_INKLING ? uint32_t(2) : uint32_t(1));
|
ms.add_kv(LLM_KV_EXPERT_SHARED_COUNT, uint32_t(1));
|
||||||
ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, uint32_t(2)); // sigmoid
|
ms.add_kv(LLM_KV_EXPERT_GATING_FUNC, uint32_t(2)); // sigmoid
|
||||||
ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f);
|
ms.add_kv(LLM_KV_EXPERT_GROUP_SCALE, 1.0f);
|
||||||
ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1));
|
ms.add_kv(LLM_KV_EXPERTS_PER_GROUP, uint32_t(1));
|
||||||
if (arch == LLM_ARCH_INKLING) {
|
|
||||||
ms.add_kv(LLM_KV_EXPERT_WEIGHTS_SCALE, 1.0f);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (arch == LLM_ARCH_INKLING) {
|
|
||||||
ms.add_kv(LLM_KV_INKLING_D_REL, uint32_t(8));
|
|
||||||
ms.add_kv(LLM_KV_INKLING_REL_EXTENT, uint32_t(32));
|
|
||||||
ms.add_kv(LLM_KV_INKLING_REL_EXTENT_SWA, uint32_t(16));
|
|
||||||
ms.add_kv(LLM_KV_INKLING_SHORTCONV_KERNEL, uint32_t(4));
|
|
||||||
ms.add_kv(LLM_KV_INKLING_DENSE_BLOCK_COUNT, uint32_t(1));
|
|
||||||
ms.add_kv(LLM_KV_INKLING_LOGIT_SCALE_DENOM, 1.0f);
|
|
||||||
ms.add_kv(LLM_KV_INKLING_LOG_SCALING_N_FLOOR, uint32_t(16));
|
|
||||||
ms.add_kv(LLM_KV_INKLING_LOG_SCALING_ALPHA, 0.1f);
|
|
||||||
ms.add_kv(LLM_KV_INKLING_UNPADDED_VOCAB_SIZE, n_vocab);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
ms.add_kv(LLM_KV_POSNET_EMBEDDING_LENGTH, n_embd);
|
ms.add_kv(LLM_KV_POSNET_EMBEDDING_LENGTH, n_embd);
|
||||||
@@ -396,7 +371,6 @@ static bool moe_mandatory(const llm_arch arch) {
|
|||||||
case LLM_ARCH_MISTRAL4:
|
case LLM_ARCH_MISTRAL4:
|
||||||
case LLM_ARCH_MELLUM:
|
case LLM_ARCH_MELLUM:
|
||||||
case LLM_ARCH_LAGUNA:
|
case LLM_ARCH_LAGUNA:
|
||||||
case LLM_ARCH_INKLING:
|
|
||||||
return true;
|
return true;
|
||||||
default:
|
default:
|
||||||
return false;
|
return false;
|
||||||
@@ -502,11 +476,7 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
|
|||||||
if (!moe && moe_mandatory(arch)) {
|
if (!moe && moe_mandatory(arch)) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
const bool can_save_fixture =
|
if (!llama_model_saver_supports_arch(arch) || !arch_supported(arch)) {
|
||||||
llama_model_saver_supports_arch(arch) ||
|
|
||||||
arch == LLM_ARCH_LAGUNA ||
|
|
||||||
arch == LLM_ARCH_INKLING;
|
|
||||||
if (!can_save_fixture || !arch_supported(arch)) {
|
|
||||||
LOG_INF("%s: %s model (%s) is unsupported, skipping\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense");
|
LOG_INF("%s: %s model (%s) is unsupported, skipping\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense");
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -514,26 +484,7 @@ static int save_models(const llm_arch target_arch, const size_t seed, const ggml
|
|||||||
auto model_and_ctx = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {});
|
auto model_and_ctx = get_model_and_ctx(gguf_ctx.get(), nullptr, seed, {});
|
||||||
const std::string path = dir + "/" + llm_arch_name(arch) + (moe ? "-moe.gguf" : "-dense.gguf");
|
const std::string path = dir + "/" + llm_arch_name(arch) + (moe ? "-moe.gguf" : "-dense.gguf");
|
||||||
LOG_INF("%s: Saving %s model (%s) to %s...\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense", path.c_str());
|
LOG_INF("%s: Saving %s model (%s) to %s...\n", __func__, llm_arch_name(arch), moe ? "MoE" : "dense", path.c_str());
|
||||||
const bool private_fixture =
|
llama_model_save_to_file(model_and_ctx.first.get(), path.c_str());
|
||||||
arch == LLM_ARCH_LAGUNA || arch == LLM_ARCH_INKLING;
|
|
||||||
if (llama_model_saver_supports_arch(arch) && !private_fixture) {
|
|
||||||
llama_model_save_to_file(model_and_ctx.first.get(), path.c_str());
|
|
||||||
} else {
|
|
||||||
// Laguna and Inkling are not supported by the production
|
|
||||||
// model saver yet.
|
|
||||||
// Preserve the exact synthetic fixture metadata and attach the
|
|
||||||
// initialized model tensors so longhaul can be tested from a
|
|
||||||
// real file without broadening the saver API.
|
|
||||||
gguf_context_ptr fixture(gguf_init_empty());
|
|
||||||
// Model initialization normalizes the caller's GGUF context,
|
|
||||||
// so regenerate the exact private metadata for the file.
|
|
||||||
gguf_context_ptr fixture_metadata = get_gguf_ctx(arch, moe);
|
|
||||||
gguf_set_kv(fixture.get(), fixture_metadata.get());
|
|
||||||
for (const auto & item : llama_internal_get_tensor_map(model_and_ctx.first.get())) {
|
|
||||||
gguf_add_tensor(fixture.get(), item.second);
|
|
||||||
}
|
|
||||||
gguf_write_to_file(fixture.get(), path.c_str(), false);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);
|
llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);
|
||||||
|
|||||||
+3
-158
@@ -1,16 +1,11 @@
|
|||||||
#include "testing.h"
|
#include "testing.h"
|
||||||
|
|
||||||
#include "../src/llama-longhaul.h"
|
#include "../src/llama-longhaul.h"
|
||||||
#include "../src/llama-model.h"
|
|
||||||
|
|
||||||
#include "ggml-backend.h"
|
#include "ggml-backend.h"
|
||||||
#include "llama-cpp.h"
|
|
||||||
|
|
||||||
#include <cmath>
|
|
||||||
#include <cstdio>
|
#include <cstdio>
|
||||||
#include <memory>
|
#include <memory>
|
||||||
#include <stdexcept>
|
|
||||||
#include <string>
|
|
||||||
#include <vector>
|
#include <vector>
|
||||||
|
|
||||||
struct longhaul_fixture {
|
struct longhaul_fixture {
|
||||||
@@ -63,7 +58,8 @@ struct longhaul_fixture {
|
|||||||
sources.push_back({ weights_b, 0, n_experts * expert_size, expert_size, 0 });
|
sources.push_back({ weights_b, 0, n_experts * expert_size, expert_size, 0 });
|
||||||
}
|
}
|
||||||
cache = std::make_unique<llama_longhaul_cache>(
|
cache = std::make_unique<llama_longhaul_cache>(
|
||||||
std::move(files), std::move(sources), n_slots, n_experts, 1);
|
std::move(files), std::move(sources), std::vector<llama_model_loader::longhaul_shard>{},
|
||||||
|
n_slots, n_experts, 1, false);
|
||||||
}
|
}
|
||||||
|
|
||||||
~longhaul_fixture() {
|
~longhaul_fixture() {
|
||||||
@@ -133,32 +129,6 @@ static void test_multiple_sources(testing & t) {
|
|||||||
t.assert_equal(uint8_t(34), fixture.slot_value(fixture.weights_b, remapped[1]));
|
t.assert_equal(uint8_t(34), fixture.slot_value(fixture.weights_b, remapped[1]));
|
||||||
}
|
}
|
||||||
|
|
||||||
static void test_concurrent_source_reads(testing & t) {
|
|
||||||
longhaul_fixture fixture(2, 4, true);
|
|
||||||
|
|
||||||
for (int iteration = 0; iteration < 64; ++iteration) {
|
|
||||||
const int32_t first = iteration % 2 == 0 ? 0 : 2;
|
|
||||||
const int32_t second = first + 1;
|
|
||||||
const auto remapped = fixture.remap({first, second});
|
|
||||||
|
|
||||||
if (!t.assert_equal(2u, remapped.size())) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
t.assert_equal(
|
|
||||||
uint8_t(1 + first),
|
|
||||||
fixture.slot_value(fixture.weights_a, remapped[0]));
|
|
||||||
t.assert_equal(
|
|
||||||
uint8_t(33 + first),
|
|
||||||
fixture.slot_value(fixture.weights_b, remapped[0]));
|
|
||||||
t.assert_equal(
|
|
||||||
uint8_t(1 + second),
|
|
||||||
fixture.slot_value(fixture.weights_a, remapped[1]));
|
|
||||||
t.assert_equal(
|
|
||||||
uint8_t(33 + second),
|
|
||||||
fixture.slot_value(fixture.weights_b, remapped[1]));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
static void test_invalid_ids(testing & t) {
|
static void test_invalid_ids(testing & t) {
|
||||||
longhaul_fixture fixture(2, 4);
|
longhaul_fixture fixture(2, 4);
|
||||||
|
|
||||||
@@ -183,116 +153,8 @@ static void test_read_failure_recovery(testing & t) {
|
|||||||
t.assert_equal(uint8_t(1), fixture.slot_value(fixture.weights_a, remapped[0]));
|
t.assert_equal(uint8_t(1), fixture.slot_value(fixture.weights_a, remapped[0]));
|
||||||
}
|
}
|
||||||
|
|
||||||
struct cpu_decode_result {
|
|
||||||
std::vector<float> logits;
|
|
||||||
uint32_t capacity = 0;
|
|
||||||
uint64_t misses = 0;
|
|
||||||
uint64_t bytes_read = 0;
|
|
||||||
bool expert_buffer_is_host = false;
|
|
||||||
};
|
|
||||||
|
|
||||||
static cpu_decode_result decode_model(
|
|
||||||
const std::string & path,
|
|
||||||
llama_load_mode load_mode,
|
|
||||||
uint64_t cache_bytes,
|
|
||||||
int32_t n_gpu_layers) {
|
|
||||||
llama_model_params model_params = llama_model_default_params();
|
|
||||||
model_params.n_gpu_layers = n_gpu_layers;
|
|
||||||
model_params.load_mode = load_mode;
|
|
||||||
model_params.longhaul_cache_bytes = cache_bytes;
|
|
||||||
model_params.use_extra_bufts = true;
|
|
||||||
|
|
||||||
llama_model_ptr model(llama_model_load_from_file(path.c_str(), model_params));
|
|
||||||
if (!model) {
|
|
||||||
throw std::runtime_error("failed to load CPU model: " + path);
|
|
||||||
}
|
|
||||||
|
|
||||||
llama_context_params context_params = llama_context_default_params();
|
|
||||||
context_params.n_ctx = 32;
|
|
||||||
context_params.n_batch = 8;
|
|
||||||
context_params.n_ubatch = 8;
|
|
||||||
context_params.n_threads = 2;
|
|
||||||
context_params.n_threads_batch = 2;
|
|
||||||
|
|
||||||
llama_context_ptr context(llama_init_from_model(model.get(), context_params));
|
|
||||||
if (!context) {
|
|
||||||
throw std::runtime_error("failed to create CPU context: " + path);
|
|
||||||
}
|
|
||||||
|
|
||||||
std::vector<llama_token> tokens = {1, 2, 3, 4, 5, 6, 7, 8};
|
|
||||||
llama_batch batch = llama_batch_get_one(tokens.data(), tokens.size());
|
|
||||||
if (llama_decode(context.get(), batch) != 0) {
|
|
||||||
throw std::runtime_error("failed to decode CPU model: " + path);
|
|
||||||
}
|
|
||||||
|
|
||||||
const int32_t n_vocab = llama_vocab_n_tokens(llama_model_get_vocab(model.get()));
|
|
||||||
const float * logits = llama_get_logits_ith(context.get(), -1);
|
|
||||||
if (!logits) {
|
|
||||||
throw std::runtime_error("CPU model produced no logits: " + path);
|
|
||||||
}
|
|
||||||
|
|
||||||
cpu_decode_result result;
|
|
||||||
result.logits.assign(logits, logits + n_vocab);
|
|
||||||
|
|
||||||
if (load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
|
|
||||||
llama_longhaul_cache * cache = model->longhaul_cache();
|
|
||||||
if (!cache) {
|
|
||||||
throw std::runtime_error("longhaul cache was not created: " + path);
|
|
||||||
}
|
|
||||||
result.capacity = cache->capacity();
|
|
||||||
result.misses = cache->misses();
|
|
||||||
result.bytes_read = cache->bytes_read_count();
|
|
||||||
|
|
||||||
for (const auto & item : llama_internal_get_tensor_map(model.get())) {
|
|
||||||
const std::string & name = item.first;
|
|
||||||
if (name.find("ffn_") != std::string::npos &&
|
|
||||||
name.find("_exps.weight") != std::string::npos &&
|
|
||||||
item.second->ne[2] == result.capacity) {
|
|
||||||
result.expert_buffer_is_host = ggml_backend_buffer_is_host(item.second->buffer);
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
static void test_model(
|
|
||||||
testing & t,
|
|
||||||
const std::string & path,
|
|
||||||
uint64_t cache_bytes,
|
|
||||||
int32_t n_gpu_layers) {
|
|
||||||
const cpu_decode_result regular =
|
|
||||||
decode_model(path, LLAMA_LOAD_MODE_NONE, 0, n_gpu_layers);
|
|
||||||
const cpu_decode_result longhaul =
|
|
||||||
decode_model(path, LLAMA_LOAD_MODE_LONGHAUL, cache_bytes, n_gpu_layers);
|
|
||||||
|
|
||||||
t.assert_equal(1u, longhaul.capacity);
|
|
||||||
t.assert_true("longhaul loaded at least one expert", longhaul.misses > 0);
|
|
||||||
t.assert_true("longhaul read expert bytes", longhaul.bytes_read > 0);
|
|
||||||
if (n_gpu_layers == 0) {
|
|
||||||
t.assert_true(
|
|
||||||
"streamed CPU experts use a directly writable host buffer",
|
|
||||||
longhaul.expert_buffer_is_host);
|
|
||||||
}
|
|
||||||
t.assert_equal(regular.logits.size(), longhaul.logits.size());
|
|
||||||
|
|
||||||
double squared_error = 0.0;
|
|
||||||
double squared_reference = 0.0;
|
|
||||||
for (size_t i = 0; i < regular.logits.size(); ++i) {
|
|
||||||
const double delta = double(regular.logits[i]) - double(longhaul.logits[i]);
|
|
||||||
squared_error += delta * delta;
|
|
||||||
squared_reference += double(regular.logits[i]) * double(regular.logits[i]);
|
|
||||||
}
|
|
||||||
const double nmse = squared_reference > 0.0 ? squared_error / squared_reference : squared_error;
|
|
||||||
t.assert_true(
|
|
||||||
"longhaul CPU logits NMSE = " + std::to_string(nmse),
|
|
||||||
std::isfinite(nmse) && nmse <= 1e-6);
|
|
||||||
}
|
|
||||||
|
|
||||||
int main(int argc, char ** argv) {
|
int main(int argc, char ** argv) {
|
||||||
testing t;
|
testing t;
|
||||||
llama_backend_init();
|
|
||||||
|
|
||||||
const char * verbose = getenv("LLAMA_TEST_VERBOSE");
|
const char * verbose = getenv("LLAMA_TEST_VERBOSE");
|
||||||
if (verbose) {
|
if (verbose) {
|
||||||
@@ -302,31 +164,14 @@ int main(int argc, char ** argv) {
|
|||||||
llama_log_set([](ggml_log_level, const char *, void *) {}, nullptr);
|
llama_log_set([](ggml_log_level, const char *, void *) {}, nullptr);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (argc == 4 &&
|
|
||||||
(std::string(argv[1]) == "--cpu-model" ||
|
|
||||||
std::string(argv[1]) == "--metal-model")) {
|
|
||||||
const std::string path = argv[2];
|
|
||||||
const uint64_t cache_bytes = std::stoull(argv[3]);
|
|
||||||
const bool metal = std::string(argv[1]) == "--metal-model";
|
|
||||||
t.test(metal ? "metal_model" : "cpu_model", [&](testing & current) {
|
|
||||||
test_model(current, path, cache_bytes, metal ? 99 : 0);
|
|
||||||
});
|
|
||||||
const int result = t.summary();
|
|
||||||
llama_backend_free();
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (argc > 1) {
|
if (argc > 1) {
|
||||||
t.set_filter(argv[1]);
|
t.set_filter(argv[1]);
|
||||||
}
|
}
|
||||||
|
|
||||||
t.test("batch_planning", test_batch_planning);
|
t.test("batch_planning", test_batch_planning);
|
||||||
t.test("multiple_sources", test_multiple_sources);
|
t.test("multiple_sources", test_multiple_sources);
|
||||||
t.test("concurrent_reads", test_concurrent_source_reads);
|
|
||||||
t.test("invalid_ids", test_invalid_ids);
|
t.test("invalid_ids", test_invalid_ids);
|
||||||
t.test("read_failure", test_read_failure_recovery);
|
t.test("read_failure", test_read_failure_recovery);
|
||||||
|
|
||||||
const int result = t.summary();
|
return t.summary();
|
||||||
llama_backend_free();
|
|
||||||
return result;
|
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -61,7 +61,7 @@
|
|||||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
| `--longhaul` | stream routed Qwen3.5 MoE experts through a bounded Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||||
|
|||||||
@@ -144,7 +144,7 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
|
|||||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
| `--longhaul` | stream routed Qwen3.5 MoE experts through a bounded Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||||
|
|||||||
+11
-1
@@ -36,6 +36,17 @@ flowchart TD
|
|||||||
By default, `ggml-rpc-server` exposes all available accelerator devices on the host.
|
By default, `ggml-rpc-server` exposes all available accelerator devices on the host.
|
||||||
If there are no accelerators, it exposes a single `CPU` device.
|
If there are no accelerators, it exposes a single `CPU` device.
|
||||||
|
|
||||||
|
For server-resident longhaul MoE paging, place the same GGUF shards on the RPC
|
||||||
|
host and point the server at their directory:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
$ bin/ggml-rpc-server --device MTL0 --longhaul-root /path/to/model-directory
|
||||||
|
```
|
||||||
|
|
||||||
|
The client can then use `--rpc`, `--longhaul`, and `--longhaul-cache` together.
|
||||||
|
RPC longhaul v1 requires one remote Metal device. Expert cache misses are read
|
||||||
|
from the RPC host's local files rather than uploaded by the client.
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
### Remote hosts
|
### Remote hosts
|
||||||
@@ -107,4 +118,3 @@ Use the `GGML_RPC_DEBUG` environment variable to enable debug messages from `ggm
|
|||||||
```bash
|
```bash
|
||||||
$ GGML_RPC_DEBUG=1 bin/ggml-rpc-server
|
$ GGML_RPC_DEBUG=1 bin/ggml-rpc-server
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|||||||
@@ -13,6 +13,7 @@
|
|||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
#include <clocale>
|
#include <clocale>
|
||||||
#include <codecvt>
|
#include <codecvt>
|
||||||
|
#include <cstring>
|
||||||
#include <filesystem>
|
#include <filesystem>
|
||||||
#include <regex>
|
#include <regex>
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
@@ -173,6 +174,7 @@ struct rpc_server_params {
|
|||||||
std::string host = "127.0.0.1";
|
std::string host = "127.0.0.1";
|
||||||
int port = 50052;
|
int port = 50052;
|
||||||
bool use_cache = false;
|
bool use_cache = false;
|
||||||
|
std::string longhaul_root;
|
||||||
int n_threads = std::max(1U, std::thread::hardware_concurrency()/2);
|
int n_threads = std::max(1U, std::thread::hardware_concurrency()/2);
|
||||||
std::vector<std::string> devices;
|
std::vector<std::string> devices;
|
||||||
};
|
};
|
||||||
@@ -186,6 +188,7 @@ static void print_usage(int /*argc*/, char ** argv, rpc_server_params params) {
|
|||||||
fprintf(stderr, " -H, --host HOST host to bind to (default: %s)\n", params.host.c_str());
|
fprintf(stderr, " -H, --host HOST host to bind to (default: %s)\n", params.host.c_str());
|
||||||
fprintf(stderr, " -p, --port PORT port to bind to (default: %d)\n", params.port);
|
fprintf(stderr, " -p, --port PORT port to bind to (default: %d)\n", params.port);
|
||||||
fprintf(stderr, " -c, --cache enable local file cache\n");
|
fprintf(stderr, " -c, --cache enable local file cache\n");
|
||||||
|
fprintf(stderr, " --longhaul-root PATH serve longhaul expert data from this model directory\n");
|
||||||
fprintf(stderr, "\n");
|
fprintf(stderr, "\n");
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -233,6 +236,11 @@ static bool rpc_server_params_parse(int argc, char ** argv, rpc_server_params &
|
|||||||
}
|
}
|
||||||
} else if (arg == "-c" || arg == "--cache") {
|
} else if (arg == "-c" || arg == "--cache") {
|
||||||
params.use_cache = true;
|
params.use_cache = true;
|
||||||
|
} else if (arg == "--longhaul-root") {
|
||||||
|
if (++i >= argc) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
params.longhaul_root = argv[i];
|
||||||
} else if (arg == "-h" || arg == "--help") {
|
} else if (arg == "-h" || arg == "--help") {
|
||||||
print_usage(argc, argv, params);
|
print_usage(argc, argv, params);
|
||||||
exit(0);
|
exit(0);
|
||||||
@@ -313,6 +321,23 @@ int main(int argc, char * argv[]) {
|
|||||||
fprintf(stderr, "No devices found\n");
|
fprintf(stderr, "No devices found\n");
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (!params.longhaul_root.empty()) {
|
||||||
|
std::error_code ec;
|
||||||
|
const auto root = std::filesystem::canonical(params.longhaul_root, ec);
|
||||||
|
if (ec || !std::filesystem::is_directory(root, ec)) {
|
||||||
|
fprintf(stderr, "Invalid longhaul root: %s\n", params.longhaul_root.c_str());
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
params.longhaul_root = root.string();
|
||||||
|
for (auto * dev : devices) {
|
||||||
|
auto * dev_reg = ggml_backend_dev_backend_reg(dev);
|
||||||
|
if (!dev_reg || strcmp(ggml_backend_reg_name(dev_reg), "MTL") != 0) {
|
||||||
|
fprintf(stderr, "--longhaul-root currently requires Metal devices\n");
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
std::string endpoint = params.host + ":" + std::to_string(params.port);
|
std::string endpoint = params.host + ":" + std::to_string(params.port);
|
||||||
const char * cache_dir = nullptr;
|
const char * cache_dir = nullptr;
|
||||||
std::string cache_dir_str;
|
std::string cache_dir_str;
|
||||||
@@ -331,12 +356,19 @@ int main(int argc, char * argv[]) {
|
|||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
auto start_server_fn = (decltype(ggml_backend_rpc_start_server)*) ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_start_server");
|
auto start_server_fn = (decltype(ggml_backend_rpc_start_server_with_options)*)
|
||||||
|
ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_start_server_with_options");
|
||||||
if (!start_server_fn) {
|
if (!start_server_fn) {
|
||||||
fprintf(stderr, "Failed to obtain RPC backend start server function\n");
|
fprintf(stderr, "Failed to obtain RPC backend start server function with options\n");
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
start_server_fn(endpoint.c_str(), cache_dir, params.n_threads, devices.size(), devices.data());
|
start_server_fn(
|
||||||
|
endpoint.c_str(),
|
||||||
|
cache_dir,
|
||||||
|
params.longhaul_root.empty() ? nullptr : params.longhaul_root.c_str(),
|
||||||
|
params.n_threads,
|
||||||
|
devices.size(),
|
||||||
|
devices.data());
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -78,7 +78,7 @@ For the full list of features, please refer to [server's changelog](https://gith
|
|||||||
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
| `--mmap, --no-mmap` | DEPRECATED in favor of `--load-mode`: whether to memory-map model. (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>(env: LLAMA_ARG_MMAP) |
|
||||||
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
| `-dio, --direct-io, -ndio, --no-direct-io` | DEPRECATED in favor of `--load-mode`: use DirectIO if available<br/>(env: LLAMA_ARG_DIO) |
|
||||||
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
| `-lm, --load-mode MODE` | model loading mode (default: mmap)<br/>- none: no special loading mode<br/>- mmap: memory-map model (if mmap disabled, slower load but may reduce pageouts if not using mlock)<br/>- mlock: force system to keep model in RAM rather than swapping or compressing<br/>- mmap+mlock: mmap + force system to keep model in RAM rather than swapping or compressing<br/>- dio: use DirectIO if available<br/>- longhaul: stream routed MoE experts through a bounded cache<br/><br/>(env: LLAMA_ARG_LOAD_MODE) |
|
||||||
| `--longhaul` | stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
| `--longhaul` | stream routed Qwen3.5 MoE experts through a bounded Metal cache<br/>(env: LLAMA_ARG_LONGHAUL) |
|
||||||
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
| `--longhaul-cache N` | longhaul expert cache size in GiB<br/>(env: LLAMA_ARG_LONGHAUL_CACHE) |
|
||||||
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
| `--numa TYPE` | attempt optimizations that help on some NUMA systems<br/>- distribute: spread execution evenly over all nodes<br/>- isolate: only spawn threads on CPUs on the node that execution started on<br/>- numactl: use the CPU map provided by numactl<br/>if run without this previously, it is recommended to drop the system page cache before using this<br/>see https://github.com/ggml-org/llama.cpp/issues/1437<br/>(env: LLAMA_ARG_NUMA) |
|
||||||
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
| `-dev, --device <dev1,dev2,..>` | comma-separated list of devices to use for offloading (none = don't offload)<br/>use --list-devices to see a list of available devices<br/>(env: LLAMA_ARG_DEVICE) |
|
||||||
|
|||||||
Reference in New Issue
Block a user