Author SHA1 Message Date
Owen Qwen e0665b1042 Add tokenizer longhaul cache 2026-07-30 23:32:52 -05:00
33 changed files with 1121 additions and 1324 deletions
+26
View File
@@ -817,6 +817,15 @@ static bool common_params_parse_ex(int argc, char ** argv, common_params_context
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL && params.longhaul_cache_bytes == 0) { if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL && params.longhaul_cache_bytes == 0) {
throw std::invalid_argument("error: --longhaul requires --longhaul-cache N\n"); throw std::invalid_argument("error: --longhaul requires --longhaul-cache N\n");
} }
if (params.tokenizer_longhaul != (params.tokenizer_longhaul_cache_bytes != 0)) {
throw std::invalid_argument(
params.tokenizer_longhaul
? "error: --tokenizer-longhaul requires --tokenizer-longhaul-cache N\n"
: "error: --tokenizer-longhaul-cache requires --tokenizer-longhaul\n");
}
if (params.tokenizer_longhaul && params.token_cache_size_mib == 0) {
throw std::invalid_argument("error: --tokenizer-longhaul requires an enabled persistent token cache\n");
}
postprocess_cpu_params(params.cpuparams, nullptr); postprocess_cpu_params(params.cpuparams, nullptr);
postprocess_cpu_params(params.cpuparams_batch, &params.cpuparams); postprocess_cpu_params(params.cpuparams_batch, &params.cpuparams);
@@ -2646,6 +2655,23 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.longhaul_cache_bytes = uint64_t(value) * 1024 * 1024 * 1024; params.longhaul_cache_bytes = uint64_t(value) * 1024 * 1024 * 1024;
} }
).set_env("LLAMA_ARG_LONGHAUL_CACHE")); ).set_env("LLAMA_ARG_LONGHAUL_CACHE"));
add_opt(common_arg(
{"--tokenizer-longhaul"},
"build and page the Qwen3.5/Laguna tokenizer through a bounded cache",
[](common_params & params) {
params.tokenizer_longhaul = true;
}
).set_env("LLAMA_ARG_TOKENIZER_LONGHAUL"));
add_opt(common_arg(
{"--tokenizer-longhaul-cache"}, "N",
"tokenizer-longhaul RAM cache size in MiB",
[](common_params & params, int value) {
if (value <= 0) {
throw std::invalid_argument("tokenizer-longhaul cache size must be positive");
}
params.tokenizer_longhaul_cache_bytes = uint64_t(value) * 1024 * 1024;
}
).set_env("LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE"));
add_opt(common_arg( add_opt(common_arg(
{"--numa"}, "TYPE", {"--numa"}, "TYPE",
"attempt optimizations that help on some NUMA systems\n" "attempt optimizations that help on some NUMA systems\n"
+2
View File
@@ -1592,6 +1592,8 @@ struct llama_model_params common_model_params_to_llama(common_params & params) {
mparams.split_mode = params.split_mode; mparams.split_mode = params.split_mode;
mparams.load_mode = params.load_mode; mparams.load_mode = params.load_mode;
mparams.longhaul_cache_bytes = params.longhaul_cache_bytes; mparams.longhaul_cache_bytes = params.longhaul_cache_bytes;
mparams.tokenizer_longhaul = params.tokenizer_longhaul;
mparams.tokenizer_longhaul_cache_bytes = params.tokenizer_longhaul_cache_bytes;
mparams.tensor_split = params.tensor_split; mparams.tensor_split = params.tensor_split;
mparams.check_tensors = params.check_tensors; mparams.check_tensors = params.check_tensors;
mparams.use_extra_bufts = !params.no_extra_bufts; mparams.use_extra_bufts = !params.no_extra_bufts;
+2
View File
@@ -486,6 +486,8 @@ struct common_params {
enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER; // how to split the model across GPUs enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER; // how to split the model across GPUs
enum llama_load_mode load_mode = LLAMA_LOAD_MODE_MMAP; // how to load the model enum llama_load_mode load_mode = LLAMA_LOAD_MODE_MMAP; // how to load the model
uint64_t longhaul_cache_bytes = 0; uint64_t longhaul_cache_bytes = 0;
bool tokenizer_longhaul = false;
uint64_t tokenizer_longhaul_cache_bytes = 0;
common_cpu_params cpuparams; common_cpu_params cpuparams;
common_cpu_params cpuparams_batch; common_cpu_params cpuparams_batch;
+2 -37
View File
@@ -16,41 +16,6 @@ llama-cli \
`--longhaul-cache` is the expert cache budget in GiB. It is required when `--longhaul` is used. The same options are accepted by `llama-server`. `--longhaul-cache` is the expert cache budget in GiB. It is required when `--longhaul` is used. The same options are accepted by `llama-server`.
## RPC mode
RPC longhaul keeps the paging loop on the accelerator host so expert payloads do
not cross the network during generation. Build both hosts with `-DGGML_RPC=ON`.
Place the same GGUF shard files on both hosts, preserving their basenames, then
start the Metal host with the directory containing those shards:
```sh
ggml-rpc-server \
--device MTL0 \
--longhaul-root /path/to/model-directory \
--host 192.168.1.20
```
Start the client normally:
```sh
llama-server \
--model /local/path/model.gguf \
--rpc 192.168.1.20:50052 \
--longhaul \
--longhaul-cache 2 \
--n-gpu-layers 99
```
RPC longhaul currently requires all repeating layers on one remote Metal device.
The server validates each shard basename, size, GGUF metadata layout digest, and
every registered expert range before decoding. An older server, a server without
`--longhaul-root`, or a shard mismatch is a fatal model-load error; the client
does not fall back to sending expert slices over RPC.
`--longhaul-root` grants connected RPC clients read access to registered byte
ranges in files in that directory. The RPC server remains experimental and
insecure and must not be exposed to an untrusted network.
Longhaul may reduce `--ubatch-size` so that every expert selected by one graph segment can be present in the cache at the same time. The effective value is logged during context creation. If one token selects more experts than the cache has slots, the routed MoE computation is split into multiple stages and the partial results are summed. This permits smaller caches at the cost of additional graph work. Longhaul may reduce `--ubatch-size` so that every expert selected by one graph segment can be present in the cache at the same time. The effective value is logged during context creation. If one token selects more experts than the cache has slots, the routed MoE computation is split into multiple stages and the partial results are summed. This permits smaller caches at the cost of additional graph work.
The normal startup warmup is skipped automatically in longhaul mode. Routed expert weights are not read until the first real decode. The normal startup warmup is skipped automatically in longhaul mode. Routed expert weights are not read until the first real decode.
@@ -59,9 +24,9 @@ The normal startup warmup is skipped automatically in longhaul mode. Routed expe
Longhaul currently requires: Longhaul currently requires:
- the Metal backend, either local on macOS or on one RPC host - macOS with the Metal backend
- Qwen3.5 MoE or Laguna architecture - Qwen3.5 MoE or Laguna architecture
- all repeating model layers assigned to local Metal or one RPC Metal device - all repeating model layers assigned to Metal
- text generation without embeddings or LoRA adapters - text generation without embeddings or LoRA adapters
Longhaul does not restrict the GGUF quantization type. Individual tensor types must still be supported by Metal. Longhaul does not restrict the GGUF quantization type. Individual tensor types must still be supported by Metal.
+21
View File
@@ -30,3 +30,24 @@ call `llama_token_cache_flush()`.
The 5 GiB limit accounts for logical entry payloads. SQLite metadata and bounded The 5 GiB limit accounts for logical entry payloads. SQLite metadata and bounded
journal files are not included. journal files are not included.
## Tokenizer longhaul
Qwen3.5 MoE and Laguna models can move their reverse-token and BPE-merge indexes
into a persistent, disk-backed index while keeping a bounded lookup working set:
```text
--tokenizer-longhaul
--tokenizer-longhaul-cache N
```
`N` is the application-owned lookup cache budget in MiB and is required when
the mode is enabled. The mode is independent of model `--longhaul`, so the two
can be used together. The finite tokenizer index builds in the background while
model loading continues. If tokenizer work arrives first, it waits for atomic
index publication; build or storage failures are reported instead of silently
falling back to an unbounded tokenizer.
The persistent index uses the directory selected by `--token-cache-dir` and is
subject to `--token-cache-size`. Tokenized-text results continue to be cached
on demand and therefore do not have a finite completion point.
+1 -44
View File
@@ -8,7 +8,7 @@ extern "C" {
#define RPC_PROTO_MAJOR_VERSION 4 #define RPC_PROTO_MAJOR_VERSION 4
#define RPC_PROTO_MINOR_VERSION 0 #define RPC_PROTO_MINOR_VERSION 0
#define RPC_PROTO_PATCH_VERSION 4 #define RPC_PROTO_PATCH_VERSION 3
#ifdef __cplusplus #ifdef __cplusplus
static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT has changed - update RPC_PROTO_PATCH_VERSION"); static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT has changed - update RPC_PROTO_PATCH_VERSION");
@@ -27,49 +27,6 @@ GGML_BACKEND_API void ggml_backend_rpc_get_device_memory(const char * endpoint,
GGML_BACKEND_API void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir, GGML_BACKEND_API void ggml_backend_rpc_start_server(const char * endpoint, const char * cache_dir,
size_t n_threads, size_t n_devices, ggml_backend_dev_t * devices); size_t n_threads, size_t n_devices, ggml_backend_dev_t * devices);
struct ggml_backend_rpc_longhaul_shard {
const char * name;
uint64_t size;
uint64_t metadata_size;
uint8_t metadata_sha256[32];
};
struct ggml_backend_rpc_longhaul_source {
struct ggml_tensor * tensor;
uint32_t shard;
uint64_t offset;
uint64_t expert_size;
int32_t layer;
};
struct ggml_backend_rpc_longhaul_params {
uint32_t n_layers;
uint32_t n_experts;
uint32_t n_slots;
size_t n_shards;
const struct ggml_backend_rpc_longhaul_shard * shards;
size_t n_sources;
const struct ggml_backend_rpc_longhaul_source * sources;
};
GGML_BACKEND_API bool ggml_backend_rpc_longhaul_register(
const struct ggml_backend_rpc_longhaul_params * params,
uint64_t * registration_id,
char * error,
size_t error_size);
GGML_BACKEND_API void ggml_backend_rpc_longhaul_unregister(
struct ggml_tensor * tensor,
uint64_t registration_id);
GGML_BACKEND_API void ggml_backend_rpc_start_server_with_options(
const char * endpoint,
const char * cache_dir,
const char * longhaul_root,
size_t n_threads,
size_t n_devices,
ggml_backend_dev_t * devices);
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_reg(void); GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_reg(void);
GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint); GGML_BACKEND_API ggml_backend_reg_t ggml_backend_rpc_add_server(const char * endpoint);
File diff suppressed because it is too large Load Diff
+2
View File
@@ -360,6 +360,8 @@ extern "C" {
bool no_alloc; // only load metadata and simulate memory allocations bool no_alloc; // only load metadata and simulate memory allocations
uint64_t longhaul_cache_bytes; // routed expert cache size for LLAMA_LOAD_MODE_LONGHAUL uint64_t longhaul_cache_bytes; // routed expert cache size for LLAMA_LOAD_MODE_LONGHAUL
bool tokenizer_longhaul; // page Qwen3.5/Laguna tokenizer indexes through a bounded cache
uint64_t tokenizer_longhaul_cache_bytes; // tokenizer-longhaul RAM cache budget
}; };
struct llama_sampler_seq_config { struct llama_sampler_seq_config {
+1
View File
@@ -38,6 +38,7 @@ add_library(llama
llama-quant.cpp llama-quant.cpp
llama-sampler.cpp llama-sampler.cpp
llama-token-cache.cpp llama-token-cache.cpp
llama-tokenizer-longhaul.cpp
llama-vocab.cpp llama-vocab.cpp
unicode-data.cpp unicode-data.cpp
unicode.cpp unicode.cpp
+3 -5
View File
@@ -1359,12 +1359,10 @@ llm_graph_result * llama_context::process_ubatch(const llama_ubatch & ubatch, ll
res->reset(); res->reset();
ggml_backend_sched_reset(sched.get()); ggml_backend_sched_reset(sched.get());
auto * longhaul = model.longhaul_cache();
const bool local_longhaul = longhaul && !longhaul->is_remote();
ggml_backend_sched_set_eval_callback( ggml_backend_sched_set_eval_callback(
sched.get(), sched.get(),
local_longhaul ? longhaul_eval_callback : cparams.cb_eval, model.longhaul_cache() ? longhaul_eval_callback : cparams.cb_eval,
local_longhaul ? this : cparams.cb_eval_user_data); model.longhaul_cache() ? this : cparams.cb_eval_user_data);
//const auto t_start_us = ggml_time_us(); //const auto t_start_us = ggml_time_us();
@@ -2463,7 +2461,7 @@ llm_graph_params llama_context::graph_params(
bool llama_context::longhaul_eval_callback(ggml_tensor * tensor, bool ask, void * user_data) { bool llama_context::longhaul_eval_callback(ggml_tensor * tensor, bool ask, void * user_data) {
auto * ctx = static_cast<llama_context *>(user_data); auto * ctx = static_cast<llama_context *>(user_data);
auto * cache = ctx->model.longhaul_cache(); auto * cache = ctx->model.longhaul_cache();
GGML_ASSERT(cache != nullptr && !cache->is_remote()); GGML_ASSERT(cache != nullptr);
int layer = -1; int layer = -1;
const bool remap = sscanf(tensor->name, "longhaul.remap.%d", &layer) == 1; const bool remap = sscanf(tensor->name, "longhaul.remap.%d", &layer) == 1;
+2 -102
View File
@@ -1,108 +1,23 @@
#include "llama-longhaul.h" #include "llama-longhaul.h"
#include "llama-impl.h" #include "llama-impl.h"
#include "llama-token-cache.h"
#include "ggml-rpc.h"
#include <algorithm> #include <algorithm>
#include <array>
#include <cstring> #include <cstring>
#include <inttypes.h> #include <inttypes.h>
#include <stdexcept>
llama_longhaul_cache::llama_longhaul_cache( llama_longhaul_cache::llama_longhaul_cache(
llama_files files, llama_files files,
std::vector<llama_model_loader::longhaul_source> sources, std::vector<llama_model_loader::longhaul_source> sources,
std::vector<llama_model_loader::longhaul_shard> shards,
size_t n_slots, size_t n_slots,
uint32_t n_experts, uint32_t n_experts,
uint32_t n_layers, uint32_t n_layers) :
bool remote) :
files(std::move(files)), files(std::move(files)),
sources(std::move(sources)), sources(std::move(sources)),
shards(std::move(shards)),
sources_by_layer(n_layers), sources_by_layer(n_layers),
layers(n_layers), layers(n_layers),
n_slots(n_slots), n_slots(n_slots),
n_experts(n_experts), n_experts(n_experts) {
remote(remote) {
if (remote) {
if (this->shards.size() != this->files.size() || this->sources.empty()) {
throw std::runtime_error("RPC longhaul requires named GGUF shards");
}
std::vector<std::array<uint8_t, 32>> digests(this->shards.size());
std::vector<ggml_backend_rpc_longhaul_shard> rpc_shards(this->shards.size());
for (size_t i = 0; i < this->shards.size(); ++i) {
const auto & shard = this->shards[i];
if (shard.metadata_size > shard.size) {
throw std::runtime_error("invalid RPC longhaul shard metadata range");
}
std::vector<uint8_t> metadata(shard.metadata_size);
if (!metadata.empty()) {
this->files[i]->read_at(metadata.data(), metadata.size(), 0);
}
const std::string hex = llama_token_cache_hash(metadata.data(), metadata.size());
if (hex.size() != 64) {
throw std::runtime_error("failed to fingerprint RPC longhaul shard");
}
for (size_t j = 0; j < digests[i].size(); ++j) {
auto nibble = [](char c) -> uint8_t {
if (c >= '0' && c <= '9') return c - '0';
if (c >= 'a' && c <= 'f') return c - 'a' + 10;
if (c >= 'A' && c <= 'F') return c - 'A' + 10;
return 0xff;
};
const uint8_t hi = nibble(hex[2*j]);
const uint8_t lo = nibble(hex[2*j + 1]);
if (hi > 0x0f || lo > 0x0f) {
throw std::runtime_error("invalid RPC longhaul shard fingerprint");
}
digests[i][j] = (hi << 4) | lo;
}
rpc_shards[i] = {
shard.name.c_str(),
shard.size,
shard.metadata_size,
{},
};
memcpy(rpc_shards[i].metadata_sha256, digests[i].data(), digests[i].size());
}
std::vector<ggml_backend_rpc_longhaul_source> rpc_sources;
rpc_sources.reserve(this->sources.size());
for (const auto & source : this->sources) {
rpc_sources.push_back({
source.tensor,
source.file_idx,
source.offset,
source.expert_size,
source.layer,
});
}
ggml_backend_rpc_longhaul_params params = {
n_layers,
n_experts,
(uint32_t) n_slots,
rpc_shards.size(),
rpc_shards.data(),
rpc_sources.size(),
rpc_sources.data(),
};
ggml_backend_reg_t reg = ggml_backend_reg_by_name("RPC");
auto register_fn = reg ? (decltype(ggml_backend_rpc_longhaul_register) *)
ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_longhaul_register") : nullptr;
if (!register_fn) {
throw std::runtime_error("RPC backend does not expose longhaul registration");
}
char error[256] = {};
if (!register_fn(&params, &remote_registration_id, error, sizeof(error))) {
throw std::runtime_error(error[0] ? error : "RPC longhaul registration failed");
}
this->files.clear();
return;
}
for (auto & file : this->files) { for (auto & file : this->files) {
file->set_no_cache(); file->set_no_cache();
} }
@@ -130,17 +45,6 @@ llama_longhaul_cache::llama_longhaul_cache(
} }
llama_longhaul_cache::~llama_longhaul_cache() { llama_longhaul_cache::~llama_longhaul_cache() {
if (remote) {
if (!sources.empty() && remote_registration_id != 0) {
ggml_backend_reg_t reg = ggml_backend_reg_by_name("RPC");
auto unregister_fn = reg ? (decltype(ggml_backend_rpc_longhaul_unregister) *)
ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_longhaul_unregister") : nullptr;
if (unregister_fn) {
unregister_fn(sources.front().tensor, remote_registration_id);
}
}
return;
}
{ {
std::lock_guard<std::mutex> lock(io_mutex); std::lock_guard<std::mutex> lock(io_mutex);
io_stopping = true; io_stopping = true;
@@ -417,7 +321,3 @@ uint64_t llama_longhaul_cache::misses() const {
uint64_t llama_longhaul_cache::bytes_read_count() const { uint64_t llama_longhaul_cache::bytes_read_count() const {
return bytes_read; return bytes_read;
} }
bool llama_longhaul_cache::is_remote() const {
return remote;
}
+1 -7
View File
@@ -17,11 +17,9 @@ struct llama_longhaul_cache {
llama_longhaul_cache( llama_longhaul_cache(
llama_files files, llama_files files,
std::vector<llama_model_loader::longhaul_source> sources, std::vector<llama_model_loader::longhaul_source> sources,
std::vector<llama_model_loader::longhaul_shard> shards,
size_t n_slots, size_t n_slots,
uint32_t n_experts, uint32_t n_experts,
uint32_t n_layers, uint32_t n_layers);
bool remote);
~llama_longhaul_cache(); ~llama_longhaul_cache();
bool remap(int layer, ggml_tensor * ids); bool remap(int layer, ggml_tensor * ids);
@@ -33,7 +31,6 @@ struct llama_longhaul_cache {
bool failed() const; bool failed() const;
uint64_t misses() const; uint64_t misses() const;
uint64_t bytes_read_count() const; uint64_t bytes_read_count() const;
bool is_remote() const;
private: private:
struct layer_state { struct layer_state {
@@ -52,7 +49,6 @@ private:
llama_files files; llama_files files;
std::vector<llama_model_loader::longhaul_source> sources; std::vector<llama_model_loader::longhaul_source> sources;
std::vector<llama_model_loader::longhaul_shard> shards;
std::vector<std::vector<const llama_model_loader::longhaul_source *>> sources_by_layer; std::vector<std::vector<const llama_model_loader::longhaul_source *>> sources_by_layer;
std::vector<layer_state> layers; std::vector<layer_state> layers;
size_t n_slots; size_t n_slots;
@@ -88,8 +84,6 @@ private:
std::condition_variable io_done; std::condition_variable io_done;
size_t io_pending = 0; size_t io_pending = 0;
bool io_stopping = false; bool io_stopping = false;
bool remote = false;
uint64_t remote_registration_id = 0;
bool source_is_direct(const llama_model_loader::longhaul_source & source) const; bool source_is_direct(const llama_model_loader::longhaul_source & source) const;
bool load_plan_sources(); bool load_plan_sources();
-10
View File
@@ -594,11 +594,6 @@ llama_model_loader::llama_model_loader(
files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io)); files.emplace_back(new llama_file(fname.c_str(), "rb", use_direct_io));
contexts.emplace_back(ctx); contexts.emplace_back(ctx);
longhaul_shards.push_back({
std::filesystem::path(fname).filename().string(),
files.back()->size(),
gguf_get_data_offset(metadata),
});
// Save tensors data offset of the main file. // Save tensors data offset of the main file.
// For subsidiary files, `meta` tensor data offset must not be used, // For subsidiary files, `meta` tensor data offset must not be used,
@@ -667,11 +662,6 @@ llama_model_loader::llama_model_loader(
files.emplace_back(new llama_file(fname_split, "rb", use_direct_io)); files.emplace_back(new llama_file(fname_split, "rb", use_direct_io));
contexts.emplace_back(ctx); contexts.emplace_back(ctx);
longhaul_shards.push_back({
std::filesystem::path(fname_split).filename().string(),
files.back()->size(),
gguf_get_data_offset(ctx_gguf.get()),
});
// Save tensors data offset info of the shard. // Save tensors data offset info of the shard.
for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) { for (ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
-8
View File
@@ -9,7 +9,6 @@
#include "ggml-cpp.h" #include "ggml-cpp.h"
#include <array>
#include <cstddef> #include <cstddef>
#include <cstring> #include <cstring>
#include <map> #include <map>
@@ -78,12 +77,6 @@ struct llama_model_loader {
int layer; int layer;
}; };
struct longhaul_shard {
std::string name;
uint64_t size;
uint64_t metadata_size;
};
int n_kv = 0; int n_kv = 0;
int n_tensors = 0; int n_tensors = 0;
int n_created = 0; int n_created = 0;
@@ -122,7 +115,6 @@ struct llama_model_loader {
std::vector<std::pair<size_t, size_t>> mmaps_used; std::vector<std::pair<size_t, size_t>> mmaps_used;
size_t longhaul_slots = 0; size_t longhaul_slots = 0;
std::vector<longhaul_source> longhaul_sources; std::vector<longhaul_source> longhaul_sources;
std::vector<longhaul_shard> longhaul_shards;
// define a comparator for the buft -> ctx map to ensure that the order is well-defined: // define a comparator for the buft -> ctx map to ensure that the order is well-defined:
struct ggml_backend_buft_comparator { struct ggml_backend_buft_comparator {
+10 -30
View File
@@ -1039,7 +1039,6 @@ struct llama_model::impl {
std::vector<float> tensor_split_owned; std::vector<float> tensor_split_owned;
std::unique_ptr<llama_longhaul_cache> longhaul; std::unique_ptr<llama_longhaul_cache> longhaul;
bool longhaul_remote = false;
}; };
llama_model::llama_model(const llama_model_params & params) : params(params), pimpl(std::make_unique<impl>()) { llama_model::llama_model(const llama_model_params & params) : params(params), pimpl(std::make_unique<impl>()) {
@@ -1250,7 +1249,7 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
void llama_model_base::load_vocab(llama_model_loader & ml) { void llama_model_base::load_vocab(llama_model_loader & ml) {
const auto kv = LLM_KV(arch); const auto kv = LLM_KV(arch);
vocab.load(ml, kv); vocab.load(ml, kv, params.tokenizer_longhaul, params.tokenizer_longhaul_cache_bytes);
} }
bool llama_model_base::load_tensors(llama_model_loader & ml) { bool llama_model_base::load_tensors(llama_model_loader & ml) {
@@ -1357,6 +1356,9 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
} }
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) { if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
#if !defined(__APPLE__)
throw std::runtime_error("longhaul is only supported on macOS");
#endif
if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA) { if (arch != LLM_ARCH_QWEN35MOE && arch != LLM_ARCH_LAGUNA) {
throw std::runtime_error("longhaul currently requires a qwen35moe or laguna model"); throw std::runtime_error("longhaul currently requires a qwen35moe or laguna model");
} }
@@ -1366,35 +1368,12 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
if (params.check_tensors || params.no_alloc || params.vocab_only) { if (params.check_tensors || params.no_alloc || params.vocab_only) {
throw std::runtime_error("longhaul does not support check-tensors, no-alloc, or vocab-only loading"); throw std::runtime_error("longhaul does not support check-tensors, no-alloc, or vocab-only loading");
} }
bool all_metal = true;
bool all_rpc = true;
for (int il = 0; il < n_layer_all; ++il) { for (int il = 0; il < n_layer_all; ++il) {
ggml_backend_dev_t dev = pimpl->dev_layer[il].dev; ggml_backend_dev_t dev = pimpl->dev_layer[il].dev;
const char * reg_name = ggml_backend_reg_name(ggml_backend_dev_backend_reg(dev)); if (ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_GPU ||
all_metal = all_metal && strcmp(ggml_backend_reg_name(ggml_backend_dev_backend_reg(dev)), "MTL") != 0) {
ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_GPU && throw std::runtime_error("longhaul requires every model layer on Metal");
strcmp(reg_name, "MTL") == 0;
all_rpc = all_rpc &&
ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_GPU &&
strcmp(reg_name, "RPC") == 0;
} }
if (!all_metal && !all_rpc) {
throw std::runtime_error(
"longhaul requires all model layers on local Metal or one RPC device");
}
if (all_metal) {
#if !defined(__APPLE__)
throw std::runtime_error("local longhaul is only supported on macOS");
#endif
} else {
ggml_backend_dev_t first = pimpl->dev_layer.front().dev;
for (int il = 1; il < n_layer_all; ++il) {
if (pimpl->dev_layer[il].dev != first) {
throw std::runtime_error(
"RPC longhaul v1 requires every model layer on one RPC device");
}
}
pimpl->longhaul_remote = true;
} }
ml.configure_longhaul(params.longhaul_cache_bytes, n_expert, n_layer_all); ml.configure_longhaul(params.longhaul_cache_bytes, n_expert, n_layer_all);
if (ml.longhaul_slots < (size_t) n_expert_used) { if (ml.longhaul_slots < (size_t) n_expert_used) {
@@ -1715,8 +1694,7 @@ bool llama_model_base::load_tensors(llama_model_loader & ml) {
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) { if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL) {
pimpl->longhaul = std::make_unique<llama_longhaul_cache>( pimpl->longhaul = std::make_unique<llama_longhaul_cache>(
std::move(ml.files), std::move(ml.longhaul_sources), std::move(ml.longhaul_shards), std::move(ml.files), std::move(ml.longhaul_sources), ml.longhaul_slots, hparams.n_expert, hparams.n_layer_all);
ml.longhaul_slots, hparams.n_expert, hparams.n_layer_all, pimpl->longhaul_remote);
} }
return true; return true;
@@ -2402,6 +2380,8 @@ llama_model_params llama_model_default_params() {
/*.no_host =*/ false, /*.no_host =*/ false,
/*.no_alloc =*/ false, /*.no_alloc =*/ false,
/*.longhaul_cache_bytes =*/ 0, /*.longhaul_cache_bytes =*/ 0,
/*.tokenizer_longhaul =*/ false,
/*.tokenizer_longhaul_cache_bytes =*/ 0,
}; };
return result; return result;
+77 -1
View File
@@ -214,6 +214,10 @@ public:
if (!ensure_open()) return params.capacity_bytes == 0; if (!ensure_open()) return params.capacity_bytes == 0;
std::lock_guard<std::mutex> db_lock(db_mutex); std::lock_guard<std::mutex> db_lock(db_mutex);
if (sqlite3_exec(db, "DELETE FROM entries; PRAGMA incremental_vacuum;", nullptr, nullptr, nullptr) != SQLITE_OK) return db_error(); if (sqlite3_exec(db, "DELETE FROM entries; PRAGMA incremental_vacuum;", nullptr, nullptr, nullptr) != SQLITE_OK) return db_error();
std::error_code ec;
fs::remove_all(fs::path(path).parent_path() / "tokenizer-indexes", ec);
if (ec) return false;
sqlite3_exec(db, "INSERT OR REPLACE INTO meta(key,value) VALUES('external_bytes',0)", nullptr, nullptr, nullptr);
return true; return true;
} }
@@ -239,6 +243,11 @@ public:
result.bytes_used = sqlite3_column_int64(stmt, 1); result.bytes_used = sqlite3_column_int64(stmt, 1);
} }
sqlite3_finalize(stmt); sqlite3_finalize(stmt);
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='external_bytes'", -1, &stmt, nullptr) == SQLITE_OK &&
sqlite3_step(stmt) == SQLITE_ROW) {
result.bytes_used += static_cast<uint64_t>(sqlite3_column_int64(stmt, 0));
}
sqlite3_finalize(stmt);
} }
return result; return result;
} }
@@ -247,6 +256,45 @@ public:
if (hit) ++stats.tokenizer_hits; else ++stats.tokenizer_misses; if (hit) ++stats.tokenizer_hits; else ++stats.tokenizer_misses;
} }
std::string database_path() {
return ensure_open() ? path : std::string();
}
uint64_t capacity() const {
return params.capacity_bytes;
}
void set_external_bytes(uint64_t bytes) {
if (!ensure_open()) return;
std::lock_guard<std::mutex> db_lock(db_mutex);
sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr);
sqlite3_stmt * stmt = nullptr;
sqlite3_prepare_v2(db, "INSERT OR REPLACE INTO meta(key,value) VALUES('external_bytes',?1)", -1, &stmt, nullptr);
sqlite3_bind_int64(stmt, 1, static_cast<sqlite3_int64>(bytes));
sqlite3_step(stmt);
sqlite3_finalize(stmt);
uint64_t total = 0;
sqlite3_prepare_v2(db, "SELECT COALESCE(SUM(size),0) FROM entries", -1, &stmt, nullptr);
if (sqlite3_step(stmt) == SQLITE_ROW) total = sqlite3_column_int64(stmt, 0);
sqlite3_finalize(stmt);
const uint64_t entry_capacity = bytes >= params.capacity_bytes ? 0 : params.capacity_bytes - bytes;
while (total > entry_capacity) {
uint64_t victim_size = 0;
sqlite3_prepare_v2(db, "SELECT size FROM entries ORDER BY last_access LIMIT 1", -1, &stmt, nullptr);
if (sqlite3_step(stmt) == SQLITE_ROW) victim_size = sqlite3_column_int64(stmt, 0);
sqlite3_finalize(stmt);
if (victim_size == 0 || sqlite3_exec(db,
"DELETE FROM entries WHERE rowid=(SELECT rowid FROM entries ORDER BY last_access LIMIT 1)",
nullptr, nullptr, nullptr) != SQLITE_OK) {
break;
}
total -= std::min(total, victim_size);
++stats.evictions;
}
sqlite3_exec(db, "COMMIT", nullptr, nullptr, nullptr);
}
private: private:
static int64_t now_tick() { static int64_t now_tick() {
return std::chrono::duration_cast<std::chrono::microseconds>( return std::chrono::duration_cast<std::chrono::microseconds>(
@@ -311,13 +359,22 @@ private:
sqlite3_bind_int64(stmt, 1, static_cast<sqlite3_int64>(params.capacity_bytes)); sqlite3_bind_int64(stmt, 1, static_cast<sqlite3_int64>(params.capacity_bytes));
sqlite3_step(stmt); sqlite3_step(stmt);
sqlite3_finalize(stmt); sqlite3_finalize(stmt);
uint64_t external_bytes = 0;
sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='external_bytes'", -1, &stmt, nullptr);
if (sqlite3_step(stmt) == SQLITE_ROW) {
external_bytes = static_cast<uint64_t>(sqlite3_column_int64(stmt, 0));
}
sqlite3_finalize(stmt);
const uint64_t entry_capacity = external_bytes >= params.capacity_bytes
? 0
: params.capacity_bytes - external_bytes;
sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr); sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr);
while (true) { while (true) {
sqlite3_prepare_v2(db, "SELECT COALESCE(SUM(size),0) FROM entries", -1, &stmt, nullptr); sqlite3_prepare_v2(db, "SELECT COALESCE(SUM(size),0) FROM entries", -1, &stmt, nullptr);
uint64_t total = 0; uint64_t total = 0;
if (sqlite3_step(stmt) == SQLITE_ROW) total = sqlite3_column_int64(stmt, 0); if (sqlite3_step(stmt) == SQLITE_ROW) total = sqlite3_column_int64(stmt, 0);
sqlite3_finalize(stmt); sqlite3_finalize(stmt);
if (total <= params.capacity_bytes) break; if (total <= entry_capacity) break;
if (sqlite3_exec(db, if (sqlite3_exec(db,
"DELETE FROM entries WHERE rowid=(SELECT rowid FROM entries ORDER BY last_access LIMIT 1)", "DELETE FROM entries WHERE rowid=(SELECT rowid FROM entries ORDER BY last_access LIMIT 1)",
nullptr, nullptr, nullptr) != SQLITE_OK) { nullptr, nullptr, nullptr) != SQLITE_OK) {
@@ -355,12 +412,19 @@ private:
std::lock_guard<std::mutex> db_lock(db_mutex); std::lock_guard<std::mutex> db_lock(db_mutex);
if (sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr) != SQLITE_OK) { db_error(); return; } if (sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr) != SQLITE_OK) { db_error(); return; }
uint64_t capacity = params.capacity_bytes; uint64_t capacity = params.capacity_bytes;
uint64_t external_bytes = 0;
sqlite3_stmt * capacity_stmt = nullptr; sqlite3_stmt * capacity_stmt = nullptr;
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='capacity'", -1, &capacity_stmt, nullptr) == SQLITE_OK && if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='capacity'", -1, &capacity_stmt, nullptr) == SQLITE_OK &&
sqlite3_step(capacity_stmt) == SQLITE_ROW) { sqlite3_step(capacity_stmt) == SQLITE_ROW) {
capacity = static_cast<uint64_t>(sqlite3_column_int64(capacity_stmt, 0)); capacity = static_cast<uint64_t>(sqlite3_column_int64(capacity_stmt, 0));
} }
sqlite3_finalize(capacity_stmt); sqlite3_finalize(capacity_stmt);
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='external_bytes'", -1, &capacity_stmt, nullptr) == SQLITE_OK &&
sqlite3_step(capacity_stmt) == SQLITE_ROW) {
external_bytes = static_cast<uint64_t>(sqlite3_column_int64(capacity_stmt, 0));
}
sqlite3_finalize(capacity_stmt);
capacity = external_bytes >= capacity ? 0 : capacity - external_bytes;
if (item.payload.size() > capacity) { if (item.payload.size() > capacity) {
sqlite3_exec(db, "ROLLBACK", nullptr, nullptr, nullptr); sqlite3_exec(db, "ROLLBACK", nullptr, nullptr, nullptr);
return; return;
@@ -492,6 +556,18 @@ void llama_token_cache_note_tokenizer_hit(bool hit) {
cache().note_tokenizer(hit); cache().note_tokenizer(hit);
} }
std::string llama_token_cache_database_path() {
return cache().database_path();
}
uint64_t llama_token_cache_capacity() {
return cache().capacity();
}
void llama_token_cache_set_external_bytes(uint64_t bytes) {
cache().set_external_bytes(bytes);
}
llama_token_cache_params llama_token_cache_default_params() { llama_token_cache_params llama_token_cache_default_params() {
return {nullptr, DEFAULT_CAPACITY}; return {nullptr, DEFAULT_CAPACITY};
} }
+5
View File
@@ -27,3 +27,8 @@ void llama_token_cache_store(
void llama_token_cache_note_tokenizer_hit(bool hit); void llama_token_cache_note_tokenizer_hit(bool hit);
// Internal helpers used by tokenizer-longhaul artifacts.
// Opening the cache here also fixes the configured path for the process.
std::string llama_token_cache_database_path();
uint64_t llama_token_cache_capacity();
void llama_token_cache_set_external_bytes(uint64_t bytes);
+471
View File
@@ -0,0 +1,471 @@
#include "llama-tokenizer-longhaul.h"
#include "llama-token-cache.h"
#include <sqlite3.h>
#include <algorithm>
#include <chrono>
#include <filesystem>
#include <list>
#include <mutex>
#include <thread>
#include <unordered_map>
#if defined(_WIN32)
# include <process.h>
#else
# include <unistd.h>
#endif
namespace fs = std::filesystem;
namespace {
constexpr int INDEX_VERSION = 1;
int process_id() {
#if defined(_WIN32)
return _getpid();
#else
return getpid();
#endif
}
bool exec_sql(sqlite3 * db, const char * sql, std::string & error) {
char * message = nullptr;
if (sqlite3_exec(db, sql, nullptr, nullptr, &message) == SQLITE_OK) {
return true;
}
error = message ? message : sqlite3_errmsg(db);
sqlite3_free(message);
return false;
}
struct lock_directory {
fs::path path;
bool owned = false;
~lock_directory() {
if (owned) {
std::error_code ec;
fs::remove(path, ec);
}
}
};
uint64_t tokenizer_artifact_bytes(const fs::path & directory) {
uint64_t total = 0;
std::error_code ec;
for (const auto & entry : fs::directory_iterator(directory, ec)) {
if (ec) break;
if (!entry.is_regular_file(ec) || entry.path().extension() != ".sqlite3") continue;
total += entry.file_size(ec);
if (ec) {
ec.clear();
}
}
return total;
}
} // namespace
struct llama_tokenizer_longhaul_index::impl {
struct cache_entry {
int value;
uint64_t bytes;
std::list<std::string>::iterator position;
};
std::string source_key;
uint64_t capacity;
fs::path artifact;
sqlite3 * db = nullptr;
std::string id;
uint64_t merge_count = 0;
mutable std::mutex mutex;
mutable uint64_t used = 0;
mutable std::list<std::string> lru;
mutable std::unordered_map<std::string, cache_entry> values;
impl(std::string source_key, uint64_t capacity)
: source_key(std::move(source_key)), capacity(capacity) {
}
~impl() {
if (db) {
sqlite3_close(db);
}
}
bool open_existing(std::string & error) {
if (!fs::exists(artifact)) {
return false;
}
sqlite3 * opened = nullptr;
if (sqlite3_open_v2(artifact.string().c_str(), &opened,
SQLITE_OPEN_READONLY | SQLITE_OPEN_FULLMUTEX, nullptr) != SQLITE_OK) {
error = opened ? sqlite3_errmsg(opened) : "failed to open tokenizer index";
if (opened) sqlite3_close(opened);
return false;
}
sqlite3_stmt * stmt = nullptr;
const char * sql =
"SELECT version, tokenizer_id, n_merges FROM metadata WHERE complete=1 LIMIT 1";
bool valid = sqlite3_prepare_v2(opened, sql, -1, &stmt, nullptr) == SQLITE_OK &&
sqlite3_step(stmt) == SQLITE_ROW &&
sqlite3_column_int(stmt, 0) == INDEX_VERSION;
if (valid) {
const unsigned char * value = sqlite3_column_text(stmt, 1);
id = value ? reinterpret_cast<const char *>(value) : "";
merge_count = static_cast<uint64_t>(sqlite3_column_int64(stmt, 2));
valid = !id.empty();
}
sqlite3_finalize(stmt);
if (!valid) {
error = "tokenizer index is incomplete or has an unsupported version";
sqlite3_close(opened);
return false;
}
db = opened;
sqlite3_busy_timeout(db, 5000);
return true;
}
bool build(
const std::vector<llama_tokenizer_longhaul_reverse> & reverse,
const std::vector<llama_tokenizer_longhaul_merge> & merges,
const std::string & tokenizer_id,
std::string & error) {
const fs::path temp = artifact.string() + ".tmp." + std::to_string(process_id());
std::error_code ec;
fs::remove(temp, ec);
sqlite3 * output = nullptr;
if (sqlite3_open_v2(temp.string().c_str(), &output,
SQLITE_OPEN_READWRITE | SQLITE_OPEN_CREATE | SQLITE_OPEN_EXCLUSIVE, nullptr) != SQLITE_OK) {
error = output ? sqlite3_errmsg(output) : "failed to create tokenizer index";
if (output) sqlite3_close(output);
return false;
}
auto fail = [&](const std::string & message) {
error = message;
sqlite3_close(output);
output = nullptr;
std::error_code remove_ec;
fs::remove(temp, remove_ec);
return false;
};
if (!exec_sql(output,
"PRAGMA journal_mode=OFF;"
"PRAGMA synchronous=OFF;"
"PRAGMA locking_mode=EXCLUSIVE;"
"CREATE TABLE metadata("
" version INTEGER NOT NULL,complete INTEGER NOT NULL,tokenizer_id TEXT NOT NULL,"
" n_reverse INTEGER NOT NULL,n_merges INTEGER NOT NULL);"
"CREATE TABLE reverse_tokens(text BLOB PRIMARY KEY,id INTEGER NOT NULL) WITHOUT ROWID;"
"CREATE TABLE merges(left_text BLOB NOT NULL,right_text BLOB NOT NULL,rank INTEGER NOT NULL,"
" PRIMARY KEY(left_text,right_text)) WITHOUT ROWID;"
"BEGIN IMMEDIATE;", error)) {
return fail(error);
}
sqlite3_stmt * insert_reverse = nullptr;
sqlite3_stmt * insert_merge = nullptr;
if (sqlite3_prepare_v2(output,
"INSERT INTO reverse_tokens(text,id) VALUES(?1,?2)", -1, &insert_reverse, nullptr) != SQLITE_OK ||
sqlite3_prepare_v2(output,
"INSERT INTO merges(left_text,right_text,rank) VALUES(?1,?2,?3)",
-1, &insert_merge, nullptr) != SQLITE_OK) {
if (insert_reverse) sqlite3_finalize(insert_reverse);
if (insert_merge) sqlite3_finalize(insert_merge);
return fail(sqlite3_errmsg(output));
}
for (const auto & item : reverse) {
sqlite3_bind_blob(insert_reverse, 1, item.text.data(), int(item.text.size()), SQLITE_STATIC);
sqlite3_bind_int(insert_reverse, 2, item.id);
if (sqlite3_step(insert_reverse) != SQLITE_DONE) {
const std::string message = sqlite3_errmsg(output);
sqlite3_finalize(insert_reverse);
sqlite3_finalize(insert_merge);
return fail(message);
}
sqlite3_reset(insert_reverse);
sqlite3_clear_bindings(insert_reverse);
}
for (const auto & item : merges) {
sqlite3_bind_blob(insert_merge, 1, item.left.data(), int(item.left.size()), SQLITE_STATIC);
sqlite3_bind_blob(insert_merge, 2, item.right.data(), int(item.right.size()), SQLITE_STATIC);
sqlite3_bind_int(insert_merge, 3, item.rank);
if (sqlite3_step(insert_merge) != SQLITE_DONE) {
const std::string message = sqlite3_errmsg(output);
sqlite3_finalize(insert_reverse);
sqlite3_finalize(insert_merge);
return fail(message);
}
sqlite3_reset(insert_merge);
sqlite3_clear_bindings(insert_merge);
}
sqlite3_finalize(insert_reverse);
sqlite3_finalize(insert_merge);
sqlite3_stmt * metadata = nullptr;
if (sqlite3_prepare_v2(output,
"INSERT INTO metadata VALUES(?1,1,?2,?3,?4)", -1, &metadata, nullptr) != SQLITE_OK) {
return fail(sqlite3_errmsg(output));
}
sqlite3_bind_int(metadata, 1, INDEX_VERSION);
sqlite3_bind_text(metadata, 2, tokenizer_id.c_str(), -1, SQLITE_STATIC);
sqlite3_bind_int64(metadata, 3, static_cast<sqlite3_int64>(reverse.size()));
sqlite3_bind_int64(metadata, 4, static_cast<sqlite3_int64>(merges.size()));
const bool metadata_ok = sqlite3_step(metadata) == SQLITE_DONE;
sqlite3_finalize(metadata);
if (!metadata_ok || !exec_sql(output, "COMMIT;", error)) {
return fail(metadata_ok ? error : sqlite3_errmsg(output));
}
sqlite3_close(output);
output = nullptr;
const uint64_t persistent_capacity = llama_token_cache_capacity();
const uint64_t artifact_size = fs::file_size(temp, ec);
if (ec || artifact_size > persistent_capacity) {
fs::remove(temp, ec);
error = "completed tokenizer index does not fit in --token-cache-size";
return false;
}
struct artifact_candidate {
fs::path path;
fs::file_time_type modified;
uint64_t size;
};
std::vector<artifact_candidate> candidates;
uint64_t existing_bytes = 0;
for (const auto & entry : fs::directory_iterator(artifact.parent_path(), ec)) {
if (ec) break;
if (!entry.is_regular_file(ec) || entry.path().extension() != ".sqlite3" ||
entry.path() == artifact || entry.path() == temp) {
continue;
}
const uint64_t size = entry.file_size(ec);
if (ec) {
ec.clear();
continue;
}
existing_bytes += size;
candidates.push_back({entry.path(), entry.last_write_time(ec), size});
if (ec) ec.clear();
}
std::sort(candidates.begin(), candidates.end(),
[](const artifact_candidate & a, const artifact_candidate & b) {
return a.modified < b.modified;
});
for (const auto & candidate : candidates) {
if (existing_bytes + artifact_size <= persistent_capacity) break;
fs::remove(candidate.path, ec);
if (!ec) existing_bytes -= std::min(existing_bytes, candidate.size);
ec.clear();
}
if (existing_bytes + artifact_size > persistent_capacity) {
fs::remove(temp, ec);
error = "tokenizer indexes do not fit in --token-cache-size";
return false;
}
fs::rename(temp, artifact, ec);
if (ec) {
const std::string rename_error = ec.message();
// A competing process may have published the same immutable artifact.
std::error_code remove_ec;
fs::remove(temp, remove_ec);
std::string ignored;
if (open_existing(ignored)) {
return true;
}
error = "failed to publish tokenizer index: " + rename_error;
return false;
}
const bool opened = open_existing(error);
if (opened) {
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(artifact.parent_path()));
}
return opened;
}
int lookup(const std::string & cache_key, const char * sql,
const std::string & first, const std::string * second, int missing) const {
std::lock_guard<std::mutex> lock(mutex);
auto found = values.find(cache_key);
if (found != values.end()) {
lru.splice(lru.begin(), lru, found->second.position);
return found->second.value;
}
sqlite3_stmt * stmt = nullptr;
int value = missing;
if (sqlite3_prepare_v2(db, sql, -1, &stmt, nullptr) == SQLITE_OK) {
sqlite3_bind_blob(stmt, 1, first.data(), int(first.size()), SQLITE_STATIC);
if (second) {
sqlite3_bind_blob(stmt, 2, second->data(), int(second->size()), SQLITE_STATIC);
}
if (sqlite3_step(stmt) == SQLITE_ROW) {
value = sqlite3_column_int(stmt, 0);
}
}
sqlite3_finalize(stmt);
const uint64_t bytes = cache_key.size() + sizeof(cache_entry) + sizeof(std::string);
if (bytes <= capacity) {
while (!lru.empty() && used + bytes > capacity) {
auto victim = values.find(lru.back());
if (victim != values.end()) {
used -= victim->second.bytes;
values.erase(victim);
}
lru.pop_back();
}
lru.push_front(cache_key);
values.emplace(cache_key, cache_entry{value, bytes, lru.begin()});
used += bytes;
}
return value;
}
};
llama_tokenizer_longhaul_index::llama_tokenizer_longhaul_index(
std::string source_key, uint64_t cache_bytes)
: pimpl(std::make_unique<impl>(std::move(source_key), cache_bytes)) {
}
llama_tokenizer_longhaul_index::~llama_tokenizer_longhaul_index() = default;
bool llama_tokenizer_longhaul_index::prepare(
const std::vector<llama_tokenizer_longhaul_reverse> & reverse,
const std::vector<llama_tokenizer_longhaul_merge> & merges,
const std::string & tokenizer_id,
std::string & error) {
const std::string cache_db = llama_token_cache_database_path();
if (cache_db.empty()) {
error = "persistent token cache is disabled or unavailable";
return false;
}
const fs::path directory = fs::path(cache_db).parent_path() / "tokenizer-indexes";
std::error_code ec;
fs::create_directories(directory, ec);
if (ec) {
error = "failed to create tokenizer index directory: " + ec.message();
return false;
}
pimpl->artifact = directory / (llama_token_cache_hash(pimpl->source_key) + ".sqlite3");
if (pimpl->open_existing(error)) {
if (pimpl->id == tokenizer_id) {
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(directory));
return true;
}
sqlite3_close(pimpl->db);
pimpl->db = nullptr;
std::error_code stale_ec;
fs::remove(pimpl->artifact, stale_ec);
}
lock_directory lock{pimpl->artifact.string() + ".lock", false};
lock.owned = fs::create_directory(lock.path, ec);
if (!lock.owned) {
// Another process is building. Wait for its atomic publication.
for (int i = 0; i < 24000; ++i) {
std::this_thread::sleep_for(std::chrono::milliseconds(25));
std::string ignored;
if (pimpl->open_existing(ignored)) {
if (pimpl->id == tokenizer_id) {
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(directory));
return true;
}
sqlite3_close(pimpl->db);
pimpl->db = nullptr;
}
if (!fs::exists(lock.path)) {
lock.owned = fs::create_directory(lock.path, ec);
if (lock.owned) break;
}
}
if (!lock.owned) {
error = "timed out waiting for another tokenizer index builder";
return false;
}
}
error.clear();
if (pimpl->open_existing(error)) {
if (pimpl->id == tokenizer_id) {
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(directory));
return true;
}
sqlite3_close(pimpl->db);
pimpl->db = nullptr;
}
std::error_code remove_ec;
fs::remove(pimpl->artifact, remove_ec);
return pimpl->build(reverse, merges, tokenizer_id, error);
}
llama_token llama_tokenizer_longhaul_index::text_to_token(const std::string & text) const {
const std::string key = "t" + text;
return pimpl->lookup(key,
"SELECT id FROM reverse_tokens WHERE text=?1", text, nullptr, LLAMA_TOKEN_NULL);
}
int llama_tokenizer_longhaul_index::find_bpe_rank(
const std::string & left, const std::string & right) const {
std::string key;
key.reserve(left.size() + right.size() + 2);
key.push_back('m');
key += left;
key.push_back('\0');
key += right;
return pimpl->lookup(key,
"SELECT rank FROM merges WHERE left_text=?1 AND right_text=?2", left, &right, -1);
}
std::vector<std::string> llama_tokenizer_longhaul_index::get_bpe_merges() const {
std::lock_guard<std::mutex> lock(pimpl->mutex);
std::vector<std::string> result;
result.reserve(pimpl->merge_count);
sqlite3_stmt * stmt = nullptr;
if (sqlite3_prepare_v2(pimpl->db,
"SELECT left_text,right_text FROM merges ORDER BY rank", -1, &stmt, nullptr) != SQLITE_OK) {
return result;
}
while (sqlite3_step(stmt) == SQLITE_ROW) {
const char * left = static_cast<const char *>(sqlite3_column_blob(stmt, 0));
const int left_size = sqlite3_column_bytes(stmt, 0);
const char * right = static_cast<const char *>(sqlite3_column_blob(stmt, 1));
const int right_size = sqlite3_column_bytes(stmt, 1);
std::string item(left, left_size);
item.push_back(' ');
item.append(right, right_size);
result.push_back(std::move(item));
}
sqlite3_finalize(stmt);
return result;
}
uint64_t llama_tokenizer_longhaul_index::cache_bytes_used() const {
std::lock_guard<std::mutex> lock(pimpl->mutex);
return pimpl->used;
}
uint64_t llama_tokenizer_longhaul_index::cache_capacity() const {
return pimpl->capacity;
}
uint64_t llama_tokenizer_longhaul_index::n_merges() const {
return pimpl->merge_count;
}
const std::string & llama_tokenizer_longhaul_index::tokenizer_id() const {
return pimpl->id;
}
+47
View File
@@ -0,0 +1,47 @@
#pragma once
#include "llama.h"
#include <cstdint>
#include <memory>
#include <string>
#include <vector>
struct llama_tokenizer_longhaul_reverse {
std::string text;
llama_token id;
};
struct llama_tokenizer_longhaul_merge {
std::string left;
std::string right;
int rank;
};
class llama_tokenizer_longhaul_index {
public:
llama_tokenizer_longhaul_index(std::string source_key, uint64_t cache_bytes);
~llama_tokenizer_longhaul_index();
llama_tokenizer_longhaul_index(const llama_tokenizer_longhaul_index &) = delete;
llama_tokenizer_longhaul_index & operator=(const llama_tokenizer_longhaul_index &) = delete;
bool prepare(
const std::vector<llama_tokenizer_longhaul_reverse> & reverse,
const std::vector<llama_tokenizer_longhaul_merge> & merges,
const std::string & tokenizer_id,
std::string & error);
llama_token text_to_token(const std::string & text) const;
int find_bpe_rank(const std::string & left, const std::string & right) const;
std::vector<std::string> get_bpe_merges() const;
uint64_t cache_bytes_used() const;
uint64_t cache_capacity() const;
uint64_t n_merges() const;
const std::string & tokenizer_id() const;
private:
struct impl;
std::unique_ptr<impl> pimpl;
};
+193 -16
View File
@@ -1,5 +1,6 @@
#include "llama-vocab.h" #include "llama-vocab.h"
#include "llama-token-cache.h" #include "llama-token-cache.h"
#include "llama-tokenizer-longhaul.h"
#include "ggml.h" #include "ggml.h"
#include "gguf.h" #include "gguf.h"
@@ -13,13 +14,16 @@
#include <cctype> #include <cctype>
#include <cfloat> #include <cfloat>
#include <cmath> #include <cmath>
#include <condition_variable>
#include <cstdarg> #include <cstdarg>
#include <cstring> #include <cstring>
#include <forward_list> #include <forward_list>
#include <limits> #include <limits>
#include <map> #include <map>
#include <mutex>
#include <queue> #include <queue>
#include <set> #include <set>
#include <thread>
#include <unordered_map> #include <unordered_map>
// //
@@ -1846,12 +1850,37 @@ struct llama_vocab::impl {
std::vector<char> precompiled_charsmap; std::vector<char> precompiled_charsmap;
enum class longhaul_state {
disabled,
building,
ready,
failed,
};
bool tokenizer_longhaul = false;
uint64_t tokenizer_longhaul_cache_bytes = 0;
uint64_t tokenizer_longhaul_n_merges = 0;
mutable std::mutex tokenizer_longhaul_mutex;
mutable std::condition_variable tokenizer_longhaul_ready;
mutable longhaul_state tokenizer_longhaul_state = longhaul_state::disabled;
mutable std::string tokenizer_longhaul_error;
mutable std::unique_ptr<llama_tokenizer_longhaul_index> tokenizer_longhaul_index;
std::thread tokenizer_longhaul_worker;
impl(const llama_vocab & vocab) : vocab(vocab) { impl(const llama_vocab & vocab) : vocab(vocab) {
} }
~impl() = default; ~impl() {
if (tokenizer_longhaul_worker.joinable()) {
tokenizer_longhaul_worker.join();
}
}
void load(llama_model_loader & ml, const LLM_KV & kv); void load(
llama_model_loader & ml,
const LLM_KV & kv,
bool tokenizer_longhaul,
uint64_t tokenizer_longhaul_cache_bytes);
void wait_for_tokenizer_longhaul() const;
bool load_cache_snapshot(const std::vector<uint8_t> & snapshot); bool load_cache_snapshot(const std::vector<uint8_t> & snapshot);
void reset_cache_snapshot_state(); void reset_cache_snapshot_state();
@@ -1908,7 +1937,7 @@ struct llama_vocab::impl {
bool special) const; bool special) const;
// use cached data // use cached data
const std::string & token_to_piece(llama_token token) const; std::string token_to_piece(llama_token token) const;
int32_t detokenize( int32_t detokenize(
const llama_token * tokens, const llama_token * tokens,
@@ -1928,11 +1957,29 @@ private:
const llama_vocab & vocab; const llama_vocab & vocab;
}; };
void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) { void llama_vocab::impl::load(
llama_model_loader & ml,
const LLM_KV & kv,
bool tokenizer_longhaul_value,
uint64_t tokenizer_longhaul_cache_bytes_value) {
struct gguf_context * ctx = ml.metadata; struct gguf_context * ctx = ml.metadata;
tokenizer_longhaul = tokenizer_longhaul_value;
tokenizer_longhaul_cache_bytes = tokenizer_longhaul_cache_bytes_value;
if (tokenizer_longhaul) {
if (tokenizer_longhaul_cache_bytes == 0) {
throw std::runtime_error("tokenizer-longhaul cache budget must be positive");
}
if (ml.get_arch() != LLM_ARCH_QWEN35MOE && ml.get_arch() != LLM_ARCH_LAGUNA) {
throw std::runtime_error("tokenizer-longhaul currently requires a qwen35moe or laguna model");
}
if (ml.get_token_cache_source_key().empty()) {
throw std::runtime_error("tokenizer-longhaul requires a file-backed GGUF model");
}
}
const std::string & source_cache_key = ml.get_token_cache_source_key(); const std::string & source_cache_key = ml.get_token_cache_source_key();
if (!source_cache_key.empty()) { if (!tokenizer_longhaul && !source_cache_key.empty()) {
std::vector<uint8_t> snapshot; std::vector<uint8_t> snapshot;
bool found = llama_token_cache_lookup(LLAMA_TOKEN_CACHE_KIND_TOKENIZER, source_cache_key, snapshot); bool found = llama_token_cache_lookup(LLAMA_TOKEN_CACHE_KIND_TOKENIZER, source_cache_key, snapshot);
if (found && snapshot.size() == 65 && snapshot[0] == 'A') { if (found && snapshot.size() == 65 && snapshot[0] == 'A') {
@@ -1958,6 +2005,9 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
ml.get_key(LLM_KV_TOKENIZER_TOKEN_TYPE_COUNT, n_token_types, false); ml.get_key(LLM_KV_TOKENIZER_TOKEN_TYPE_COUNT, n_token_types, false);
if (tokenizer_model == "no_vocab" || tokenizer_model == "none") { if (tokenizer_model == "no_vocab" || tokenizer_model == "none") {
if (tokenizer_longhaul) {
throw std::runtime_error("tokenizer-longhaul requires a BPE tokenizer");
}
type = LLAMA_VOCAB_TYPE_NONE; type = LLAMA_VOCAB_TYPE_NONE;
// default special tokens // default special tokens
@@ -3063,6 +3113,69 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
} }
} }
if (tokenizer_longhaul) {
if (type != LLAMA_VOCAB_TYPE_BPE ||
(tokenizer_pre != "qwen35" && tokenizer_pre != "laguna")) {
throw std::runtime_error(
"tokenizer-longhaul currently supports only qwen35 and laguna BPE tokenizers");
}
std::vector<llama_tokenizer_longhaul_reverse> reverse;
reverse.reserve(token_to_id.size());
while (!token_to_id.empty()) {
auto node = token_to_id.extract(token_to_id.begin());
reverse.push_back({std::move(node.key()), node.mapped()});
}
token_to_id.rehash(0);
std::vector<llama_tokenizer_longhaul_merge> merges;
merges.reserve(bpe_ranks.size());
while (!bpe_ranks.empty()) {
auto node = bpe_ranks.extract(bpe_ranks.begin());
auto key = std::move(node.key());
merges.push_back({std::move(key.first), std::move(key.second), node.mapped()});
}
bpe_ranks.rehash(0);
tokenizer_longhaul_n_merges = merges.size();
cache_token_to_piece.clear();
cache_token_to_piece.shrink_to_fit();
token_cache_id = llama_token_cache_hash(
std::string("tokenizer-longhaul-v1\n") + source_cache_key);
{
std::lock_guard<std::mutex> lock(tokenizer_longhaul_mutex);
tokenizer_longhaul_state = longhaul_state::building;
}
const uint64_t cache_bytes = tokenizer_longhaul_cache_bytes;
const std::string source_key = source_cache_key;
const std::string tokenizer_id = token_cache_id;
tokenizer_longhaul_worker = std::thread(
[this, source_key, cache_bytes, tokenizer_id,
reverse = std::move(reverse), merges = std::move(merges)]() mutable {
auto index = std::make_unique<llama_tokenizer_longhaul_index>(source_key, cache_bytes);
std::string error;
const bool ok = index->prepare(reverse, merges, tokenizer_id, error);
{
std::lock_guard<std::mutex> lock(tokenizer_longhaul_mutex);
if (ok) {
tokenizer_longhaul_index = std::move(index);
tokenizer_longhaul_state = longhaul_state::ready;
LLAMA_LOG_INFO("%s: tokenizer-longhaul index is ready\n", "llama_vocab");
} else {
tokenizer_longhaul_error = error.empty()
? "unknown tokenizer-longhaul build error"
: std::move(error);
tokenizer_longhaul_state = longhaul_state::failed;
LLAMA_LOG_ERROR("%s: tokenizer-longhaul index failed: %s\n",
"llama_vocab", tokenizer_longhaul_error.c_str());
}
}
tokenizer_longhaul_ready.notify_all();
});
LLAMA_LOG_INFO("%s: tokenizer-longhaul index is building in the background\n", __func__);
return;
}
// Persist a versioned canonical snapshot of the effective tokenizer state. // Persist a versioned canonical snapshot of the effective tokenizer state.
// The snapshot also serves as the namespace for text-to-token cache entries. // The snapshot also serves as the namespace for text-to-token cache entries.
{ {
@@ -3151,6 +3264,23 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
} }
} }
void llama_vocab::impl::wait_for_tokenizer_longhaul() const {
if (!tokenizer_longhaul) {
return;
}
std::unique_lock<std::mutex> lock(tokenizer_longhaul_mutex);
if (tokenizer_longhaul_state == longhaul_state::disabled) {
return;
}
tokenizer_longhaul_ready.wait(lock, [this] {
return tokenizer_longhaul_state == longhaul_state::ready ||
tokenizer_longhaul_state == longhaul_state::failed;
});
if (tokenizer_longhaul_state == longhaul_state::failed) {
throw std::runtime_error("tokenizer-longhaul failed: " + tokenizer_longhaul_error);
}
}
void llama_vocab::impl::reset_cache_snapshot_state() { void llama_vocab::impl::reset_cache_snapshot_state() {
n_token_types = 0; n_token_types = 0;
tokenizer_model.clear(); tokenizer_model.clear();
@@ -3282,36 +3412,43 @@ std::string llama_vocab::impl::type_name() const{
} }
bool llama_vocab::impl::is_normal(llama_token id) const { bool llama_vocab::impl::is_normal(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_NORMAL; return id_to_token[id].attr & LLAMA_TOKEN_ATTR_NORMAL;
} }
bool llama_vocab::impl::is_unknown(llama_token id) const { bool llama_vocab::impl::is_unknown(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNKNOWN; return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNKNOWN;
} }
bool llama_vocab::impl::is_control(llama_token id) const { bool llama_vocab::impl::is_control(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_CONTROL; return id_to_token[id].attr & LLAMA_TOKEN_ATTR_CONTROL;
} }
bool llama_vocab::impl::is_byte(llama_token id) const { bool llama_vocab::impl::is_byte(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_BYTE; return id_to_token[id].attr & LLAMA_TOKEN_ATTR_BYTE;
} }
bool llama_vocab::impl::is_user_defined(llama_token id) const { bool llama_vocab::impl::is_user_defined(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_USER_DEFINED; return id_to_token[id].attr & LLAMA_TOKEN_ATTR_USER_DEFINED;
} }
bool llama_vocab::impl::is_unused(llama_token id) const { bool llama_vocab::impl::is_unused(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNUSED; return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNUSED;
} }
bool llama_vocab::impl::is_eog(llama_token id) const { bool llama_vocab::impl::is_eog(llama_token id) const {
wait_for_tokenizer_longhaul();
return id != LLAMA_TOKEN_NULL && special_eog_ids.count(id) > 0; return id != LLAMA_TOKEN_NULL && special_eog_ids.count(id) > 0;
} }
@@ -3339,6 +3476,7 @@ uint8_t llama_vocab::impl::token_to_byte(llama_token id) const {
} }
llama_token_attr llama_vocab::impl::token_get_attr(llama_token id) const { llama_token_attr llama_vocab::impl::token_get_attr(llama_token id) const {
wait_for_tokenizer_longhaul();
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
return id_to_token.at(id).attr; return id_to_token.at(id).attr;
} }
@@ -3749,6 +3887,7 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
const std::string & raw_text, const std::string & raw_text,
bool add_special, bool add_special,
bool parse_special) const { bool parse_special) const {
wait_for_tokenizer_longhaul();
if (raw_text.empty() || token_cache_id.empty()) { if (raw_text.empty() || token_cache_id.empty()) {
return tokenize_uncached(raw_text, add_special, parse_special); return tokenize_uncached(raw_text, add_special, parse_special);
} }
@@ -3792,6 +3931,7 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
} }
int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t length, int32_t lstrip, bool special) const { int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t length, int32_t lstrip, bool special) const {
wait_for_tokenizer_longhaul();
// ref: https://github.com/ggml-org/llama.cpp/pull/7587#discussion_r1620983843 // ref: https://github.com/ggml-org/llama.cpp/pull/7587#discussion_r1620983843
static const int attr_special = LLAMA_TOKEN_ATTR_UNKNOWN | LLAMA_TOKEN_ATTR_CONTROL; static const int attr_special = LLAMA_TOKEN_ATTR_UNKNOWN | LLAMA_TOKEN_ATTR_CONTROL;
const llama_token_attr attr = token_get_attr(token); const llama_token_attr attr = token_get_attr(token);
@@ -3908,7 +4048,11 @@ int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t
return 0; return 0;
} }
const std::string & llama_vocab::impl::token_to_piece(llama_token token) const { std::string llama_vocab::impl::token_to_piece(llama_token token) const {
wait_for_tokenizer_longhaul();
if (tokenizer_longhaul) {
return token_to_piece_for_cache(token, true);
}
return cache_token_to_piece.at(token); return cache_token_to_piece.at(token);
} }
@@ -4027,7 +4171,15 @@ int32_t llama_vocab::impl::detokenize(
void llama_vocab::impl::print_info() const { void llama_vocab::impl::print_info() const {
LLAMA_LOG_INFO("%s: vocab type = %s\n", __func__, type_name().c_str()); LLAMA_LOG_INFO("%s: vocab type = %s\n", __func__, type_name().c_str());
LLAMA_LOG_INFO("%s: n_vocab = %u\n", __func__, vocab.n_tokens()); LLAMA_LOG_INFO("%s: n_vocab = %u\n", __func__, vocab.n_tokens());
LLAMA_LOG_INFO("%s: n_merges = %u\n", __func__, (uint32_t) bpe_ranks.size()); LLAMA_LOG_INFO("%s: n_merges = %u\n", __func__,
(uint32_t) (tokenizer_longhaul ? tokenizer_longhaul_n_merges : bpe_ranks.size()));
if (tokenizer_longhaul) {
std::lock_guard<std::mutex> lock(tokenizer_longhaul_mutex);
LLAMA_LOG_INFO("%s: tokenizer longhaul = %s, cache = %.2f MiB\n", __func__,
tokenizer_longhaul_state == longhaul_state::ready ? "ready" :
tokenizer_longhaul_state == longhaul_state::failed ? "failed" : "building",
tokenizer_longhaul_cache_bytes / 1024.0 / 1024.0);
}
// special tokens // special tokens
if (special_bos_id != LLAMA_TOKEN_NULL) { LLAMA_LOG_INFO( "%s: BOS token = %d '%s'\n", __func__, special_bos_id, id_to_token.at(special_bos_id).text.c_str() ); } if (special_bos_id != LLAMA_TOKEN_NULL) { LLAMA_LOG_INFO( "%s: BOS token = %d '%s'\n", __func__, special_bos_id, id_to_token.at(special_bos_id).text.c_str() ); }
@@ -4060,8 +4212,12 @@ llama_vocab::llama_vocab() : pimpl(new impl(*this)) {
llama_vocab::~llama_vocab() = default; llama_vocab::~llama_vocab() = default;
void llama_vocab::load(llama_model_loader & ml, const LLM_KV & kv) { void llama_vocab::load(
pimpl->load(ml, kv); llama_model_loader & ml,
const LLM_KV & kv,
bool tokenizer_longhaul,
uint64_t tokenizer_longhaul_cache_bytes) {
pimpl->load(ml, kv, tokenizer_longhaul, tokenizer_longhaul_cache_bytes);
} }
std::string llama_vocab::get_tokenizer_model() const { std::string llama_vocab::get_tokenizer_model() const {
@@ -4131,23 +4287,29 @@ llama_token llama_vocab::byte_to_token(uint8_t ch) const {
case LLAMA_VOCAB_TYPE_SPM: case LLAMA_VOCAB_TYPE_SPM:
case LLAMA_VOCAB_TYPE_UGM: { case LLAMA_VOCAB_TYPE_UGM: {
const char buf[7] = { '<', '0', 'x', hex[ch >> 4], hex[ch & 15], '>', 0 }; const char buf[7] = { '<', '0', 'x', hex[ch >> 4], hex[ch & 15], '>', 0 };
auto token = pimpl->token_to_id.find(buf); const auto token = text_to_token(buf);
if (token != pimpl->token_to_id.end()) { if (token != LLAMA_TOKEN_NULL) {
return (*token).second; return token;
} }
// Try to fall back to just the byte as a string // Try to fall back to just the byte as a string
const char buf2[2] = { (char)ch, 0 }; const char buf2[2] = { (char)ch, 0 };
return pimpl->token_to_id.at(buf2); const auto fallback = text_to_token(buf2);
if (fallback == LLAMA_TOKEN_NULL) throw std::out_of_range("byte token not found");
return fallback;
} }
case LLAMA_VOCAB_TYPE_WPM: case LLAMA_VOCAB_TYPE_WPM:
case LLAMA_VOCAB_TYPE_BPE: { case LLAMA_VOCAB_TYPE_BPE: {
return pimpl->token_to_id.at(unicode_byte_to_utf8(ch)); const auto token = text_to_token(unicode_byte_to_utf8(ch));
if (token == LLAMA_TOKEN_NULL) throw std::out_of_range("byte token not found");
return token;
} }
case LLAMA_VOCAB_TYPE_PLAMO2: { case LLAMA_VOCAB_TYPE_PLAMO2: {
// PLaMo-2 uses byte tokens in format <0xXX> // PLaMo-2 uses byte tokens in format <0xXX>
char hex_str[8]; char hex_str[8];
snprintf(hex_str, sizeof(hex_str), "<0x%02X>", ch); snprintf(hex_str, sizeof(hex_str), "<0x%02X>", ch);
return pimpl->token_to_id.at(hex_str); const auto token = text_to_token(hex_str);
if (token == LLAMA_TOKEN_NULL) throw std::out_of_range("byte token not found");
return token;
} }
default: default:
GGML_ABORT("fatal error"); GGML_ABORT("fatal error");
@@ -4156,6 +4318,10 @@ llama_token llama_vocab::byte_to_token(uint8_t ch) const {
llama_token llama_vocab::text_to_token(const std::string & text) const { llama_token llama_vocab::text_to_token(const std::string & text) const {
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
pimpl->wait_for_tokenizer_longhaul();
if (pimpl->tokenizer_longhaul_index) {
return pimpl->tokenizer_longhaul_index->text_to_token(text);
}
auto it = pimpl->token_to_id.find(text); auto it = pimpl->token_to_id.find(text);
if (it != pimpl->token_to_id.end()) { if (it != pimpl->token_to_id.end()) {
return (*it).second; return (*it).second;
@@ -4165,16 +4331,19 @@ llama_token llama_vocab::text_to_token(const std::string & text) const {
const llama_vocab::token_data & llama_vocab::get_token_data(llama_token id) const { const llama_vocab::token_data & llama_vocab::get_token_data(llama_token id) const {
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
pimpl->wait_for_tokenizer_longhaul();
return pimpl->id_to_token.at(id); return pimpl->id_to_token.at(id);
} }
const char * llama_vocab::token_get_text(llama_token id) const { const char * llama_vocab::token_get_text(llama_token id) const {
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
pimpl->wait_for_tokenizer_longhaul();
return pimpl->id_to_token.at(id).text.c_str(); return pimpl->id_to_token.at(id).text.c_str();
} }
float llama_vocab::token_get_score(llama_token id) const { float llama_vocab::token_get_score(llama_token id) const {
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE); GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
pimpl->wait_for_tokenizer_longhaul();
return pimpl->id_to_token.at(id).score; return pimpl->id_to_token.at(id).score;
} }
@@ -4306,6 +4475,10 @@ int llama_vocab::find_bpe_rank(const std::string & token_left, const std::string
GGML_ASSERT(token_left.find(' ') == std::string::npos); GGML_ASSERT(token_left.find(' ') == std::string::npos);
GGML_ASSERT(token_right.find(' ') == std::string::npos); GGML_ASSERT(token_right.find(' ') == std::string::npos);
pimpl->wait_for_tokenizer_longhaul();
if (pimpl->tokenizer_longhaul_index) {
return pimpl->tokenizer_longhaul_index->find_bpe_rank(token_left, token_right);
}
auto it = pimpl->bpe_ranks.find(std::make_pair(token_left, token_right)); auto it = pimpl->bpe_ranks.find(std::make_pair(token_left, token_right));
if (it == pimpl->bpe_ranks.end()) { if (it == pimpl->bpe_ranks.end()) {
return -1; return -1;
@@ -4315,6 +4488,10 @@ int llama_vocab::find_bpe_rank(const std::string & token_left, const std::string
} }
std::vector<std::string> llama_vocab::get_bpe_merges() const { std::vector<std::string> llama_vocab::get_bpe_merges() const {
pimpl->wait_for_tokenizer_longhaul();
if (pimpl->tokenizer_longhaul_index) {
return pimpl->tokenizer_longhaul_index->get_bpe_merges();
}
int max_rank = -1; int max_rank = -1;
for (const auto & pair : pimpl->bpe_ranks) { for (const auto & pair : pimpl->bpe_ranks) {
max_rank = std::max(max_rank, pair.second); max_rank = std::max(max_rank, pair.second);
@@ -4364,7 +4541,7 @@ std::vector<llama_token> llama_vocab::tokenize(
return pimpl->tokenize(raw_text, add_special, parse_special); return pimpl->tokenize(raw_text, add_special, parse_special);
} }
const std::string & llama_vocab::token_to_piece(llama_token token) const { std::string llama_vocab::token_to_piece(llama_token token) const {
return pimpl->token_to_piece(token); return pimpl->token_to_piece(token);
} }
+6 -2
View File
@@ -86,7 +86,11 @@ struct llama_vocab {
llama_vocab(); llama_vocab();
~llama_vocab(); ~llama_vocab();
void load(llama_model_loader & ml, const LLM_KV & kv); void load(
llama_model_loader & ml,
const LLM_KV & kv,
bool tokenizer_longhaul = false,
uint64_t tokenizer_longhaul_cache_bytes = 0);
std::string get_tokenizer_model() const; std::string get_tokenizer_model() const;
std::string get_tokenizer_pre() const; std::string get_tokenizer_pre() const;
@@ -181,7 +185,7 @@ struct llama_vocab {
bool special) const; bool special) const;
// use cached data // use cached data
const std::string & token_to_piece(llama_token token) const; std::string token_to_piece(llama_token token) const;
int32_t detokenize( int32_t detokenize(
const llama_token * tokens, const llama_token * tokens,
-5
View File
@@ -263,10 +263,6 @@ static bool llama_prepare_model_devices(const llama_model_params & params, llama
} }
} }
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL && !rpc_servers.empty()) {
// RPC longhaul v1 keeps the entire paging domain on one remote device.
model->devices.push_back(rpc_servers.front());
} else {
// add RPC servers at the front of the list to minimize network transfers // add RPC servers at the front of the list to minimize network transfers
model->devices.insert(model->devices.begin(), rpc_servers.begin(), rpc_servers.end()); model->devices.insert(model->devices.begin(), rpc_servers.begin(), rpc_servers.end());
@@ -279,7 +275,6 @@ static bool llama_prepare_model_devices(const llama_model_params & params, llama
model->devices.insert(model->devices.end(), igpus.begin(), igpus.end()); model->devices.insert(model->devices.end(), igpus.begin(), igpus.end());
} }
} }
}
// if using single GPU mode, remove all except the main GPU // if using single GPU mode, remove all except the main GPU
if (params.split_mode == LLAMA_SPLIT_MODE_NONE && !model->devices.empty()) { if (params.split_mode == LLAMA_SPLIT_MODE_NONE && !model->devices.empty()) {
+2
View File
@@ -259,6 +259,8 @@ set_tests_properties(test-thread-safety PROPERTIES FIXTURES_REQUIRED test-downlo
llama_build_and_test(test-arg-parser.cpp) llama_build_and_test(test-arg-parser.cpp)
llama_build_and_test(test-longhaul.cpp) llama_build_and_test(test-longhaul.cpp)
llama_build_and_test(test-tokenizer-longhaul.cpp
ARGS ${PROJECT_SOURCE_DIR}/models/ggml-vocab-qwen35.gguf)
llama_build_and_test(test-token-cache.cpp) llama_build_and_test(test-token-cache.cpp)
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src) target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
+15
View File
@@ -160,6 +160,21 @@ static void test(void) {
assert(params.load_mode == LLAMA_LOAD_MODE_LONGHAUL); assert(params.load_mode == LLAMA_LOAD_MODE_LONGHAUL);
assert(params.longhaul_cache_bytes == 2ULL * 1024 * 1024 * 1024); assert(params.longhaul_cache_bytes == 2ULL * 1024 * 1024 * 1024);
params.tokenizer_longhaul = false;
params.tokenizer_longhaul_cache_bytes = 0;
argv = {"binary_name", "--tokenizer-longhaul"};
assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
argv = {"binary_name", "--tokenizer-longhaul", "--tokenizer-longhaul-cache", "64"};
assert(true == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
assert(params.tokenizer_longhaul);
assert(params.tokenizer_longhaul_cache_bytes == 64ULL * 1024 * 1024);
params.tokenizer_longhaul = false;
params.tokenizer_longhaul_cache_bytes = 0;
argv = {"binary_name", "--tokenizer-longhaul-cache", "64"};
assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
argv = {"binary_name", "--token-cache-size", "256", "--token-cache-dir", "/tmp/llama-token-cache-test.sqlite3"}; argv = {"binary_name", "--token-cache-size", "256", "--token-cache-dir", "/tmp/llama-token-cache-test.sqlite3"};
assert(true == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON)); assert(true == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
assert(params.token_cache_size_mib == 256); assert(params.token_cache_size_mib == 256);
+1 -2
View File
@@ -58,8 +58,7 @@ struct longhaul_fixture {
sources.push_back({ weights_b, 0, n_experts * expert_size, expert_size, 0 }); sources.push_back({ weights_b, 0, n_experts * expert_size, expert_size, 0 });
} }
cache = std::make_unique<llama_longhaul_cache>( cache = std::make_unique<llama_longhaul_cache>(
std::move(files), std::move(sources), std::vector<llama_model_loader::longhaul_shard>{}, std::move(files), std::move(sources), n_slots, n_experts, 1);
n_slots, n_experts, 1, false);
} }
~longhaul_fixture() { ~longhaul_fixture() {
+144
View File
@@ -0,0 +1,144 @@
#include "llama.h"
#include "../src/llama-token-cache.h"
#include "../src/llama-tokenizer-longhaul.h"
#include <chrono>
#include <cstring>
#include <filesystem>
#include <string>
#include <thread>
#include <vector>
#undef NDEBUG
#include <cassert>
static std::vector<llama_token> tokenize(const llama_vocab * vocab, const std::string & text) {
int32_t count = llama_tokenize(vocab, text.data(), int32_t(text.size()), nullptr, 0, false, true);
assert(count < 0);
std::vector<llama_token> result(size_t(-count));
count = llama_tokenize(vocab, text.data(), int32_t(text.size()), result.data(), result.size(), false, true);
assert(count == int32_t(result.size()));
return result;
}
int main(int argc, char ** argv) {
namespace fs = std::filesystem;
const fs::path root = fs::temp_directory_path() /
("llama-tokenizer-longhaul-test-" + std::to_string(
std::chrono::high_resolution_clock::now().time_since_epoch().count()));
const std::string db = (root / "token-cache.sqlite3").string();
const llama_token_cache_params params = {
db.c_str(),
128 * 1024 * 1024,
};
assert(llama_token_cache_configure(params));
assert(llama_token_cache_clear());
std::vector<llama_tokenizer_longhaul_reverse> reverse;
std::vector<llama_tokenizer_longhaul_merge> merges;
for (int i = 0; i < 200; ++i) {
reverse.push_back({"token-" + std::to_string(i), i});
if (i > 0) {
merges.push_back({
"token-" + std::to_string(i - 1),
"token-" + std::to_string(i),
i - 1,
});
}
}
{
llama_tokenizer_longhaul_index index("source:test", 512);
std::string error;
assert(index.prepare(reverse, merges, "tokenizer-id", error));
assert(error.empty());
assert(index.n_merges() == merges.size());
for (int i = 0; i < 200; ++i) {
assert(index.text_to_token("token-" + std::to_string(i)) == i);
}
assert(index.text_to_token("missing") == LLAMA_TOKEN_NULL);
for (int i = 1; i < 200; ++i) {
assert(index.find_bpe_rank(
"token-" + std::to_string(i - 1),
"token-" + std::to_string(i)) == i - 1);
}
assert(index.find_bpe_rank("missing", "pair") == -1);
assert(index.cache_bytes_used() <= index.cache_capacity());
const auto loaded_merges = index.get_bpe_merges();
assert(loaded_merges.size() == merges.size());
assert(loaded_merges.front() == "token-0 token-1");
assert(loaded_merges.back() == "token-198 token-199");
}
// A warm open must use the completed artifact even without rebuild inputs.
{
llama_tokenizer_longhaul_index index("source:test", 128);
std::string error;
assert(index.prepare({}, {}, "tokenizer-id", error));
assert(index.text_to_token("token-42") == 42);
std::vector<std::thread> workers;
for (int thread = 0; thread < 4; ++thread) {
workers.emplace_back([&index, thread] {
for (int i = thread; i < 200; i += 4) {
assert(index.text_to_token("token-" + std::to_string(i)) == i);
}
});
}
for (auto & worker : workers) worker.join();
assert(index.cache_bytes_used() <= index.cache_capacity());
}
if (argc > 1) {
llama_backend_init();
llama_model_params normal_params = llama_model_default_params();
normal_params.vocab_only = true;
llama_model * normal = llama_model_load_from_file(argv[1], normal_params);
assert(normal != nullptr);
llama_model_kv_override overrides[2] = {};
std::strcpy(overrides[0].key, "general.architecture");
overrides[0].tag = LLAMA_KV_OVERRIDE_TYPE_STR;
std::strcpy(overrides[0].val_str, "qwen35moe");
llama_model_params longhaul_params = llama_model_default_params();
longhaul_params.vocab_only = true;
longhaul_params.kv_overrides = overrides;
longhaul_params.tokenizer_longhaul = true;
longhaul_params.tokenizer_longhaul_cache_bytes = 1024;
llama_model * longhaul = llama_model_load_from_file(argv[1], longhaul_params);
assert(longhaul != nullptr);
const llama_vocab * normal_vocab = llama_model_get_vocab(normal);
const llama_vocab * longhaul_vocab = llama_model_get_vocab(longhaul);
for (const std::string & text : {
"Hello, world!",
" tokenizer longhaul",
"Qwen3.5: 12345\n",
"<|im_end|>",
"Unicode: Καλημέρα 世界 🚚",
}) {
const auto expected = tokenize(normal_vocab, text);
const auto actual = tokenize(longhaul_vocab, text);
assert(actual == expected);
for (const auto token : actual) {
assert(std::strcmp(
llama_vocab_get_text(normal_vocab, token),
llama_vocab_get_text(longhaul_vocab, token)) == 0);
}
}
llama_model_free(longhaul);
llama_model_free(normal);
llama_backend_free();
}
assert(llama_token_cache_clear());
std::error_code ec;
fs::remove_all(root, ec);
return 0;
}
+2
View File
@@ -35,6 +35,8 @@
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) | | `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) | | `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) | | `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
| `--tokenizer-longhaul` | build and page the Qwen3.5/Laguna tokenizer through a bounded cache<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL) |
| `--tokenizer-longhaul-cache N` | tokenizer-longhaul RAM cache size in MiB<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE) |
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) | | `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
| `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys | | `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys |
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) | | `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
+2
View File
@@ -118,6 +118,8 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) | | `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) | | `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) | | `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
| `--tokenizer-longhaul` | build and page the Qwen3.5/Laguna tokenizer through a bounded cache<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL) |
| `--tokenizer-longhaul-cache N` | tokenizer-longhaul RAM cache size in MiB<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE) |
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) | | `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
| `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys | | `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys |
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) | | `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
+2
View File
@@ -57,6 +57,8 @@ test parameters:
-ctk, --cache-type-k <t> (default: f16) -ctk, --cache-type-k <t> (default: f16)
-ctv, --cache-type-v <t> (default: f16) -ctv, --cache-type-v <t> (default: f16)
--longhaul-cache <GiB> expert cache size for longhaul mode --longhaul-cache <GiB> expert cache size for longhaul mode
--tokenizer-longhaul page tokenizer indexes through a bounded cache
--tokenizer-longhaul-cache <MiB> tokenizer lookup cache size
-t, --threads <n> (default: system dependent) -t, --threads <n> (default: system dependent)
-C, --cpu-mask <hex,hex> (default: 0x0) -C, --cpu-mask <hex,hex> (default: 0x0)
--cpu-strict <0|1> (default: 0) --cpu-strict <0|1> (default: 0)
+53 -3
View File
@@ -342,6 +342,8 @@ struct cmd_params {
std::vector<llama_split_mode> split_mode; std::vector<llama_split_mode> split_mode;
std::vector<llama_load_mode> load_mode; std::vector<llama_load_mode> load_mode;
uint64_t longhaul_cache_bytes; uint64_t longhaul_cache_bytes;
bool tokenizer_longhaul;
uint64_t tokenizer_longhaul_cache_bytes;
std::vector<int> main_gpu; std::vector<int> main_gpu;
std::vector<bool> no_kv_offload; std::vector<bool> no_kv_offload;
std::vector<llama_flash_attn_type> flash_attn; std::vector<llama_flash_attn_type> flash_attn;
@@ -387,6 +389,8 @@ static const cmd_params cmd_params_defaults = {
/* split_mode */ { LLAMA_SPLIT_MODE_LAYER }, /* split_mode */ { LLAMA_SPLIT_MODE_LAYER },
/* load_mode */ { LLAMA_LOAD_MODE_MMAP }, /* load_mode */ { LLAMA_LOAD_MODE_MMAP },
/* longhaul_cache_bytes */ 0, /* longhaul_cache_bytes */ 0,
/* tokenizer_longhaul */ false,
/* tokenizer_longhaul_cache_bytes */ 0,
/* main_gpu */ { 0 }, /* main_gpu */ { 0 },
/* no_kv_offload */ { false }, /* no_kv_offload */ { false },
/* flash_attn */ { LLAMA_FLASH_ATTN_TYPE_AUTO }, /* flash_attn */ { LLAMA_FLASH_ATTN_TYPE_AUTO },
@@ -463,6 +467,8 @@ static void print_usage(int /* argc */, char ** argv) {
printf(" -dev, --device <dev0/dev1/...> (default: auto)\n"); printf(" -dev, --device <dev0/dev1/...> (default: auto)\n");
printf(" -lm, --load-mode <none|mmap|mlock|mmap+mlock|dio|longhaul> (default: %s)\n", join(transform_to_str(cmd_params_defaults.load_mode, llama_load_mode_name), ",").c_str()); printf(" -lm, --load-mode <none|mmap|mlock|mmap+mlock|dio|longhaul> (default: %s)\n", join(transform_to_str(cmd_params_defaults.load_mode, llama_load_mode_name), ",").c_str());
printf(" --longhaul-cache <GiB> expert cache size for longhaul mode\n"); printf(" --longhaul-cache <GiB> expert cache size for longhaul mode\n");
printf(" --tokenizer-longhaul page tokenizer indexes through a bounded cache\n");
printf(" --tokenizer-longhaul-cache <MiB> tokenizer lookup cache size (required with --tokenizer-longhaul)\n");
printf(" -mmp, --mmap <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n"); printf(" -mmp, --mmap <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n");
printf(" -dio, --direct-io <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n"); printf(" -dio, --direct-io <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n");
printf(" -embd, --embeddings <0|1> (default: %s)\n", join(cmd_params_defaults.embeddings, ",").c_str()); printf(" -embd, --embeddings <0|1> (default: %s)\n", join(cmd_params_defaults.embeddings, ",").c_str());
@@ -525,6 +531,8 @@ static cmd_params parse_cmd_params(int argc, char ** argv) {
params.no_warmup = cmd_params_defaults.no_warmup; params.no_warmup = cmd_params_defaults.no_warmup;
params.offline = cmd_params_defaults.offline; params.offline = cmd_params_defaults.offline;
params.longhaul_cache_bytes = cmd_params_defaults.longhaul_cache_bytes; params.longhaul_cache_bytes = cmd_params_defaults.longhaul_cache_bytes;
params.tokenizer_longhaul = cmd_params_defaults.tokenizer_longhaul;
params.tokenizer_longhaul_cache_bytes = cmd_params_defaults.tokenizer_longhaul_cache_bytes;
if (const char * env = getenv("HF_TOKEN")) { if (const char * env = getenv("HF_TOKEN")) {
params.hf_token = env; params.hf_token = env;
@@ -801,6 +809,19 @@ static cmd_params parse_cmd_params(int argc, char ** argv) {
break; break;
} }
params.longhaul_cache_bytes = gib * 1024 * 1024 * 1024; params.longhaul_cache_bytes = gib * 1024 * 1024 * 1024;
} else if (arg == "--tokenizer-longhaul") {
params.tokenizer_longhaul = true;
} else if (arg == "--tokenizer-longhaul-cache") {
if (++i >= argc) {
invalid_param = true;
break;
}
const uint64_t mib = std::stoull(argv[i]);
if (mib == 0 || mib > UINT64_MAX / 1024 / 1024) {
invalid_param = true;
break;
}
params.tokenizer_longhaul_cache_bytes = mib * 1024 * 1024;
} else if (arg == "-mg" || arg == "--main-gpu") { } else if (arg == "-mg" || arg == "--main-gpu") {
if (++i >= argc) { if (++i >= argc) {
invalid_param = true; invalid_param = true;
@@ -1157,6 +1178,10 @@ static cmd_params parse_cmd_params(int argc, char ** argv) {
fprintf(stderr, "error: longhaul load mode requires --longhaul-cache N\n"); fprintf(stderr, "error: longhaul load mode requires --longhaul-cache N\n");
exit(1); exit(1);
} }
if (params.tokenizer_longhaul != (params.tokenizer_longhaul_cache_bytes != 0)) {
fprintf(stderr, "error: --tokenizer-longhaul and --tokenizer-longhaul-cache must be used together\n");
exit(1);
}
if (params.main_gpu.empty()) { if (params.main_gpu.empty()) {
params.main_gpu = cmd_params_defaults.main_gpu; params.main_gpu = cmd_params_defaults.main_gpu;
} }
@@ -1224,6 +1249,8 @@ struct cmd_params_instance {
llama_split_mode split_mode; llama_split_mode split_mode;
llama_load_mode load_mode; llama_load_mode load_mode;
uint64_t longhaul_cache_bytes; uint64_t longhaul_cache_bytes;
bool tokenizer_longhaul;
uint64_t tokenizer_longhaul_cache_bytes;
int main_gpu; int main_gpu;
bool no_kv_offload; bool no_kv_offload;
llama_flash_attn_type flash_attn; llama_flash_attn_type flash_attn;
@@ -1246,6 +1273,8 @@ struct cmd_params_instance {
mparams.split_mode = split_mode; mparams.split_mode = split_mode;
mparams.load_mode = load_mode; mparams.load_mode = load_mode;
mparams.longhaul_cache_bytes = longhaul_cache_bytes; mparams.longhaul_cache_bytes = longhaul_cache_bytes;
mparams.tokenizer_longhaul = tokenizer_longhaul;
mparams.tokenizer_longhaul_cache_bytes = tokenizer_longhaul_cache_bytes;
mparams.main_gpu = main_gpu; mparams.main_gpu = main_gpu;
mparams.tensor_split = tensor_split.data(); mparams.tensor_split = tensor_split.data();
mparams.no_host = no_host; mparams.no_host = no_host;
@@ -1294,6 +1323,8 @@ struct cmd_params_instance {
split_mode == other.split_mode && split_mode == other.split_mode &&
main_gpu == other.main_gpu && tensor_split == other.tensor_split && main_gpu == other.main_gpu && tensor_split == other.tensor_split &&
load_mode == other.load_mode && longhaul_cache_bytes == other.longhaul_cache_bytes && load_mode == other.load_mode && longhaul_cache_bytes == other.longhaul_cache_bytes &&
tokenizer_longhaul == other.tokenizer_longhaul &&
tokenizer_longhaul_cache_bytes == other.tokenizer_longhaul_cache_bytes &&
devices == other.devices && no_host == other.no_host && devices == other.devices && no_host == other.no_host &&
vec_tensor_buft_override_equal(tensor_buft_overrides, other.tensor_buft_overrides); vec_tensor_buft_override_equal(tensor_buft_overrides, other.tensor_buft_overrides);
} }
@@ -1368,6 +1399,8 @@ static std::vector<cmd_params_instance> get_cmd_params_instances(const cmd_param
/* .split_mode = */ sm, /* .split_mode = */ sm,
/* .load_mode = */ lm, /* .load_mode = */ lm,
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes, /* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
/* .tokenizer_longhaul = */ params.tokenizer_longhaul,
/* .tokenizer_longhaul_cache_bytes = */ params.tokenizer_longhaul_cache_bytes,
/* .main_gpu = */ mg, /* .main_gpu = */ mg,
/* .no_kv_offload = */ nkvo, /* .no_kv_offload = */ nkvo,
/* .flash_attn = */ fa, /* .flash_attn = */ fa,
@@ -1405,6 +1438,8 @@ static std::vector<cmd_params_instance> get_cmd_params_instances(const cmd_param
/* .split_mode = */ sm, /* .split_mode = */ sm,
/* .load_mode = */ lm, /* .load_mode = */ lm,
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes, /* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
/* .tokenizer_longhaul = */ params.tokenizer_longhaul,
/* .tokenizer_longhaul_cache_bytes = */ params.tokenizer_longhaul_cache_bytes,
/* .main_gpu = */ mg, /* .main_gpu = */ mg,
/* .no_kv_offload = */ nkvo, /* .no_kv_offload = */ nkvo,
/* .flash_attn = */ fa, /* .flash_attn = */ fa,
@@ -1442,6 +1477,8 @@ static std::vector<cmd_params_instance> get_cmd_params_instances(const cmd_param
/* .split_mode = */ sm, /* .split_mode = */ sm,
/* .load_mode = */ lm, /* .load_mode = */ lm,
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes, /* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
/* .tokenizer_longhaul = */ params.tokenizer_longhaul,
/* .tokenizer_longhaul_cache_bytes = */ params.tokenizer_longhaul_cache_bytes,
/* .main_gpu = */ mg, /* .main_gpu = */ mg,
/* .no_kv_offload = */ nkvo, /* .no_kv_offload = */ nkvo,
/* .flash_attn = */ fa, /* .flash_attn = */ fa,
@@ -1484,6 +1521,8 @@ struct test {
llama_split_mode split_mode; llama_split_mode split_mode;
llama_load_mode load_mode; llama_load_mode load_mode;
uint64_t longhaul_cache_bytes; uint64_t longhaul_cache_bytes;
bool tokenizer_longhaul;
uint64_t tokenizer_longhaul_cache_bytes;
int main_gpu; int main_gpu;
bool no_kv_offload; bool no_kv_offload;
llama_flash_attn_type flash_attn; llama_flash_attn_type flash_attn;
@@ -1524,6 +1563,8 @@ struct test {
split_mode = inst.split_mode; split_mode = inst.split_mode;
load_mode = inst.load_mode; load_mode = inst.load_mode;
longhaul_cache_bytes = inst.longhaul_cache_bytes; longhaul_cache_bytes = inst.longhaul_cache_bytes;
tokenizer_longhaul = inst.tokenizer_longhaul;
tokenizer_longhaul_cache_bytes = inst.tokenizer_longhaul_cache_bytes;
main_gpu = inst.main_gpu; main_gpu = inst.main_gpu;
no_kv_offload = inst.no_kv_offload; no_kv_offload = inst.no_kv_offload;
flash_attn = inst.flash_attn; flash_attn = inst.flash_attn;
@@ -1591,7 +1632,8 @@ struct test {
"n_ubatch", "n_threads", "cpu_mask", "cpu_strict", "poll", "n_ubatch", "n_threads", "cpu_mask", "cpu_strict", "poll",
"type_k", "type_v", "n_gpu_layers", "n_cpu_moe", "split_mode", "type_k", "type_v", "n_gpu_layers", "n_cpu_moe", "split_mode",
"main_gpu", "no_kv_offload", "flash_attn", "devices", "tensor_split", "main_gpu", "no_kv_offload", "flash_attn", "devices", "tensor_split",
"tensor_buft_overrides", "load_mode", "longhaul_cache_bytes", "embeddings", "tensor_buft_overrides", "load_mode", "longhaul_cache_bytes",
"tokenizer_longhaul", "tokenizer_longhaul_cache_bytes", "embeddings",
"no_op_offload", "no_host", "fit_target", "fit_min_ctx", "no_op_offload", "no_host", "fit_target", "fit_min_ctx",
"n_prompt", "n_gen", "n_depth", "n_prompt", "n_gen", "n_depth",
"test_time", "avg_ns", "stddev_ns", "avg_ts", "stddev_ts" "test_time", "avg_ns", "stddev_ns", "avg_ts", "stddev_ts"
@@ -1607,11 +1649,11 @@ struct test {
field == "main_gpu" || field == "n_prompt" || field == "n_gen" || field == "n_depth" || field == "avg_ns" || field == "main_gpu" || field == "n_prompt" || field == "n_gen" || field == "n_depth" || field == "avg_ns" ||
field == "stddev_ns" || field == "no_op_offload" || field == "n_cpu_moe" || field == "stddev_ns" || field == "no_op_offload" || field == "n_cpu_moe" ||
field == "fit_target" || field == "fit_min_ctx" || field == "flash_attn" || field == "fit_target" || field == "fit_min_ctx" || field == "flash_attn" ||
field == "longhaul_cache_bytes") { field == "longhaul_cache_bytes" || field == "tokenizer_longhaul_cache_bytes") {
return INT; return INT;
} }
if (field == "f16_kv" || field == "no_kv_offload" || field == "cpu_strict" || if (field == "f16_kv" || field == "no_kv_offload" || field == "cpu_strict" ||
field == "embeddings" || field == "no_host") { field == "embeddings" || field == "no_host" || field == "tokenizer_longhaul") {
return BOOL; return BOOL;
} }
if (field == "avg_ts" || field == "stddev_ts") { if (field == "avg_ts" || field == "stddev_ts") {
@@ -1688,6 +1730,8 @@ struct test {
tensor_buft_overrides_str, tensor_buft_overrides_str,
llama_load_mode_name(load_mode), llama_load_mode_name(load_mode),
std::to_string(longhaul_cache_bytes), std::to_string(longhaul_cache_bytes),
std::to_string(tokenizer_longhaul),
std::to_string(tokenizer_longhaul_cache_bytes),
std::to_string(embeddings), std::to_string(embeddings),
std::to_string(no_op_offload), std::to_string(no_op_offload),
std::to_string(no_host), std::to_string(no_host),
@@ -2005,6 +2049,12 @@ struct markdown_printer : public printer {
if (params.longhaul_cache_bytes != cmd_params_defaults.longhaul_cache_bytes) { if (params.longhaul_cache_bytes != cmd_params_defaults.longhaul_cache_bytes) {
fields.emplace_back("longhaul_cache_bytes"); fields.emplace_back("longhaul_cache_bytes");
} }
if (params.tokenizer_longhaul != cmd_params_defaults.tokenizer_longhaul) {
fields.emplace_back("tokenizer_longhaul");
}
if (params.tokenizer_longhaul_cache_bytes != cmd_params_defaults.tokenizer_longhaul_cache_bytes) {
fields.emplace_back("tokenizer_longhaul_cache_bytes");
}
if (params.embeddings.size() > 1 || params.embeddings != cmd_params_defaults.embeddings) { if (params.embeddings.size() > 1 || params.embeddings != cmd_params_defaults.embeddings) {
fields.emplace_back("embeddings"); fields.emplace_back("embeddings");
} }
+1 -11
View File
@@ -36,17 +36,6 @@ flowchart TD
By default, `ggml-rpc-server` exposes all available accelerator devices on the host. By default, `ggml-rpc-server` exposes all available accelerator devices on the host.
If there are no accelerators, it exposes a single `CPU` device. If there are no accelerators, it exposes a single `CPU` device.
For server-resident longhaul MoE paging, place the same GGUF shards on the RPC
host and point the server at their directory:
```sh
$ bin/ggml-rpc-server --device MTL0 --longhaul-root /path/to/model-directory
```
The client can then use `--rpc`, `--longhaul`, and `--longhaul-cache` together.
RPC longhaul v1 requires one remote Metal device. Expert cache misses are read
from the RPC host's local files rather than uploaded by the client.
## Usage ## Usage
### Remote hosts ### Remote hosts
@@ -118,3 +107,4 @@ Use the `GGML_RPC_DEBUG` environment variable to enable debug messages from `ggm
```bash ```bash
$ GGML_RPC_DEBUG=1 bin/ggml-rpc-server $ GGML_RPC_DEBUG=1 bin/ggml-rpc-server
``` ```
+3 -35
View File
@@ -13,7 +13,6 @@
#include <algorithm> #include <algorithm>
#include <clocale> #include <clocale>
#include <codecvt> #include <codecvt>
#include <cstring>
#include <filesystem> #include <filesystem>
#include <regex> #include <regex>
#include <stdio.h> #include <stdio.h>
@@ -174,7 +173,6 @@ struct rpc_server_params {
std::string host = "127.0.0.1"; std::string host = "127.0.0.1";
int port = 50052; int port = 50052;
bool use_cache = false; bool use_cache = false;
std::string longhaul_root;
int n_threads = std::max(1U, std::thread::hardware_concurrency()/2); int n_threads = std::max(1U, std::thread::hardware_concurrency()/2);
std::vector<std::string> devices; std::vector<std::string> devices;
}; };
@@ -188,7 +186,6 @@ static void print_usage(int /*argc*/, char ** argv, rpc_server_params params) {
fprintf(stderr, " -H, --host HOST host to bind to (default: %s)\n", params.host.c_str()); fprintf(stderr, " -H, --host HOST host to bind to (default: %s)\n", params.host.c_str());
fprintf(stderr, " -p, --port PORT port to bind to (default: %d)\n", params.port); fprintf(stderr, " -p, --port PORT port to bind to (default: %d)\n", params.port);
fprintf(stderr, " -c, --cache enable local file cache\n"); fprintf(stderr, " -c, --cache enable local file cache\n");
fprintf(stderr, " --longhaul-root PATH serve longhaul expert data from this model directory\n");
fprintf(stderr, "\n"); fprintf(stderr, "\n");
} }
@@ -236,11 +233,6 @@ static bool rpc_server_params_parse(int argc, char ** argv, rpc_server_params &
} }
} else if (arg == "-c" || arg == "--cache") { } else if (arg == "-c" || arg == "--cache") {
params.use_cache = true; params.use_cache = true;
} else if (arg == "--longhaul-root") {
if (++i >= argc) {
return false;
}
params.longhaul_root = argv[i];
} else if (arg == "-h" || arg == "--help") { } else if (arg == "-h" || arg == "--help") {
print_usage(argc, argv, params); print_usage(argc, argv, params);
exit(0); exit(0);
@@ -321,23 +313,6 @@ int main(int argc, char * argv[]) {
fprintf(stderr, "No devices found\n"); fprintf(stderr, "No devices found\n");
return 1; return 1;
} }
if (!params.longhaul_root.empty()) {
std::error_code ec;
const auto root = std::filesystem::canonical(params.longhaul_root, ec);
if (ec || !std::filesystem::is_directory(root, ec)) {
fprintf(stderr, "Invalid longhaul root: %s\n", params.longhaul_root.c_str());
return 1;
}
params.longhaul_root = root.string();
for (auto * dev : devices) {
auto * dev_reg = ggml_backend_dev_backend_reg(dev);
if (!dev_reg || strcmp(ggml_backend_reg_name(dev_reg), "MTL") != 0) {
fprintf(stderr, "--longhaul-root currently requires Metal devices\n");
return 1;
}
}
}
std::string endpoint = params.host + ":" + std::to_string(params.port); std::string endpoint = params.host + ":" + std::to_string(params.port);
const char * cache_dir = nullptr; const char * cache_dir = nullptr;
std::string cache_dir_str; std::string cache_dir_str;
@@ -356,19 +331,12 @@ int main(int argc, char * argv[]) {
return 1; return 1;
} }
auto start_server_fn = (decltype(ggml_backend_rpc_start_server_with_options)*) auto start_server_fn = (decltype(ggml_backend_rpc_start_server)*) ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_start_server");
ggml_backend_reg_get_proc_address(reg, "ggml_backend_rpc_start_server_with_options");
if (!start_server_fn) { if (!start_server_fn) {
fprintf(stderr, "Failed to obtain RPC backend start server function with options\n"); fprintf(stderr, "Failed to obtain RPC backend start server function\n");
return 1; return 1;
} }
start_server_fn( start_server_fn(endpoint.c_str(), cache_dir, params.n_threads, devices.size(), devices.data());
endpoint.c_str(),
cache_dir,
params.longhaul_root.empty() ? nullptr : params.longhaul_root.c_str(),
params.n_threads,
devices.size(),
devices.data());
return 0; return 0;
} }
+2
View File
@@ -56,6 +56,8 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) | | `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) | | `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) | | `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
| `--tokenizer-longhaul` | build and page the Qwen3.5/Laguna tokenizer through a bounded cache<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL) |
| `--tokenizer-longhaul-cache N` | tokenizer-longhaul RAM cache size in MiB<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE) |
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) | | `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) | | `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
| `-e, --escape, --no-escape` | whether to process escapes sequences (\n, \r, \t, \', \", \\) (default: true) | | `-e, --escape, --no-escape` | whether to process escapes sequences (\n, \r, \t, \', \", \\) (default: true) |