Add tokenizer longhaul cache
This commit is contained in:
@@ -817,6 +817,15 @@ static bool common_params_parse_ex(int argc, char ** argv, common_params_context
|
|||||||
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL && params.longhaul_cache_bytes == 0) {
|
if (params.load_mode == LLAMA_LOAD_MODE_LONGHAUL && params.longhaul_cache_bytes == 0) {
|
||||||
throw std::invalid_argument("error: --longhaul requires --longhaul-cache N\n");
|
throw std::invalid_argument("error: --longhaul requires --longhaul-cache N\n");
|
||||||
}
|
}
|
||||||
|
if (params.tokenizer_longhaul != (params.tokenizer_longhaul_cache_bytes != 0)) {
|
||||||
|
throw std::invalid_argument(
|
||||||
|
params.tokenizer_longhaul
|
||||||
|
? "error: --tokenizer-longhaul requires --tokenizer-longhaul-cache N\n"
|
||||||
|
: "error: --tokenizer-longhaul-cache requires --tokenizer-longhaul\n");
|
||||||
|
}
|
||||||
|
if (params.tokenizer_longhaul && params.token_cache_size_mib == 0) {
|
||||||
|
throw std::invalid_argument("error: --tokenizer-longhaul requires an enabled persistent token cache\n");
|
||||||
|
}
|
||||||
|
|
||||||
postprocess_cpu_params(params.cpuparams, nullptr);
|
postprocess_cpu_params(params.cpuparams, nullptr);
|
||||||
postprocess_cpu_params(params.cpuparams_batch, ¶ms.cpuparams);
|
postprocess_cpu_params(params.cpuparams_batch, ¶ms.cpuparams);
|
||||||
@@ -2646,6 +2655,23 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
|||||||
params.longhaul_cache_bytes = uint64_t(value) * 1024 * 1024 * 1024;
|
params.longhaul_cache_bytes = uint64_t(value) * 1024 * 1024 * 1024;
|
||||||
}
|
}
|
||||||
).set_env("LLAMA_ARG_LONGHAUL_CACHE"));
|
).set_env("LLAMA_ARG_LONGHAUL_CACHE"));
|
||||||
|
add_opt(common_arg(
|
||||||
|
{"--tokenizer-longhaul"},
|
||||||
|
"build and page the Qwen3.5/Laguna tokenizer through a bounded cache",
|
||||||
|
[](common_params & params) {
|
||||||
|
params.tokenizer_longhaul = true;
|
||||||
|
}
|
||||||
|
).set_env("LLAMA_ARG_TOKENIZER_LONGHAUL"));
|
||||||
|
add_opt(common_arg(
|
||||||
|
{"--tokenizer-longhaul-cache"}, "N",
|
||||||
|
"tokenizer-longhaul RAM cache size in MiB",
|
||||||
|
[](common_params & params, int value) {
|
||||||
|
if (value <= 0) {
|
||||||
|
throw std::invalid_argument("tokenizer-longhaul cache size must be positive");
|
||||||
|
}
|
||||||
|
params.tokenizer_longhaul_cache_bytes = uint64_t(value) * 1024 * 1024;
|
||||||
|
}
|
||||||
|
).set_env("LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE"));
|
||||||
add_opt(common_arg(
|
add_opt(common_arg(
|
||||||
{"--numa"}, "TYPE",
|
{"--numa"}, "TYPE",
|
||||||
"attempt optimizations that help on some NUMA systems\n"
|
"attempt optimizations that help on some NUMA systems\n"
|
||||||
|
|||||||
@@ -1592,6 +1592,8 @@ struct llama_model_params common_model_params_to_llama(common_params & params) {
|
|||||||
mparams.split_mode = params.split_mode;
|
mparams.split_mode = params.split_mode;
|
||||||
mparams.load_mode = params.load_mode;
|
mparams.load_mode = params.load_mode;
|
||||||
mparams.longhaul_cache_bytes = params.longhaul_cache_bytes;
|
mparams.longhaul_cache_bytes = params.longhaul_cache_bytes;
|
||||||
|
mparams.tokenizer_longhaul = params.tokenizer_longhaul;
|
||||||
|
mparams.tokenizer_longhaul_cache_bytes = params.tokenizer_longhaul_cache_bytes;
|
||||||
mparams.tensor_split = params.tensor_split;
|
mparams.tensor_split = params.tensor_split;
|
||||||
mparams.check_tensors = params.check_tensors;
|
mparams.check_tensors = params.check_tensors;
|
||||||
mparams.use_extra_bufts = !params.no_extra_bufts;
|
mparams.use_extra_bufts = !params.no_extra_bufts;
|
||||||
|
|||||||
@@ -486,6 +486,8 @@ struct common_params {
|
|||||||
enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER; // how to split the model across GPUs
|
enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER; // how to split the model across GPUs
|
||||||
enum llama_load_mode load_mode = LLAMA_LOAD_MODE_MMAP; // how to load the model
|
enum llama_load_mode load_mode = LLAMA_LOAD_MODE_MMAP; // how to load the model
|
||||||
uint64_t longhaul_cache_bytes = 0;
|
uint64_t longhaul_cache_bytes = 0;
|
||||||
|
bool tokenizer_longhaul = false;
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes = 0;
|
||||||
|
|
||||||
common_cpu_params cpuparams;
|
common_cpu_params cpuparams;
|
||||||
common_cpu_params cpuparams_batch;
|
common_cpu_params cpuparams_batch;
|
||||||
|
|||||||
@@ -30,3 +30,24 @@ call `llama_token_cache_flush()`.
|
|||||||
|
|
||||||
The 5 GiB limit accounts for logical entry payloads. SQLite metadata and bounded
|
The 5 GiB limit accounts for logical entry payloads. SQLite metadata and bounded
|
||||||
journal files are not included.
|
journal files are not included.
|
||||||
|
|
||||||
|
## Tokenizer longhaul
|
||||||
|
|
||||||
|
Qwen3.5 MoE and Laguna models can move their reverse-token and BPE-merge indexes
|
||||||
|
into a persistent, disk-backed index while keeping a bounded lookup working set:
|
||||||
|
|
||||||
|
```text
|
||||||
|
--tokenizer-longhaul
|
||||||
|
--tokenizer-longhaul-cache N
|
||||||
|
```
|
||||||
|
|
||||||
|
`N` is the application-owned lookup cache budget in MiB and is required when
|
||||||
|
the mode is enabled. The mode is independent of model `--longhaul`, so the two
|
||||||
|
can be used together. The finite tokenizer index builds in the background while
|
||||||
|
model loading continues. If tokenizer work arrives first, it waits for atomic
|
||||||
|
index publication; build or storage failures are reported instead of silently
|
||||||
|
falling back to an unbounded tokenizer.
|
||||||
|
|
||||||
|
The persistent index uses the directory selected by `--token-cache-dir` and is
|
||||||
|
subject to `--token-cache-size`. Tokenized-text results continue to be cached
|
||||||
|
on demand and therefore do not have a finite completion point.
|
||||||
|
|||||||
@@ -360,6 +360,8 @@ extern "C" {
|
|||||||
bool no_alloc; // only load metadata and simulate memory allocations
|
bool no_alloc; // only load metadata and simulate memory allocations
|
||||||
|
|
||||||
uint64_t longhaul_cache_bytes; // routed expert cache size for LLAMA_LOAD_MODE_LONGHAUL
|
uint64_t longhaul_cache_bytes; // routed expert cache size for LLAMA_LOAD_MODE_LONGHAUL
|
||||||
|
bool tokenizer_longhaul; // page Qwen3.5/Laguna tokenizer indexes through a bounded cache
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes; // tokenizer-longhaul RAM cache budget
|
||||||
};
|
};
|
||||||
|
|
||||||
struct llama_sampler_seq_config {
|
struct llama_sampler_seq_config {
|
||||||
|
|||||||
@@ -38,6 +38,7 @@ add_library(llama
|
|||||||
llama-quant.cpp
|
llama-quant.cpp
|
||||||
llama-sampler.cpp
|
llama-sampler.cpp
|
||||||
llama-token-cache.cpp
|
llama-token-cache.cpp
|
||||||
|
llama-tokenizer-longhaul.cpp
|
||||||
llama-vocab.cpp
|
llama-vocab.cpp
|
||||||
unicode-data.cpp
|
unicode-data.cpp
|
||||||
unicode.cpp
|
unicode.cpp
|
||||||
|
|||||||
+3
-1
@@ -1249,7 +1249,7 @@ void llama_model_base::load_hparams(llama_model_loader & ml) {
|
|||||||
void llama_model_base::load_vocab(llama_model_loader & ml) {
|
void llama_model_base::load_vocab(llama_model_loader & ml) {
|
||||||
const auto kv = LLM_KV(arch);
|
const auto kv = LLM_KV(arch);
|
||||||
|
|
||||||
vocab.load(ml, kv);
|
vocab.load(ml, kv, params.tokenizer_longhaul, params.tokenizer_longhaul_cache_bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
bool llama_model_base::load_tensors(llama_model_loader & ml) {
|
||||||
@@ -2380,6 +2380,8 @@ llama_model_params llama_model_default_params() {
|
|||||||
/*.no_host =*/ false,
|
/*.no_host =*/ false,
|
||||||
/*.no_alloc =*/ false,
|
/*.no_alloc =*/ false,
|
||||||
/*.longhaul_cache_bytes =*/ 0,
|
/*.longhaul_cache_bytes =*/ 0,
|
||||||
|
/*.tokenizer_longhaul =*/ false,
|
||||||
|
/*.tokenizer_longhaul_cache_bytes =*/ 0,
|
||||||
};
|
};
|
||||||
|
|
||||||
return result;
|
return result;
|
||||||
|
|||||||
@@ -214,6 +214,10 @@ public:
|
|||||||
if (!ensure_open()) return params.capacity_bytes == 0;
|
if (!ensure_open()) return params.capacity_bytes == 0;
|
||||||
std::lock_guard<std::mutex> db_lock(db_mutex);
|
std::lock_guard<std::mutex> db_lock(db_mutex);
|
||||||
if (sqlite3_exec(db, "DELETE FROM entries; PRAGMA incremental_vacuum;", nullptr, nullptr, nullptr) != SQLITE_OK) return db_error();
|
if (sqlite3_exec(db, "DELETE FROM entries; PRAGMA incremental_vacuum;", nullptr, nullptr, nullptr) != SQLITE_OK) return db_error();
|
||||||
|
std::error_code ec;
|
||||||
|
fs::remove_all(fs::path(path).parent_path() / "tokenizer-indexes", ec);
|
||||||
|
if (ec) return false;
|
||||||
|
sqlite3_exec(db, "INSERT OR REPLACE INTO meta(key,value) VALUES('external_bytes',0)", nullptr, nullptr, nullptr);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -239,6 +243,11 @@ public:
|
|||||||
result.bytes_used = sqlite3_column_int64(stmt, 1);
|
result.bytes_used = sqlite3_column_int64(stmt, 1);
|
||||||
}
|
}
|
||||||
sqlite3_finalize(stmt);
|
sqlite3_finalize(stmt);
|
||||||
|
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='external_bytes'", -1, &stmt, nullptr) == SQLITE_OK &&
|
||||||
|
sqlite3_step(stmt) == SQLITE_ROW) {
|
||||||
|
result.bytes_used += static_cast<uint64_t>(sqlite3_column_int64(stmt, 0));
|
||||||
|
}
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
}
|
}
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
@@ -247,6 +256,45 @@ public:
|
|||||||
if (hit) ++stats.tokenizer_hits; else ++stats.tokenizer_misses;
|
if (hit) ++stats.tokenizer_hits; else ++stats.tokenizer_misses;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::string database_path() {
|
||||||
|
return ensure_open() ? path : std::string();
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t capacity() const {
|
||||||
|
return params.capacity_bytes;
|
||||||
|
}
|
||||||
|
|
||||||
|
void set_external_bytes(uint64_t bytes) {
|
||||||
|
if (!ensure_open()) return;
|
||||||
|
std::lock_guard<std::mutex> db_lock(db_mutex);
|
||||||
|
sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr);
|
||||||
|
sqlite3_stmt * stmt = nullptr;
|
||||||
|
sqlite3_prepare_v2(db, "INSERT OR REPLACE INTO meta(key,value) VALUES('external_bytes',?1)", -1, &stmt, nullptr);
|
||||||
|
sqlite3_bind_int64(stmt, 1, static_cast<sqlite3_int64>(bytes));
|
||||||
|
sqlite3_step(stmt);
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
|
||||||
|
uint64_t total = 0;
|
||||||
|
sqlite3_prepare_v2(db, "SELECT COALESCE(SUM(size),0) FROM entries", -1, &stmt, nullptr);
|
||||||
|
if (sqlite3_step(stmt) == SQLITE_ROW) total = sqlite3_column_int64(stmt, 0);
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
const uint64_t entry_capacity = bytes >= params.capacity_bytes ? 0 : params.capacity_bytes - bytes;
|
||||||
|
while (total > entry_capacity) {
|
||||||
|
uint64_t victim_size = 0;
|
||||||
|
sqlite3_prepare_v2(db, "SELECT size FROM entries ORDER BY last_access LIMIT 1", -1, &stmt, nullptr);
|
||||||
|
if (sqlite3_step(stmt) == SQLITE_ROW) victim_size = sqlite3_column_int64(stmt, 0);
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
if (victim_size == 0 || sqlite3_exec(db,
|
||||||
|
"DELETE FROM entries WHERE rowid=(SELECT rowid FROM entries ORDER BY last_access LIMIT 1)",
|
||||||
|
nullptr, nullptr, nullptr) != SQLITE_OK) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
total -= std::min(total, victim_size);
|
||||||
|
++stats.evictions;
|
||||||
|
}
|
||||||
|
sqlite3_exec(db, "COMMIT", nullptr, nullptr, nullptr);
|
||||||
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
static int64_t now_tick() {
|
static int64_t now_tick() {
|
||||||
return std::chrono::duration_cast<std::chrono::microseconds>(
|
return std::chrono::duration_cast<std::chrono::microseconds>(
|
||||||
@@ -311,13 +359,22 @@ private:
|
|||||||
sqlite3_bind_int64(stmt, 1, static_cast<sqlite3_int64>(params.capacity_bytes));
|
sqlite3_bind_int64(stmt, 1, static_cast<sqlite3_int64>(params.capacity_bytes));
|
||||||
sqlite3_step(stmt);
|
sqlite3_step(stmt);
|
||||||
sqlite3_finalize(stmt);
|
sqlite3_finalize(stmt);
|
||||||
|
uint64_t external_bytes = 0;
|
||||||
|
sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='external_bytes'", -1, &stmt, nullptr);
|
||||||
|
if (sqlite3_step(stmt) == SQLITE_ROW) {
|
||||||
|
external_bytes = static_cast<uint64_t>(sqlite3_column_int64(stmt, 0));
|
||||||
|
}
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
const uint64_t entry_capacity = external_bytes >= params.capacity_bytes
|
||||||
|
? 0
|
||||||
|
: params.capacity_bytes - external_bytes;
|
||||||
sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr);
|
sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr);
|
||||||
while (true) {
|
while (true) {
|
||||||
sqlite3_prepare_v2(db, "SELECT COALESCE(SUM(size),0) FROM entries", -1, &stmt, nullptr);
|
sqlite3_prepare_v2(db, "SELECT COALESCE(SUM(size),0) FROM entries", -1, &stmt, nullptr);
|
||||||
uint64_t total = 0;
|
uint64_t total = 0;
|
||||||
if (sqlite3_step(stmt) == SQLITE_ROW) total = sqlite3_column_int64(stmt, 0);
|
if (sqlite3_step(stmt) == SQLITE_ROW) total = sqlite3_column_int64(stmt, 0);
|
||||||
sqlite3_finalize(stmt);
|
sqlite3_finalize(stmt);
|
||||||
if (total <= params.capacity_bytes) break;
|
if (total <= entry_capacity) break;
|
||||||
if (sqlite3_exec(db,
|
if (sqlite3_exec(db,
|
||||||
"DELETE FROM entries WHERE rowid=(SELECT rowid FROM entries ORDER BY last_access LIMIT 1)",
|
"DELETE FROM entries WHERE rowid=(SELECT rowid FROM entries ORDER BY last_access LIMIT 1)",
|
||||||
nullptr, nullptr, nullptr) != SQLITE_OK) {
|
nullptr, nullptr, nullptr) != SQLITE_OK) {
|
||||||
@@ -355,12 +412,19 @@ private:
|
|||||||
std::lock_guard<std::mutex> db_lock(db_mutex);
|
std::lock_guard<std::mutex> db_lock(db_mutex);
|
||||||
if (sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr) != SQLITE_OK) { db_error(); return; }
|
if (sqlite3_exec(db, "BEGIN IMMEDIATE", nullptr, nullptr, nullptr) != SQLITE_OK) { db_error(); return; }
|
||||||
uint64_t capacity = params.capacity_bytes;
|
uint64_t capacity = params.capacity_bytes;
|
||||||
|
uint64_t external_bytes = 0;
|
||||||
sqlite3_stmt * capacity_stmt = nullptr;
|
sqlite3_stmt * capacity_stmt = nullptr;
|
||||||
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='capacity'", -1, &capacity_stmt, nullptr) == SQLITE_OK &&
|
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='capacity'", -1, &capacity_stmt, nullptr) == SQLITE_OK &&
|
||||||
sqlite3_step(capacity_stmt) == SQLITE_ROW) {
|
sqlite3_step(capacity_stmt) == SQLITE_ROW) {
|
||||||
capacity = static_cast<uint64_t>(sqlite3_column_int64(capacity_stmt, 0));
|
capacity = static_cast<uint64_t>(sqlite3_column_int64(capacity_stmt, 0));
|
||||||
}
|
}
|
||||||
sqlite3_finalize(capacity_stmt);
|
sqlite3_finalize(capacity_stmt);
|
||||||
|
if (sqlite3_prepare_v2(db, "SELECT value FROM meta WHERE key='external_bytes'", -1, &capacity_stmt, nullptr) == SQLITE_OK &&
|
||||||
|
sqlite3_step(capacity_stmt) == SQLITE_ROW) {
|
||||||
|
external_bytes = static_cast<uint64_t>(sqlite3_column_int64(capacity_stmt, 0));
|
||||||
|
}
|
||||||
|
sqlite3_finalize(capacity_stmt);
|
||||||
|
capacity = external_bytes >= capacity ? 0 : capacity - external_bytes;
|
||||||
if (item.payload.size() > capacity) {
|
if (item.payload.size() > capacity) {
|
||||||
sqlite3_exec(db, "ROLLBACK", nullptr, nullptr, nullptr);
|
sqlite3_exec(db, "ROLLBACK", nullptr, nullptr, nullptr);
|
||||||
return;
|
return;
|
||||||
@@ -492,6 +556,18 @@ void llama_token_cache_note_tokenizer_hit(bool hit) {
|
|||||||
cache().note_tokenizer(hit);
|
cache().note_tokenizer(hit);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
std::string llama_token_cache_database_path() {
|
||||||
|
return cache().database_path();
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t llama_token_cache_capacity() {
|
||||||
|
return cache().capacity();
|
||||||
|
}
|
||||||
|
|
||||||
|
void llama_token_cache_set_external_bytes(uint64_t bytes) {
|
||||||
|
cache().set_external_bytes(bytes);
|
||||||
|
}
|
||||||
|
|
||||||
llama_token_cache_params llama_token_cache_default_params() {
|
llama_token_cache_params llama_token_cache_default_params() {
|
||||||
return {nullptr, DEFAULT_CAPACITY};
|
return {nullptr, DEFAULT_CAPACITY};
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -27,3 +27,8 @@ void llama_token_cache_store(
|
|||||||
|
|
||||||
void llama_token_cache_note_tokenizer_hit(bool hit);
|
void llama_token_cache_note_tokenizer_hit(bool hit);
|
||||||
|
|
||||||
|
// Internal helpers used by tokenizer-longhaul artifacts.
|
||||||
|
// Opening the cache here also fixes the configured path for the process.
|
||||||
|
std::string llama_token_cache_database_path();
|
||||||
|
uint64_t llama_token_cache_capacity();
|
||||||
|
void llama_token_cache_set_external_bytes(uint64_t bytes);
|
||||||
|
|||||||
@@ -0,0 +1,471 @@
|
|||||||
|
#include "llama-tokenizer-longhaul.h"
|
||||||
|
|
||||||
|
#include "llama-token-cache.h"
|
||||||
|
|
||||||
|
#include <sqlite3.h>
|
||||||
|
|
||||||
|
#include <algorithm>
|
||||||
|
#include <chrono>
|
||||||
|
#include <filesystem>
|
||||||
|
#include <list>
|
||||||
|
#include <mutex>
|
||||||
|
#include <thread>
|
||||||
|
#include <unordered_map>
|
||||||
|
|
||||||
|
#if defined(_WIN32)
|
||||||
|
# include <process.h>
|
||||||
|
#else
|
||||||
|
# include <unistd.h>
|
||||||
|
#endif
|
||||||
|
|
||||||
|
namespace fs = std::filesystem;
|
||||||
|
|
||||||
|
namespace {
|
||||||
|
|
||||||
|
constexpr int INDEX_VERSION = 1;
|
||||||
|
|
||||||
|
int process_id() {
|
||||||
|
#if defined(_WIN32)
|
||||||
|
return _getpid();
|
||||||
|
#else
|
||||||
|
return getpid();
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
bool exec_sql(sqlite3 * db, const char * sql, std::string & error) {
|
||||||
|
char * message = nullptr;
|
||||||
|
if (sqlite3_exec(db, sql, nullptr, nullptr, &message) == SQLITE_OK) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
error = message ? message : sqlite3_errmsg(db);
|
||||||
|
sqlite3_free(message);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
struct lock_directory {
|
||||||
|
fs::path path;
|
||||||
|
bool owned = false;
|
||||||
|
|
||||||
|
~lock_directory() {
|
||||||
|
if (owned) {
|
||||||
|
std::error_code ec;
|
||||||
|
fs::remove(path, ec);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
uint64_t tokenizer_artifact_bytes(const fs::path & directory) {
|
||||||
|
uint64_t total = 0;
|
||||||
|
std::error_code ec;
|
||||||
|
for (const auto & entry : fs::directory_iterator(directory, ec)) {
|
||||||
|
if (ec) break;
|
||||||
|
if (!entry.is_regular_file(ec) || entry.path().extension() != ".sqlite3") continue;
|
||||||
|
total += entry.file_size(ec);
|
||||||
|
if (ec) {
|
||||||
|
ec.clear();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return total;
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
struct llama_tokenizer_longhaul_index::impl {
|
||||||
|
struct cache_entry {
|
||||||
|
int value;
|
||||||
|
uint64_t bytes;
|
||||||
|
std::list<std::string>::iterator position;
|
||||||
|
};
|
||||||
|
|
||||||
|
std::string source_key;
|
||||||
|
uint64_t capacity;
|
||||||
|
fs::path artifact;
|
||||||
|
sqlite3 * db = nullptr;
|
||||||
|
std::string id;
|
||||||
|
uint64_t merge_count = 0;
|
||||||
|
|
||||||
|
mutable std::mutex mutex;
|
||||||
|
mutable uint64_t used = 0;
|
||||||
|
mutable std::list<std::string> lru;
|
||||||
|
mutable std::unordered_map<std::string, cache_entry> values;
|
||||||
|
|
||||||
|
impl(std::string source_key, uint64_t capacity)
|
||||||
|
: source_key(std::move(source_key)), capacity(capacity) {
|
||||||
|
}
|
||||||
|
|
||||||
|
~impl() {
|
||||||
|
if (db) {
|
||||||
|
sqlite3_close(db);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
bool open_existing(std::string & error) {
|
||||||
|
if (!fs::exists(artifact)) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
sqlite3 * opened = nullptr;
|
||||||
|
if (sqlite3_open_v2(artifact.string().c_str(), &opened,
|
||||||
|
SQLITE_OPEN_READONLY | SQLITE_OPEN_FULLMUTEX, nullptr) != SQLITE_OK) {
|
||||||
|
error = opened ? sqlite3_errmsg(opened) : "failed to open tokenizer index";
|
||||||
|
if (opened) sqlite3_close(opened);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
sqlite3_stmt * stmt = nullptr;
|
||||||
|
const char * sql =
|
||||||
|
"SELECT version, tokenizer_id, n_merges FROM metadata WHERE complete=1 LIMIT 1";
|
||||||
|
bool valid = sqlite3_prepare_v2(opened, sql, -1, &stmt, nullptr) == SQLITE_OK &&
|
||||||
|
sqlite3_step(stmt) == SQLITE_ROW &&
|
||||||
|
sqlite3_column_int(stmt, 0) == INDEX_VERSION;
|
||||||
|
if (valid) {
|
||||||
|
const unsigned char * value = sqlite3_column_text(stmt, 1);
|
||||||
|
id = value ? reinterpret_cast<const char *>(value) : "";
|
||||||
|
merge_count = static_cast<uint64_t>(sqlite3_column_int64(stmt, 2));
|
||||||
|
valid = !id.empty();
|
||||||
|
}
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
if (!valid) {
|
||||||
|
error = "tokenizer index is incomplete or has an unsupported version";
|
||||||
|
sqlite3_close(opened);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
db = opened;
|
||||||
|
sqlite3_busy_timeout(db, 5000);
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool build(
|
||||||
|
const std::vector<llama_tokenizer_longhaul_reverse> & reverse,
|
||||||
|
const std::vector<llama_tokenizer_longhaul_merge> & merges,
|
||||||
|
const std::string & tokenizer_id,
|
||||||
|
std::string & error) {
|
||||||
|
const fs::path temp = artifact.string() + ".tmp." + std::to_string(process_id());
|
||||||
|
std::error_code ec;
|
||||||
|
fs::remove(temp, ec);
|
||||||
|
|
||||||
|
sqlite3 * output = nullptr;
|
||||||
|
if (sqlite3_open_v2(temp.string().c_str(), &output,
|
||||||
|
SQLITE_OPEN_READWRITE | SQLITE_OPEN_CREATE | SQLITE_OPEN_EXCLUSIVE, nullptr) != SQLITE_OK) {
|
||||||
|
error = output ? sqlite3_errmsg(output) : "failed to create tokenizer index";
|
||||||
|
if (output) sqlite3_close(output);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
auto fail = [&](const std::string & message) {
|
||||||
|
error = message;
|
||||||
|
sqlite3_close(output);
|
||||||
|
output = nullptr;
|
||||||
|
std::error_code remove_ec;
|
||||||
|
fs::remove(temp, remove_ec);
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
if (!exec_sql(output,
|
||||||
|
"PRAGMA journal_mode=OFF;"
|
||||||
|
"PRAGMA synchronous=OFF;"
|
||||||
|
"PRAGMA locking_mode=EXCLUSIVE;"
|
||||||
|
"CREATE TABLE metadata("
|
||||||
|
" version INTEGER NOT NULL,complete INTEGER NOT NULL,tokenizer_id TEXT NOT NULL,"
|
||||||
|
" n_reverse INTEGER NOT NULL,n_merges INTEGER NOT NULL);"
|
||||||
|
"CREATE TABLE reverse_tokens(text BLOB PRIMARY KEY,id INTEGER NOT NULL) WITHOUT ROWID;"
|
||||||
|
"CREATE TABLE merges(left_text BLOB NOT NULL,right_text BLOB NOT NULL,rank INTEGER NOT NULL,"
|
||||||
|
" PRIMARY KEY(left_text,right_text)) WITHOUT ROWID;"
|
||||||
|
"BEGIN IMMEDIATE;", error)) {
|
||||||
|
return fail(error);
|
||||||
|
}
|
||||||
|
|
||||||
|
sqlite3_stmt * insert_reverse = nullptr;
|
||||||
|
sqlite3_stmt * insert_merge = nullptr;
|
||||||
|
if (sqlite3_prepare_v2(output,
|
||||||
|
"INSERT INTO reverse_tokens(text,id) VALUES(?1,?2)", -1, &insert_reverse, nullptr) != SQLITE_OK ||
|
||||||
|
sqlite3_prepare_v2(output,
|
||||||
|
"INSERT INTO merges(left_text,right_text,rank) VALUES(?1,?2,?3)",
|
||||||
|
-1, &insert_merge, nullptr) != SQLITE_OK) {
|
||||||
|
if (insert_reverse) sqlite3_finalize(insert_reverse);
|
||||||
|
if (insert_merge) sqlite3_finalize(insert_merge);
|
||||||
|
return fail(sqlite3_errmsg(output));
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const auto & item : reverse) {
|
||||||
|
sqlite3_bind_blob(insert_reverse, 1, item.text.data(), int(item.text.size()), SQLITE_STATIC);
|
||||||
|
sqlite3_bind_int(insert_reverse, 2, item.id);
|
||||||
|
if (sqlite3_step(insert_reverse) != SQLITE_DONE) {
|
||||||
|
const std::string message = sqlite3_errmsg(output);
|
||||||
|
sqlite3_finalize(insert_reverse);
|
||||||
|
sqlite3_finalize(insert_merge);
|
||||||
|
return fail(message);
|
||||||
|
}
|
||||||
|
sqlite3_reset(insert_reverse);
|
||||||
|
sqlite3_clear_bindings(insert_reverse);
|
||||||
|
}
|
||||||
|
for (const auto & item : merges) {
|
||||||
|
sqlite3_bind_blob(insert_merge, 1, item.left.data(), int(item.left.size()), SQLITE_STATIC);
|
||||||
|
sqlite3_bind_blob(insert_merge, 2, item.right.data(), int(item.right.size()), SQLITE_STATIC);
|
||||||
|
sqlite3_bind_int(insert_merge, 3, item.rank);
|
||||||
|
if (sqlite3_step(insert_merge) != SQLITE_DONE) {
|
||||||
|
const std::string message = sqlite3_errmsg(output);
|
||||||
|
sqlite3_finalize(insert_reverse);
|
||||||
|
sqlite3_finalize(insert_merge);
|
||||||
|
return fail(message);
|
||||||
|
}
|
||||||
|
sqlite3_reset(insert_merge);
|
||||||
|
sqlite3_clear_bindings(insert_merge);
|
||||||
|
}
|
||||||
|
sqlite3_finalize(insert_reverse);
|
||||||
|
sqlite3_finalize(insert_merge);
|
||||||
|
|
||||||
|
sqlite3_stmt * metadata = nullptr;
|
||||||
|
if (sqlite3_prepare_v2(output,
|
||||||
|
"INSERT INTO metadata VALUES(?1,1,?2,?3,?4)", -1, &metadata, nullptr) != SQLITE_OK) {
|
||||||
|
return fail(sqlite3_errmsg(output));
|
||||||
|
}
|
||||||
|
sqlite3_bind_int(metadata, 1, INDEX_VERSION);
|
||||||
|
sqlite3_bind_text(metadata, 2, tokenizer_id.c_str(), -1, SQLITE_STATIC);
|
||||||
|
sqlite3_bind_int64(metadata, 3, static_cast<sqlite3_int64>(reverse.size()));
|
||||||
|
sqlite3_bind_int64(metadata, 4, static_cast<sqlite3_int64>(merges.size()));
|
||||||
|
const bool metadata_ok = sqlite3_step(metadata) == SQLITE_DONE;
|
||||||
|
sqlite3_finalize(metadata);
|
||||||
|
if (!metadata_ok || !exec_sql(output, "COMMIT;", error)) {
|
||||||
|
return fail(metadata_ok ? error : sqlite3_errmsg(output));
|
||||||
|
}
|
||||||
|
sqlite3_close(output);
|
||||||
|
output = nullptr;
|
||||||
|
|
||||||
|
const uint64_t persistent_capacity = llama_token_cache_capacity();
|
||||||
|
const uint64_t artifact_size = fs::file_size(temp, ec);
|
||||||
|
if (ec || artifact_size > persistent_capacity) {
|
||||||
|
fs::remove(temp, ec);
|
||||||
|
error = "completed tokenizer index does not fit in --token-cache-size";
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
struct artifact_candidate {
|
||||||
|
fs::path path;
|
||||||
|
fs::file_time_type modified;
|
||||||
|
uint64_t size;
|
||||||
|
};
|
||||||
|
std::vector<artifact_candidate> candidates;
|
||||||
|
uint64_t existing_bytes = 0;
|
||||||
|
for (const auto & entry : fs::directory_iterator(artifact.parent_path(), ec)) {
|
||||||
|
if (ec) break;
|
||||||
|
if (!entry.is_regular_file(ec) || entry.path().extension() != ".sqlite3" ||
|
||||||
|
entry.path() == artifact || entry.path() == temp) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
const uint64_t size = entry.file_size(ec);
|
||||||
|
if (ec) {
|
||||||
|
ec.clear();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
existing_bytes += size;
|
||||||
|
candidates.push_back({entry.path(), entry.last_write_time(ec), size});
|
||||||
|
if (ec) ec.clear();
|
||||||
|
}
|
||||||
|
std::sort(candidates.begin(), candidates.end(),
|
||||||
|
[](const artifact_candidate & a, const artifact_candidate & b) {
|
||||||
|
return a.modified < b.modified;
|
||||||
|
});
|
||||||
|
for (const auto & candidate : candidates) {
|
||||||
|
if (existing_bytes + artifact_size <= persistent_capacity) break;
|
||||||
|
fs::remove(candidate.path, ec);
|
||||||
|
if (!ec) existing_bytes -= std::min(existing_bytes, candidate.size);
|
||||||
|
ec.clear();
|
||||||
|
}
|
||||||
|
if (existing_bytes + artifact_size > persistent_capacity) {
|
||||||
|
fs::remove(temp, ec);
|
||||||
|
error = "tokenizer indexes do not fit in --token-cache-size";
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
fs::rename(temp, artifact, ec);
|
||||||
|
if (ec) {
|
||||||
|
const std::string rename_error = ec.message();
|
||||||
|
// A competing process may have published the same immutable artifact.
|
||||||
|
std::error_code remove_ec;
|
||||||
|
fs::remove(temp, remove_ec);
|
||||||
|
std::string ignored;
|
||||||
|
if (open_existing(ignored)) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
error = "failed to publish tokenizer index: " + rename_error;
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const bool opened = open_existing(error);
|
||||||
|
if (opened) {
|
||||||
|
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(artifact.parent_path()));
|
||||||
|
}
|
||||||
|
return opened;
|
||||||
|
}
|
||||||
|
|
||||||
|
int lookup(const std::string & cache_key, const char * sql,
|
||||||
|
const std::string & first, const std::string * second, int missing) const {
|
||||||
|
std::lock_guard<std::mutex> lock(mutex);
|
||||||
|
auto found = values.find(cache_key);
|
||||||
|
if (found != values.end()) {
|
||||||
|
lru.splice(lru.begin(), lru, found->second.position);
|
||||||
|
return found->second.value;
|
||||||
|
}
|
||||||
|
|
||||||
|
sqlite3_stmt * stmt = nullptr;
|
||||||
|
int value = missing;
|
||||||
|
if (sqlite3_prepare_v2(db, sql, -1, &stmt, nullptr) == SQLITE_OK) {
|
||||||
|
sqlite3_bind_blob(stmt, 1, first.data(), int(first.size()), SQLITE_STATIC);
|
||||||
|
if (second) {
|
||||||
|
sqlite3_bind_blob(stmt, 2, second->data(), int(second->size()), SQLITE_STATIC);
|
||||||
|
}
|
||||||
|
if (sqlite3_step(stmt) == SQLITE_ROW) {
|
||||||
|
value = sqlite3_column_int(stmt, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
|
||||||
|
const uint64_t bytes = cache_key.size() + sizeof(cache_entry) + sizeof(std::string);
|
||||||
|
if (bytes <= capacity) {
|
||||||
|
while (!lru.empty() && used + bytes > capacity) {
|
||||||
|
auto victim = values.find(lru.back());
|
||||||
|
if (victim != values.end()) {
|
||||||
|
used -= victim->second.bytes;
|
||||||
|
values.erase(victim);
|
||||||
|
}
|
||||||
|
lru.pop_back();
|
||||||
|
}
|
||||||
|
lru.push_front(cache_key);
|
||||||
|
values.emplace(cache_key, cache_entry{value, bytes, lru.begin()});
|
||||||
|
used += bytes;
|
||||||
|
}
|
||||||
|
return value;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
llama_tokenizer_longhaul_index::llama_tokenizer_longhaul_index(
|
||||||
|
std::string source_key, uint64_t cache_bytes)
|
||||||
|
: pimpl(std::make_unique<impl>(std::move(source_key), cache_bytes)) {
|
||||||
|
}
|
||||||
|
|
||||||
|
llama_tokenizer_longhaul_index::~llama_tokenizer_longhaul_index() = default;
|
||||||
|
|
||||||
|
bool llama_tokenizer_longhaul_index::prepare(
|
||||||
|
const std::vector<llama_tokenizer_longhaul_reverse> & reverse,
|
||||||
|
const std::vector<llama_tokenizer_longhaul_merge> & merges,
|
||||||
|
const std::string & tokenizer_id,
|
||||||
|
std::string & error) {
|
||||||
|
const std::string cache_db = llama_token_cache_database_path();
|
||||||
|
if (cache_db.empty()) {
|
||||||
|
error = "persistent token cache is disabled or unavailable";
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
const fs::path directory = fs::path(cache_db).parent_path() / "tokenizer-indexes";
|
||||||
|
std::error_code ec;
|
||||||
|
fs::create_directories(directory, ec);
|
||||||
|
if (ec) {
|
||||||
|
error = "failed to create tokenizer index directory: " + ec.message();
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
pimpl->artifact = directory / (llama_token_cache_hash(pimpl->source_key) + ".sqlite3");
|
||||||
|
if (pimpl->open_existing(error)) {
|
||||||
|
if (pimpl->id == tokenizer_id) {
|
||||||
|
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(directory));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
sqlite3_close(pimpl->db);
|
||||||
|
pimpl->db = nullptr;
|
||||||
|
std::error_code stale_ec;
|
||||||
|
fs::remove(pimpl->artifact, stale_ec);
|
||||||
|
}
|
||||||
|
|
||||||
|
lock_directory lock{pimpl->artifact.string() + ".lock", false};
|
||||||
|
lock.owned = fs::create_directory(lock.path, ec);
|
||||||
|
if (!lock.owned) {
|
||||||
|
// Another process is building. Wait for its atomic publication.
|
||||||
|
for (int i = 0; i < 24000; ++i) {
|
||||||
|
std::this_thread::sleep_for(std::chrono::milliseconds(25));
|
||||||
|
std::string ignored;
|
||||||
|
if (pimpl->open_existing(ignored)) {
|
||||||
|
if (pimpl->id == tokenizer_id) {
|
||||||
|
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(directory));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
sqlite3_close(pimpl->db);
|
||||||
|
pimpl->db = nullptr;
|
||||||
|
}
|
||||||
|
if (!fs::exists(lock.path)) {
|
||||||
|
lock.owned = fs::create_directory(lock.path, ec);
|
||||||
|
if (lock.owned) break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!lock.owned) {
|
||||||
|
error = "timed out waiting for another tokenizer index builder";
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
error.clear();
|
||||||
|
if (pimpl->open_existing(error)) {
|
||||||
|
if (pimpl->id == tokenizer_id) {
|
||||||
|
llama_token_cache_set_external_bytes(tokenizer_artifact_bytes(directory));
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
sqlite3_close(pimpl->db);
|
||||||
|
pimpl->db = nullptr;
|
||||||
|
}
|
||||||
|
std::error_code remove_ec;
|
||||||
|
fs::remove(pimpl->artifact, remove_ec);
|
||||||
|
return pimpl->build(reverse, merges, tokenizer_id, error);
|
||||||
|
}
|
||||||
|
|
||||||
|
llama_token llama_tokenizer_longhaul_index::text_to_token(const std::string & text) const {
|
||||||
|
const std::string key = "t" + text;
|
||||||
|
return pimpl->lookup(key,
|
||||||
|
"SELECT id FROM reverse_tokens WHERE text=?1", text, nullptr, LLAMA_TOKEN_NULL);
|
||||||
|
}
|
||||||
|
|
||||||
|
int llama_tokenizer_longhaul_index::find_bpe_rank(
|
||||||
|
const std::string & left, const std::string & right) const {
|
||||||
|
std::string key;
|
||||||
|
key.reserve(left.size() + right.size() + 2);
|
||||||
|
key.push_back('m');
|
||||||
|
key += left;
|
||||||
|
key.push_back('\0');
|
||||||
|
key += right;
|
||||||
|
return pimpl->lookup(key,
|
||||||
|
"SELECT rank FROM merges WHERE left_text=?1 AND right_text=?2", left, &right, -1);
|
||||||
|
}
|
||||||
|
|
||||||
|
std::vector<std::string> llama_tokenizer_longhaul_index::get_bpe_merges() const {
|
||||||
|
std::lock_guard<std::mutex> lock(pimpl->mutex);
|
||||||
|
std::vector<std::string> result;
|
||||||
|
result.reserve(pimpl->merge_count);
|
||||||
|
sqlite3_stmt * stmt = nullptr;
|
||||||
|
if (sqlite3_prepare_v2(pimpl->db,
|
||||||
|
"SELECT left_text,right_text FROM merges ORDER BY rank", -1, &stmt, nullptr) != SQLITE_OK) {
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
while (sqlite3_step(stmt) == SQLITE_ROW) {
|
||||||
|
const char * left = static_cast<const char *>(sqlite3_column_blob(stmt, 0));
|
||||||
|
const int left_size = sqlite3_column_bytes(stmt, 0);
|
||||||
|
const char * right = static_cast<const char *>(sqlite3_column_blob(stmt, 1));
|
||||||
|
const int right_size = sqlite3_column_bytes(stmt, 1);
|
||||||
|
std::string item(left, left_size);
|
||||||
|
item.push_back(' ');
|
||||||
|
item.append(right, right_size);
|
||||||
|
result.push_back(std::move(item));
|
||||||
|
}
|
||||||
|
sqlite3_finalize(stmt);
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t llama_tokenizer_longhaul_index::cache_bytes_used() const {
|
||||||
|
std::lock_guard<std::mutex> lock(pimpl->mutex);
|
||||||
|
return pimpl->used;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t llama_tokenizer_longhaul_index::cache_capacity() const {
|
||||||
|
return pimpl->capacity;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t llama_tokenizer_longhaul_index::n_merges() const {
|
||||||
|
return pimpl->merge_count;
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::string & llama_tokenizer_longhaul_index::tokenizer_id() const {
|
||||||
|
return pimpl->id;
|
||||||
|
}
|
||||||
@@ -0,0 +1,47 @@
|
|||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include "llama.h"
|
||||||
|
|
||||||
|
#include <cstdint>
|
||||||
|
#include <memory>
|
||||||
|
#include <string>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
struct llama_tokenizer_longhaul_reverse {
|
||||||
|
std::string text;
|
||||||
|
llama_token id;
|
||||||
|
};
|
||||||
|
|
||||||
|
struct llama_tokenizer_longhaul_merge {
|
||||||
|
std::string left;
|
||||||
|
std::string right;
|
||||||
|
int rank;
|
||||||
|
};
|
||||||
|
|
||||||
|
class llama_tokenizer_longhaul_index {
|
||||||
|
public:
|
||||||
|
llama_tokenizer_longhaul_index(std::string source_key, uint64_t cache_bytes);
|
||||||
|
~llama_tokenizer_longhaul_index();
|
||||||
|
|
||||||
|
llama_tokenizer_longhaul_index(const llama_tokenizer_longhaul_index &) = delete;
|
||||||
|
llama_tokenizer_longhaul_index & operator=(const llama_tokenizer_longhaul_index &) = delete;
|
||||||
|
|
||||||
|
bool prepare(
|
||||||
|
const std::vector<llama_tokenizer_longhaul_reverse> & reverse,
|
||||||
|
const std::vector<llama_tokenizer_longhaul_merge> & merges,
|
||||||
|
const std::string & tokenizer_id,
|
||||||
|
std::string & error);
|
||||||
|
|
||||||
|
llama_token text_to_token(const std::string & text) const;
|
||||||
|
int find_bpe_rank(const std::string & left, const std::string & right) const;
|
||||||
|
std::vector<std::string> get_bpe_merges() const;
|
||||||
|
|
||||||
|
uint64_t cache_bytes_used() const;
|
||||||
|
uint64_t cache_capacity() const;
|
||||||
|
uint64_t n_merges() const;
|
||||||
|
const std::string & tokenizer_id() const;
|
||||||
|
|
||||||
|
private:
|
||||||
|
struct impl;
|
||||||
|
std::unique_ptr<impl> pimpl;
|
||||||
|
};
|
||||||
+193
-16
@@ -1,5 +1,6 @@
|
|||||||
#include "llama-vocab.h"
|
#include "llama-vocab.h"
|
||||||
#include "llama-token-cache.h"
|
#include "llama-token-cache.h"
|
||||||
|
#include "llama-tokenizer-longhaul.h"
|
||||||
|
|
||||||
#include "ggml.h"
|
#include "ggml.h"
|
||||||
#include "gguf.h"
|
#include "gguf.h"
|
||||||
@@ -13,13 +14,16 @@
|
|||||||
#include <cctype>
|
#include <cctype>
|
||||||
#include <cfloat>
|
#include <cfloat>
|
||||||
#include <cmath>
|
#include <cmath>
|
||||||
|
#include <condition_variable>
|
||||||
#include <cstdarg>
|
#include <cstdarg>
|
||||||
#include <cstring>
|
#include <cstring>
|
||||||
#include <forward_list>
|
#include <forward_list>
|
||||||
#include <limits>
|
#include <limits>
|
||||||
#include <map>
|
#include <map>
|
||||||
|
#include <mutex>
|
||||||
#include <queue>
|
#include <queue>
|
||||||
#include <set>
|
#include <set>
|
||||||
|
#include <thread>
|
||||||
#include <unordered_map>
|
#include <unordered_map>
|
||||||
|
|
||||||
//
|
//
|
||||||
@@ -1846,12 +1850,37 @@ struct llama_vocab::impl {
|
|||||||
|
|
||||||
std::vector<char> precompiled_charsmap;
|
std::vector<char> precompiled_charsmap;
|
||||||
|
|
||||||
|
enum class longhaul_state {
|
||||||
|
disabled,
|
||||||
|
building,
|
||||||
|
ready,
|
||||||
|
failed,
|
||||||
|
};
|
||||||
|
bool tokenizer_longhaul = false;
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes = 0;
|
||||||
|
uint64_t tokenizer_longhaul_n_merges = 0;
|
||||||
|
mutable std::mutex tokenizer_longhaul_mutex;
|
||||||
|
mutable std::condition_variable tokenizer_longhaul_ready;
|
||||||
|
mutable longhaul_state tokenizer_longhaul_state = longhaul_state::disabled;
|
||||||
|
mutable std::string tokenizer_longhaul_error;
|
||||||
|
mutable std::unique_ptr<llama_tokenizer_longhaul_index> tokenizer_longhaul_index;
|
||||||
|
std::thread tokenizer_longhaul_worker;
|
||||||
|
|
||||||
impl(const llama_vocab & vocab) : vocab(vocab) {
|
impl(const llama_vocab & vocab) : vocab(vocab) {
|
||||||
}
|
}
|
||||||
|
|
||||||
~impl() = default;
|
~impl() {
|
||||||
|
if (tokenizer_longhaul_worker.joinable()) {
|
||||||
|
tokenizer_longhaul_worker.join();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
void load(llama_model_loader & ml, const LLM_KV & kv);
|
void load(
|
||||||
|
llama_model_loader & ml,
|
||||||
|
const LLM_KV & kv,
|
||||||
|
bool tokenizer_longhaul,
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes);
|
||||||
|
void wait_for_tokenizer_longhaul() const;
|
||||||
|
|
||||||
bool load_cache_snapshot(const std::vector<uint8_t> & snapshot);
|
bool load_cache_snapshot(const std::vector<uint8_t> & snapshot);
|
||||||
void reset_cache_snapshot_state();
|
void reset_cache_snapshot_state();
|
||||||
@@ -1908,7 +1937,7 @@ struct llama_vocab::impl {
|
|||||||
bool special) const;
|
bool special) const;
|
||||||
|
|
||||||
// use cached data
|
// use cached data
|
||||||
const std::string & token_to_piece(llama_token token) const;
|
std::string token_to_piece(llama_token token) const;
|
||||||
|
|
||||||
int32_t detokenize(
|
int32_t detokenize(
|
||||||
const llama_token * tokens,
|
const llama_token * tokens,
|
||||||
@@ -1928,11 +1957,29 @@ private:
|
|||||||
const llama_vocab & vocab;
|
const llama_vocab & vocab;
|
||||||
};
|
};
|
||||||
|
|
||||||
void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
void llama_vocab::impl::load(
|
||||||
|
llama_model_loader & ml,
|
||||||
|
const LLM_KV & kv,
|
||||||
|
bool tokenizer_longhaul_value,
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes_value) {
|
||||||
struct gguf_context * ctx = ml.metadata;
|
struct gguf_context * ctx = ml.metadata;
|
||||||
|
|
||||||
|
tokenizer_longhaul = tokenizer_longhaul_value;
|
||||||
|
tokenizer_longhaul_cache_bytes = tokenizer_longhaul_cache_bytes_value;
|
||||||
|
if (tokenizer_longhaul) {
|
||||||
|
if (tokenizer_longhaul_cache_bytes == 0) {
|
||||||
|
throw std::runtime_error("tokenizer-longhaul cache budget must be positive");
|
||||||
|
}
|
||||||
|
if (ml.get_arch() != LLM_ARCH_QWEN35MOE && ml.get_arch() != LLM_ARCH_LAGUNA) {
|
||||||
|
throw std::runtime_error("tokenizer-longhaul currently requires a qwen35moe or laguna model");
|
||||||
|
}
|
||||||
|
if (ml.get_token_cache_source_key().empty()) {
|
||||||
|
throw std::runtime_error("tokenizer-longhaul requires a file-backed GGUF model");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
const std::string & source_cache_key = ml.get_token_cache_source_key();
|
const std::string & source_cache_key = ml.get_token_cache_source_key();
|
||||||
if (!source_cache_key.empty()) {
|
if (!tokenizer_longhaul && !source_cache_key.empty()) {
|
||||||
std::vector<uint8_t> snapshot;
|
std::vector<uint8_t> snapshot;
|
||||||
bool found = llama_token_cache_lookup(LLAMA_TOKEN_CACHE_KIND_TOKENIZER, source_cache_key, snapshot);
|
bool found = llama_token_cache_lookup(LLAMA_TOKEN_CACHE_KIND_TOKENIZER, source_cache_key, snapshot);
|
||||||
if (found && snapshot.size() == 65 && snapshot[0] == 'A') {
|
if (found && snapshot.size() == 65 && snapshot[0] == 'A') {
|
||||||
@@ -1958,6 +2005,9 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
|||||||
ml.get_key(LLM_KV_TOKENIZER_TOKEN_TYPE_COUNT, n_token_types, false);
|
ml.get_key(LLM_KV_TOKENIZER_TOKEN_TYPE_COUNT, n_token_types, false);
|
||||||
|
|
||||||
if (tokenizer_model == "no_vocab" || tokenizer_model == "none") {
|
if (tokenizer_model == "no_vocab" || tokenizer_model == "none") {
|
||||||
|
if (tokenizer_longhaul) {
|
||||||
|
throw std::runtime_error("tokenizer-longhaul requires a BPE tokenizer");
|
||||||
|
}
|
||||||
type = LLAMA_VOCAB_TYPE_NONE;
|
type = LLAMA_VOCAB_TYPE_NONE;
|
||||||
|
|
||||||
// default special tokens
|
// default special tokens
|
||||||
@@ -3063,6 +3113,69 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (tokenizer_longhaul) {
|
||||||
|
if (type != LLAMA_VOCAB_TYPE_BPE ||
|
||||||
|
(tokenizer_pre != "qwen35" && tokenizer_pre != "laguna")) {
|
||||||
|
throw std::runtime_error(
|
||||||
|
"tokenizer-longhaul currently supports only qwen35 and laguna BPE tokenizers");
|
||||||
|
}
|
||||||
|
|
||||||
|
std::vector<llama_tokenizer_longhaul_reverse> reverse;
|
||||||
|
reverse.reserve(token_to_id.size());
|
||||||
|
while (!token_to_id.empty()) {
|
||||||
|
auto node = token_to_id.extract(token_to_id.begin());
|
||||||
|
reverse.push_back({std::move(node.key()), node.mapped()});
|
||||||
|
}
|
||||||
|
token_to_id.rehash(0);
|
||||||
|
|
||||||
|
std::vector<llama_tokenizer_longhaul_merge> merges;
|
||||||
|
merges.reserve(bpe_ranks.size());
|
||||||
|
while (!bpe_ranks.empty()) {
|
||||||
|
auto node = bpe_ranks.extract(bpe_ranks.begin());
|
||||||
|
auto key = std::move(node.key());
|
||||||
|
merges.push_back({std::move(key.first), std::move(key.second), node.mapped()});
|
||||||
|
}
|
||||||
|
bpe_ranks.rehash(0);
|
||||||
|
tokenizer_longhaul_n_merges = merges.size();
|
||||||
|
cache_token_to_piece.clear();
|
||||||
|
cache_token_to_piece.shrink_to_fit();
|
||||||
|
|
||||||
|
token_cache_id = llama_token_cache_hash(
|
||||||
|
std::string("tokenizer-longhaul-v1\n") + source_cache_key);
|
||||||
|
{
|
||||||
|
std::lock_guard<std::mutex> lock(tokenizer_longhaul_mutex);
|
||||||
|
tokenizer_longhaul_state = longhaul_state::building;
|
||||||
|
}
|
||||||
|
const uint64_t cache_bytes = tokenizer_longhaul_cache_bytes;
|
||||||
|
const std::string source_key = source_cache_key;
|
||||||
|
const std::string tokenizer_id = token_cache_id;
|
||||||
|
tokenizer_longhaul_worker = std::thread(
|
||||||
|
[this, source_key, cache_bytes, tokenizer_id,
|
||||||
|
reverse = std::move(reverse), merges = std::move(merges)]() mutable {
|
||||||
|
auto index = std::make_unique<llama_tokenizer_longhaul_index>(source_key, cache_bytes);
|
||||||
|
std::string error;
|
||||||
|
const bool ok = index->prepare(reverse, merges, tokenizer_id, error);
|
||||||
|
{
|
||||||
|
std::lock_guard<std::mutex> lock(tokenizer_longhaul_mutex);
|
||||||
|
if (ok) {
|
||||||
|
tokenizer_longhaul_index = std::move(index);
|
||||||
|
tokenizer_longhaul_state = longhaul_state::ready;
|
||||||
|
LLAMA_LOG_INFO("%s: tokenizer-longhaul index is ready\n", "llama_vocab");
|
||||||
|
} else {
|
||||||
|
tokenizer_longhaul_error = error.empty()
|
||||||
|
? "unknown tokenizer-longhaul build error"
|
||||||
|
: std::move(error);
|
||||||
|
tokenizer_longhaul_state = longhaul_state::failed;
|
||||||
|
LLAMA_LOG_ERROR("%s: tokenizer-longhaul index failed: %s\n",
|
||||||
|
"llama_vocab", tokenizer_longhaul_error.c_str());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
tokenizer_longhaul_ready.notify_all();
|
||||||
|
});
|
||||||
|
LLAMA_LOG_INFO("%s: tokenizer-longhaul index is building in the background\n", __func__);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
// Persist a versioned canonical snapshot of the effective tokenizer state.
|
// Persist a versioned canonical snapshot of the effective tokenizer state.
|
||||||
// The snapshot also serves as the namespace for text-to-token cache entries.
|
// The snapshot also serves as the namespace for text-to-token cache entries.
|
||||||
{
|
{
|
||||||
@@ -3151,6 +3264,23 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void llama_vocab::impl::wait_for_tokenizer_longhaul() const {
|
||||||
|
if (!tokenizer_longhaul) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
std::unique_lock<std::mutex> lock(tokenizer_longhaul_mutex);
|
||||||
|
if (tokenizer_longhaul_state == longhaul_state::disabled) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
tokenizer_longhaul_ready.wait(lock, [this] {
|
||||||
|
return tokenizer_longhaul_state == longhaul_state::ready ||
|
||||||
|
tokenizer_longhaul_state == longhaul_state::failed;
|
||||||
|
});
|
||||||
|
if (tokenizer_longhaul_state == longhaul_state::failed) {
|
||||||
|
throw std::runtime_error("tokenizer-longhaul failed: " + tokenizer_longhaul_error);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
void llama_vocab::impl::reset_cache_snapshot_state() {
|
void llama_vocab::impl::reset_cache_snapshot_state() {
|
||||||
n_token_types = 0;
|
n_token_types = 0;
|
||||||
tokenizer_model.clear();
|
tokenizer_model.clear();
|
||||||
@@ -3282,36 +3412,43 @@ std::string llama_vocab::impl::type_name() const{
|
|||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_normal(llama_token id) const {
|
bool llama_vocab::impl::is_normal(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_NORMAL;
|
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_NORMAL;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_unknown(llama_token id) const {
|
bool llama_vocab::impl::is_unknown(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNKNOWN;
|
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNKNOWN;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_control(llama_token id) const {
|
bool llama_vocab::impl::is_control(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_CONTROL;
|
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_CONTROL;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_byte(llama_token id) const {
|
bool llama_vocab::impl::is_byte(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_BYTE;
|
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_BYTE;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_user_defined(llama_token id) const {
|
bool llama_vocab::impl::is_user_defined(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_USER_DEFINED;
|
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_USER_DEFINED;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_unused(llama_token id) const {
|
bool llama_vocab::impl::is_unused(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNUSED;
|
return id_to_token[id].attr & LLAMA_TOKEN_ATTR_UNUSED;
|
||||||
}
|
}
|
||||||
|
|
||||||
bool llama_vocab::impl::is_eog(llama_token id) const {
|
bool llama_vocab::impl::is_eog(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
return id != LLAMA_TOKEN_NULL && special_eog_ids.count(id) > 0;
|
return id != LLAMA_TOKEN_NULL && special_eog_ids.count(id) > 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3339,6 +3476,7 @@ uint8_t llama_vocab::impl::token_to_byte(llama_token id) const {
|
|||||||
}
|
}
|
||||||
|
|
||||||
llama_token_attr llama_vocab::impl::token_get_attr(llama_token id) const {
|
llama_token_attr llama_vocab::impl::token_get_attr(llama_token id) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
return id_to_token.at(id).attr;
|
return id_to_token.at(id).attr;
|
||||||
}
|
}
|
||||||
@@ -3749,6 +3887,7 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
|
|||||||
const std::string & raw_text,
|
const std::string & raw_text,
|
||||||
bool add_special,
|
bool add_special,
|
||||||
bool parse_special) const {
|
bool parse_special) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
if (raw_text.empty() || token_cache_id.empty()) {
|
if (raw_text.empty() || token_cache_id.empty()) {
|
||||||
return tokenize_uncached(raw_text, add_special, parse_special);
|
return tokenize_uncached(raw_text, add_special, parse_special);
|
||||||
}
|
}
|
||||||
@@ -3792,6 +3931,7 @@ std::vector<llama_token> llama_vocab::impl::tokenize(
|
|||||||
}
|
}
|
||||||
|
|
||||||
int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t length, int32_t lstrip, bool special) const {
|
int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t length, int32_t lstrip, bool special) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
// ref: https://github.com/ggml-org/llama.cpp/pull/7587#discussion_r1620983843
|
// ref: https://github.com/ggml-org/llama.cpp/pull/7587#discussion_r1620983843
|
||||||
static const int attr_special = LLAMA_TOKEN_ATTR_UNKNOWN | LLAMA_TOKEN_ATTR_CONTROL;
|
static const int attr_special = LLAMA_TOKEN_ATTR_UNKNOWN | LLAMA_TOKEN_ATTR_CONTROL;
|
||||||
const llama_token_attr attr = token_get_attr(token);
|
const llama_token_attr attr = token_get_attr(token);
|
||||||
@@ -3908,7 +4048,11 @@ int32_t llama_vocab::impl::token_to_piece(llama_token token, char * buf, int32_t
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
const std::string & llama_vocab::impl::token_to_piece(llama_token token) const {
|
std::string llama_vocab::impl::token_to_piece(llama_token token) const {
|
||||||
|
wait_for_tokenizer_longhaul();
|
||||||
|
if (tokenizer_longhaul) {
|
||||||
|
return token_to_piece_for_cache(token, true);
|
||||||
|
}
|
||||||
return cache_token_to_piece.at(token);
|
return cache_token_to_piece.at(token);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -4027,7 +4171,15 @@ int32_t llama_vocab::impl::detokenize(
|
|||||||
void llama_vocab::impl::print_info() const {
|
void llama_vocab::impl::print_info() const {
|
||||||
LLAMA_LOG_INFO("%s: vocab type = %s\n", __func__, type_name().c_str());
|
LLAMA_LOG_INFO("%s: vocab type = %s\n", __func__, type_name().c_str());
|
||||||
LLAMA_LOG_INFO("%s: n_vocab = %u\n", __func__, vocab.n_tokens());
|
LLAMA_LOG_INFO("%s: n_vocab = %u\n", __func__, vocab.n_tokens());
|
||||||
LLAMA_LOG_INFO("%s: n_merges = %u\n", __func__, (uint32_t) bpe_ranks.size());
|
LLAMA_LOG_INFO("%s: n_merges = %u\n", __func__,
|
||||||
|
(uint32_t) (tokenizer_longhaul ? tokenizer_longhaul_n_merges : bpe_ranks.size()));
|
||||||
|
if (tokenizer_longhaul) {
|
||||||
|
std::lock_guard<std::mutex> lock(tokenizer_longhaul_mutex);
|
||||||
|
LLAMA_LOG_INFO("%s: tokenizer longhaul = %s, cache = %.2f MiB\n", __func__,
|
||||||
|
tokenizer_longhaul_state == longhaul_state::ready ? "ready" :
|
||||||
|
tokenizer_longhaul_state == longhaul_state::failed ? "failed" : "building",
|
||||||
|
tokenizer_longhaul_cache_bytes / 1024.0 / 1024.0);
|
||||||
|
}
|
||||||
|
|
||||||
// special tokens
|
// special tokens
|
||||||
if (special_bos_id != LLAMA_TOKEN_NULL) { LLAMA_LOG_INFO( "%s: BOS token = %d '%s'\n", __func__, special_bos_id, id_to_token.at(special_bos_id).text.c_str() ); }
|
if (special_bos_id != LLAMA_TOKEN_NULL) { LLAMA_LOG_INFO( "%s: BOS token = %d '%s'\n", __func__, special_bos_id, id_to_token.at(special_bos_id).text.c_str() ); }
|
||||||
@@ -4060,8 +4212,12 @@ llama_vocab::llama_vocab() : pimpl(new impl(*this)) {
|
|||||||
|
|
||||||
llama_vocab::~llama_vocab() = default;
|
llama_vocab::~llama_vocab() = default;
|
||||||
|
|
||||||
void llama_vocab::load(llama_model_loader & ml, const LLM_KV & kv) {
|
void llama_vocab::load(
|
||||||
pimpl->load(ml, kv);
|
llama_model_loader & ml,
|
||||||
|
const LLM_KV & kv,
|
||||||
|
bool tokenizer_longhaul,
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes) {
|
||||||
|
pimpl->load(ml, kv, tokenizer_longhaul, tokenizer_longhaul_cache_bytes);
|
||||||
}
|
}
|
||||||
|
|
||||||
std::string llama_vocab::get_tokenizer_model() const {
|
std::string llama_vocab::get_tokenizer_model() const {
|
||||||
@@ -4131,23 +4287,29 @@ llama_token llama_vocab::byte_to_token(uint8_t ch) const {
|
|||||||
case LLAMA_VOCAB_TYPE_SPM:
|
case LLAMA_VOCAB_TYPE_SPM:
|
||||||
case LLAMA_VOCAB_TYPE_UGM: {
|
case LLAMA_VOCAB_TYPE_UGM: {
|
||||||
const char buf[7] = { '<', '0', 'x', hex[ch >> 4], hex[ch & 15], '>', 0 };
|
const char buf[7] = { '<', '0', 'x', hex[ch >> 4], hex[ch & 15], '>', 0 };
|
||||||
auto token = pimpl->token_to_id.find(buf);
|
const auto token = text_to_token(buf);
|
||||||
if (token != pimpl->token_to_id.end()) {
|
if (token != LLAMA_TOKEN_NULL) {
|
||||||
return (*token).second;
|
return token;
|
||||||
}
|
}
|
||||||
// Try to fall back to just the byte as a string
|
// Try to fall back to just the byte as a string
|
||||||
const char buf2[2] = { (char)ch, 0 };
|
const char buf2[2] = { (char)ch, 0 };
|
||||||
return pimpl->token_to_id.at(buf2);
|
const auto fallback = text_to_token(buf2);
|
||||||
|
if (fallback == LLAMA_TOKEN_NULL) throw std::out_of_range("byte token not found");
|
||||||
|
return fallback;
|
||||||
}
|
}
|
||||||
case LLAMA_VOCAB_TYPE_WPM:
|
case LLAMA_VOCAB_TYPE_WPM:
|
||||||
case LLAMA_VOCAB_TYPE_BPE: {
|
case LLAMA_VOCAB_TYPE_BPE: {
|
||||||
return pimpl->token_to_id.at(unicode_byte_to_utf8(ch));
|
const auto token = text_to_token(unicode_byte_to_utf8(ch));
|
||||||
|
if (token == LLAMA_TOKEN_NULL) throw std::out_of_range("byte token not found");
|
||||||
|
return token;
|
||||||
}
|
}
|
||||||
case LLAMA_VOCAB_TYPE_PLAMO2: {
|
case LLAMA_VOCAB_TYPE_PLAMO2: {
|
||||||
// PLaMo-2 uses byte tokens in format <0xXX>
|
// PLaMo-2 uses byte tokens in format <0xXX>
|
||||||
char hex_str[8];
|
char hex_str[8];
|
||||||
snprintf(hex_str, sizeof(hex_str), "<0x%02X>", ch);
|
snprintf(hex_str, sizeof(hex_str), "<0x%02X>", ch);
|
||||||
return pimpl->token_to_id.at(hex_str);
|
const auto token = text_to_token(hex_str);
|
||||||
|
if (token == LLAMA_TOKEN_NULL) throw std::out_of_range("byte token not found");
|
||||||
|
return token;
|
||||||
}
|
}
|
||||||
default:
|
default:
|
||||||
GGML_ABORT("fatal error");
|
GGML_ABORT("fatal error");
|
||||||
@@ -4156,6 +4318,10 @@ llama_token llama_vocab::byte_to_token(uint8_t ch) const {
|
|||||||
|
|
||||||
llama_token llama_vocab::text_to_token(const std::string & text) const {
|
llama_token llama_vocab::text_to_token(const std::string & text) const {
|
||||||
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
|
pimpl->wait_for_tokenizer_longhaul();
|
||||||
|
if (pimpl->tokenizer_longhaul_index) {
|
||||||
|
return pimpl->tokenizer_longhaul_index->text_to_token(text);
|
||||||
|
}
|
||||||
auto it = pimpl->token_to_id.find(text);
|
auto it = pimpl->token_to_id.find(text);
|
||||||
if (it != pimpl->token_to_id.end()) {
|
if (it != pimpl->token_to_id.end()) {
|
||||||
return (*it).second;
|
return (*it).second;
|
||||||
@@ -4165,16 +4331,19 @@ llama_token llama_vocab::text_to_token(const std::string & text) const {
|
|||||||
|
|
||||||
const llama_vocab::token_data & llama_vocab::get_token_data(llama_token id) const {
|
const llama_vocab::token_data & llama_vocab::get_token_data(llama_token id) const {
|
||||||
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
|
pimpl->wait_for_tokenizer_longhaul();
|
||||||
return pimpl->id_to_token.at(id);
|
return pimpl->id_to_token.at(id);
|
||||||
}
|
}
|
||||||
|
|
||||||
const char * llama_vocab::token_get_text(llama_token id) const {
|
const char * llama_vocab::token_get_text(llama_token id) const {
|
||||||
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
|
pimpl->wait_for_tokenizer_longhaul();
|
||||||
return pimpl->id_to_token.at(id).text.c_str();
|
return pimpl->id_to_token.at(id).text.c_str();
|
||||||
}
|
}
|
||||||
|
|
||||||
float llama_vocab::token_get_score(llama_token id) const {
|
float llama_vocab::token_get_score(llama_token id) const {
|
||||||
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
GGML_ASSERT(pimpl->type != LLAMA_VOCAB_TYPE_NONE);
|
||||||
|
pimpl->wait_for_tokenizer_longhaul();
|
||||||
return pimpl->id_to_token.at(id).score;
|
return pimpl->id_to_token.at(id).score;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -4306,6 +4475,10 @@ int llama_vocab::find_bpe_rank(const std::string & token_left, const std::string
|
|||||||
GGML_ASSERT(token_left.find(' ') == std::string::npos);
|
GGML_ASSERT(token_left.find(' ') == std::string::npos);
|
||||||
GGML_ASSERT(token_right.find(' ') == std::string::npos);
|
GGML_ASSERT(token_right.find(' ') == std::string::npos);
|
||||||
|
|
||||||
|
pimpl->wait_for_tokenizer_longhaul();
|
||||||
|
if (pimpl->tokenizer_longhaul_index) {
|
||||||
|
return pimpl->tokenizer_longhaul_index->find_bpe_rank(token_left, token_right);
|
||||||
|
}
|
||||||
auto it = pimpl->bpe_ranks.find(std::make_pair(token_left, token_right));
|
auto it = pimpl->bpe_ranks.find(std::make_pair(token_left, token_right));
|
||||||
if (it == pimpl->bpe_ranks.end()) {
|
if (it == pimpl->bpe_ranks.end()) {
|
||||||
return -1;
|
return -1;
|
||||||
@@ -4315,6 +4488,10 @@ int llama_vocab::find_bpe_rank(const std::string & token_left, const std::string
|
|||||||
}
|
}
|
||||||
|
|
||||||
std::vector<std::string> llama_vocab::get_bpe_merges() const {
|
std::vector<std::string> llama_vocab::get_bpe_merges() const {
|
||||||
|
pimpl->wait_for_tokenizer_longhaul();
|
||||||
|
if (pimpl->tokenizer_longhaul_index) {
|
||||||
|
return pimpl->tokenizer_longhaul_index->get_bpe_merges();
|
||||||
|
}
|
||||||
int max_rank = -1;
|
int max_rank = -1;
|
||||||
for (const auto & pair : pimpl->bpe_ranks) {
|
for (const auto & pair : pimpl->bpe_ranks) {
|
||||||
max_rank = std::max(max_rank, pair.second);
|
max_rank = std::max(max_rank, pair.second);
|
||||||
@@ -4364,7 +4541,7 @@ std::vector<llama_token> llama_vocab::tokenize(
|
|||||||
return pimpl->tokenize(raw_text, add_special, parse_special);
|
return pimpl->tokenize(raw_text, add_special, parse_special);
|
||||||
}
|
}
|
||||||
|
|
||||||
const std::string & llama_vocab::token_to_piece(llama_token token) const {
|
std::string llama_vocab::token_to_piece(llama_token token) const {
|
||||||
return pimpl->token_to_piece(token);
|
return pimpl->token_to_piece(token);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+6
-2
@@ -86,7 +86,11 @@ struct llama_vocab {
|
|||||||
llama_vocab();
|
llama_vocab();
|
||||||
~llama_vocab();
|
~llama_vocab();
|
||||||
|
|
||||||
void load(llama_model_loader & ml, const LLM_KV & kv);
|
void load(
|
||||||
|
llama_model_loader & ml,
|
||||||
|
const LLM_KV & kv,
|
||||||
|
bool tokenizer_longhaul = false,
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes = 0);
|
||||||
|
|
||||||
std::string get_tokenizer_model() const;
|
std::string get_tokenizer_model() const;
|
||||||
std::string get_tokenizer_pre() const;
|
std::string get_tokenizer_pre() const;
|
||||||
@@ -181,7 +185,7 @@ struct llama_vocab {
|
|||||||
bool special) const;
|
bool special) const;
|
||||||
|
|
||||||
// use cached data
|
// use cached data
|
||||||
const std::string & token_to_piece(llama_token token) const;
|
std::string token_to_piece(llama_token token) const;
|
||||||
|
|
||||||
int32_t detokenize(
|
int32_t detokenize(
|
||||||
const llama_token * tokens,
|
const llama_token * tokens,
|
||||||
|
|||||||
@@ -259,6 +259,8 @@ set_tests_properties(test-thread-safety PROPERTIES FIXTURES_REQUIRED test-downlo
|
|||||||
|
|
||||||
llama_build_and_test(test-arg-parser.cpp)
|
llama_build_and_test(test-arg-parser.cpp)
|
||||||
llama_build_and_test(test-longhaul.cpp)
|
llama_build_and_test(test-longhaul.cpp)
|
||||||
|
llama_build_and_test(test-tokenizer-longhaul.cpp
|
||||||
|
ARGS ${PROJECT_SOURCE_DIR}/models/ggml-vocab-qwen35.gguf)
|
||||||
llama_build_and_test(test-token-cache.cpp)
|
llama_build_and_test(test-token-cache.cpp)
|
||||||
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
target_include_directories(test-token-cache PRIVATE ${PROJECT_SOURCE_DIR}/src)
|
||||||
|
|
||||||
|
|||||||
@@ -160,6 +160,21 @@ static void test(void) {
|
|||||||
assert(params.load_mode == LLAMA_LOAD_MODE_LONGHAUL);
|
assert(params.load_mode == LLAMA_LOAD_MODE_LONGHAUL);
|
||||||
assert(params.longhaul_cache_bytes == 2ULL * 1024 * 1024 * 1024);
|
assert(params.longhaul_cache_bytes == 2ULL * 1024 * 1024 * 1024);
|
||||||
|
|
||||||
|
params.tokenizer_longhaul = false;
|
||||||
|
params.tokenizer_longhaul_cache_bytes = 0;
|
||||||
|
argv = {"binary_name", "--tokenizer-longhaul"};
|
||||||
|
assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
|
||||||
|
|
||||||
|
argv = {"binary_name", "--tokenizer-longhaul", "--tokenizer-longhaul-cache", "64"};
|
||||||
|
assert(true == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
|
||||||
|
assert(params.tokenizer_longhaul);
|
||||||
|
assert(params.tokenizer_longhaul_cache_bytes == 64ULL * 1024 * 1024);
|
||||||
|
|
||||||
|
params.tokenizer_longhaul = false;
|
||||||
|
params.tokenizer_longhaul_cache_bytes = 0;
|
||||||
|
argv = {"binary_name", "--tokenizer-longhaul-cache", "64"};
|
||||||
|
assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
|
||||||
|
|
||||||
argv = {"binary_name", "--token-cache-size", "256", "--token-cache-dir", "/tmp/llama-token-cache-test.sqlite3"};
|
argv = {"binary_name", "--token-cache-size", "256", "--token-cache-dir", "/tmp/llama-token-cache-test.sqlite3"};
|
||||||
assert(true == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
|
assert(true == common_params_parse(argv.size(), list_str_to_char(argv).data(), params, LLAMA_EXAMPLE_COMMON));
|
||||||
assert(params.token_cache_size_mib == 256);
|
assert(params.token_cache_size_mib == 256);
|
||||||
|
|||||||
@@ -0,0 +1,144 @@
|
|||||||
|
#include "llama.h"
|
||||||
|
#include "../src/llama-token-cache.h"
|
||||||
|
#include "../src/llama-tokenizer-longhaul.h"
|
||||||
|
|
||||||
|
#include <chrono>
|
||||||
|
#include <cstring>
|
||||||
|
#include <filesystem>
|
||||||
|
#include <string>
|
||||||
|
#include <thread>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
#undef NDEBUG
|
||||||
|
#include <cassert>
|
||||||
|
|
||||||
|
static std::vector<llama_token> tokenize(const llama_vocab * vocab, const std::string & text) {
|
||||||
|
int32_t count = llama_tokenize(vocab, text.data(), int32_t(text.size()), nullptr, 0, false, true);
|
||||||
|
assert(count < 0);
|
||||||
|
std::vector<llama_token> result(size_t(-count));
|
||||||
|
count = llama_tokenize(vocab, text.data(), int32_t(text.size()), result.data(), result.size(), false, true);
|
||||||
|
assert(count == int32_t(result.size()));
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
|
int main(int argc, char ** argv) {
|
||||||
|
namespace fs = std::filesystem;
|
||||||
|
|
||||||
|
const fs::path root = fs::temp_directory_path() /
|
||||||
|
("llama-tokenizer-longhaul-test-" + std::to_string(
|
||||||
|
std::chrono::high_resolution_clock::now().time_since_epoch().count()));
|
||||||
|
const std::string db = (root / "token-cache.sqlite3").string();
|
||||||
|
const llama_token_cache_params params = {
|
||||||
|
db.c_str(),
|
||||||
|
128 * 1024 * 1024,
|
||||||
|
};
|
||||||
|
assert(llama_token_cache_configure(params));
|
||||||
|
assert(llama_token_cache_clear());
|
||||||
|
|
||||||
|
std::vector<llama_tokenizer_longhaul_reverse> reverse;
|
||||||
|
std::vector<llama_tokenizer_longhaul_merge> merges;
|
||||||
|
for (int i = 0; i < 200; ++i) {
|
||||||
|
reverse.push_back({"token-" + std::to_string(i), i});
|
||||||
|
if (i > 0) {
|
||||||
|
merges.push_back({
|
||||||
|
"token-" + std::to_string(i - 1),
|
||||||
|
"token-" + std::to_string(i),
|
||||||
|
i - 1,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
{
|
||||||
|
llama_tokenizer_longhaul_index index("source:test", 512);
|
||||||
|
std::string error;
|
||||||
|
assert(index.prepare(reverse, merges, "tokenizer-id", error));
|
||||||
|
assert(error.empty());
|
||||||
|
assert(index.n_merges() == merges.size());
|
||||||
|
|
||||||
|
for (int i = 0; i < 200; ++i) {
|
||||||
|
assert(index.text_to_token("token-" + std::to_string(i)) == i);
|
||||||
|
}
|
||||||
|
assert(index.text_to_token("missing") == LLAMA_TOKEN_NULL);
|
||||||
|
for (int i = 1; i < 200; ++i) {
|
||||||
|
assert(index.find_bpe_rank(
|
||||||
|
"token-" + std::to_string(i - 1),
|
||||||
|
"token-" + std::to_string(i)) == i - 1);
|
||||||
|
}
|
||||||
|
assert(index.find_bpe_rank("missing", "pair") == -1);
|
||||||
|
assert(index.cache_bytes_used() <= index.cache_capacity());
|
||||||
|
|
||||||
|
const auto loaded_merges = index.get_bpe_merges();
|
||||||
|
assert(loaded_merges.size() == merges.size());
|
||||||
|
assert(loaded_merges.front() == "token-0 token-1");
|
||||||
|
assert(loaded_merges.back() == "token-198 token-199");
|
||||||
|
}
|
||||||
|
|
||||||
|
// A warm open must use the completed artifact even without rebuild inputs.
|
||||||
|
{
|
||||||
|
llama_tokenizer_longhaul_index index("source:test", 128);
|
||||||
|
std::string error;
|
||||||
|
assert(index.prepare({}, {}, "tokenizer-id", error));
|
||||||
|
assert(index.text_to_token("token-42") == 42);
|
||||||
|
|
||||||
|
std::vector<std::thread> workers;
|
||||||
|
for (int thread = 0; thread < 4; ++thread) {
|
||||||
|
workers.emplace_back([&index, thread] {
|
||||||
|
for (int i = thread; i < 200; i += 4) {
|
||||||
|
assert(index.text_to_token("token-" + std::to_string(i)) == i);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
for (auto & worker : workers) worker.join();
|
||||||
|
assert(index.cache_bytes_used() <= index.cache_capacity());
|
||||||
|
}
|
||||||
|
|
||||||
|
if (argc > 1) {
|
||||||
|
llama_backend_init();
|
||||||
|
|
||||||
|
llama_model_params normal_params = llama_model_default_params();
|
||||||
|
normal_params.vocab_only = true;
|
||||||
|
llama_model * normal = llama_model_load_from_file(argv[1], normal_params);
|
||||||
|
assert(normal != nullptr);
|
||||||
|
|
||||||
|
llama_model_kv_override overrides[2] = {};
|
||||||
|
std::strcpy(overrides[0].key, "general.architecture");
|
||||||
|
overrides[0].tag = LLAMA_KV_OVERRIDE_TYPE_STR;
|
||||||
|
std::strcpy(overrides[0].val_str, "qwen35moe");
|
||||||
|
|
||||||
|
llama_model_params longhaul_params = llama_model_default_params();
|
||||||
|
longhaul_params.vocab_only = true;
|
||||||
|
longhaul_params.kv_overrides = overrides;
|
||||||
|
longhaul_params.tokenizer_longhaul = true;
|
||||||
|
longhaul_params.tokenizer_longhaul_cache_bytes = 1024;
|
||||||
|
llama_model * longhaul = llama_model_load_from_file(argv[1], longhaul_params);
|
||||||
|
assert(longhaul != nullptr);
|
||||||
|
|
||||||
|
const llama_vocab * normal_vocab = llama_model_get_vocab(normal);
|
||||||
|
const llama_vocab * longhaul_vocab = llama_model_get_vocab(longhaul);
|
||||||
|
for (const std::string & text : {
|
||||||
|
"Hello, world!",
|
||||||
|
" tokenizer longhaul",
|
||||||
|
"Qwen3.5: 12345\n",
|
||||||
|
"<|im_end|>",
|
||||||
|
"Unicode: Καλημέρα 世界 🚚",
|
||||||
|
}) {
|
||||||
|
const auto expected = tokenize(normal_vocab, text);
|
||||||
|
const auto actual = tokenize(longhaul_vocab, text);
|
||||||
|
assert(actual == expected);
|
||||||
|
for (const auto token : actual) {
|
||||||
|
assert(std::strcmp(
|
||||||
|
llama_vocab_get_text(normal_vocab, token),
|
||||||
|
llama_vocab_get_text(longhaul_vocab, token)) == 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
llama_model_free(longhaul);
|
||||||
|
llama_model_free(normal);
|
||||||
|
llama_backend_free();
|
||||||
|
}
|
||||||
|
|
||||||
|
assert(llama_token_cache_clear());
|
||||||
|
std::error_code ec;
|
||||||
|
fs::remove_all(root, ec);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
@@ -35,6 +35,8 @@
|
|||||||
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
|
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
|
||||||
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
|
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
|
||||||
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
|
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
|
||||||
|
| `--tokenizer-longhaul` | build and page the Qwen3.5/Laguna tokenizer through a bounded cache<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL) |
|
||||||
|
| `--tokenizer-longhaul-cache N` | tokenizer-longhaul RAM cache size in MiB<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE) |
|
||||||
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
|
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
|
||||||
| `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys |
|
| `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys |
|
||||||
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
|
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
|
||||||
|
|||||||
@@ -118,6 +118,8 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1
|
|||||||
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
|
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
|
||||||
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
|
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
|
||||||
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
|
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
|
||||||
|
| `--tokenizer-longhaul` | build and page the Qwen3.5/Laguna tokenizer through a bounded cache<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL) |
|
||||||
|
| `--tokenizer-longhaul-cache N` | tokenizer-longhaul RAM cache size in MiB<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE) |
|
||||||
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
|
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
|
||||||
| `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys |
|
| `-p, --prompt PROMPT` | prompt to start generation with; for system message, use -sys |
|
||||||
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
|
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
|
||||||
|
|||||||
@@ -57,6 +57,8 @@ test parameters:
|
|||||||
-ctk, --cache-type-k <t> (default: f16)
|
-ctk, --cache-type-k <t> (default: f16)
|
||||||
-ctv, --cache-type-v <t> (default: f16)
|
-ctv, --cache-type-v <t> (default: f16)
|
||||||
--longhaul-cache <GiB> expert cache size for longhaul mode
|
--longhaul-cache <GiB> expert cache size for longhaul mode
|
||||||
|
--tokenizer-longhaul page tokenizer indexes through a bounded cache
|
||||||
|
--tokenizer-longhaul-cache <MiB> tokenizer lookup cache size
|
||||||
-t, --threads <n> (default: system dependent)
|
-t, --threads <n> (default: system dependent)
|
||||||
-C, --cpu-mask <hex,hex> (default: 0x0)
|
-C, --cpu-mask <hex,hex> (default: 0x0)
|
||||||
--cpu-strict <0|1> (default: 0)
|
--cpu-strict <0|1> (default: 0)
|
||||||
|
|||||||
@@ -342,6 +342,8 @@ struct cmd_params {
|
|||||||
std::vector<llama_split_mode> split_mode;
|
std::vector<llama_split_mode> split_mode;
|
||||||
std::vector<llama_load_mode> load_mode;
|
std::vector<llama_load_mode> load_mode;
|
||||||
uint64_t longhaul_cache_bytes;
|
uint64_t longhaul_cache_bytes;
|
||||||
|
bool tokenizer_longhaul;
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes;
|
||||||
std::vector<int> main_gpu;
|
std::vector<int> main_gpu;
|
||||||
std::vector<bool> no_kv_offload;
|
std::vector<bool> no_kv_offload;
|
||||||
std::vector<llama_flash_attn_type> flash_attn;
|
std::vector<llama_flash_attn_type> flash_attn;
|
||||||
@@ -387,6 +389,8 @@ static const cmd_params cmd_params_defaults = {
|
|||||||
/* split_mode */ { LLAMA_SPLIT_MODE_LAYER },
|
/* split_mode */ { LLAMA_SPLIT_MODE_LAYER },
|
||||||
/* load_mode */ { LLAMA_LOAD_MODE_MMAP },
|
/* load_mode */ { LLAMA_LOAD_MODE_MMAP },
|
||||||
/* longhaul_cache_bytes */ 0,
|
/* longhaul_cache_bytes */ 0,
|
||||||
|
/* tokenizer_longhaul */ false,
|
||||||
|
/* tokenizer_longhaul_cache_bytes */ 0,
|
||||||
/* main_gpu */ { 0 },
|
/* main_gpu */ { 0 },
|
||||||
/* no_kv_offload */ { false },
|
/* no_kv_offload */ { false },
|
||||||
/* flash_attn */ { LLAMA_FLASH_ATTN_TYPE_AUTO },
|
/* flash_attn */ { LLAMA_FLASH_ATTN_TYPE_AUTO },
|
||||||
@@ -463,6 +467,8 @@ static void print_usage(int /* argc */, char ** argv) {
|
|||||||
printf(" -dev, --device <dev0/dev1/...> (default: auto)\n");
|
printf(" -dev, --device <dev0/dev1/...> (default: auto)\n");
|
||||||
printf(" -lm, --load-mode <none|mmap|mlock|mmap+mlock|dio|longhaul> (default: %s)\n", join(transform_to_str(cmd_params_defaults.load_mode, llama_load_mode_name), ",").c_str());
|
printf(" -lm, --load-mode <none|mmap|mlock|mmap+mlock|dio|longhaul> (default: %s)\n", join(transform_to_str(cmd_params_defaults.load_mode, llama_load_mode_name), ",").c_str());
|
||||||
printf(" --longhaul-cache <GiB> expert cache size for longhaul mode\n");
|
printf(" --longhaul-cache <GiB> expert cache size for longhaul mode\n");
|
||||||
|
printf(" --tokenizer-longhaul page tokenizer indexes through a bounded cache\n");
|
||||||
|
printf(" --tokenizer-longhaul-cache <MiB> tokenizer lookup cache size (required with --tokenizer-longhaul)\n");
|
||||||
printf(" -mmp, --mmap <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n");
|
printf(" -mmp, --mmap <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n");
|
||||||
printf(" -dio, --direct-io <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n");
|
printf(" -dio, --direct-io <0|1> (DEPRECATED IN FAVOUR OF --load-mode)\n");
|
||||||
printf(" -embd, --embeddings <0|1> (default: %s)\n", join(cmd_params_defaults.embeddings, ",").c_str());
|
printf(" -embd, --embeddings <0|1> (default: %s)\n", join(cmd_params_defaults.embeddings, ",").c_str());
|
||||||
@@ -525,6 +531,8 @@ static cmd_params parse_cmd_params(int argc, char ** argv) {
|
|||||||
params.no_warmup = cmd_params_defaults.no_warmup;
|
params.no_warmup = cmd_params_defaults.no_warmup;
|
||||||
params.offline = cmd_params_defaults.offline;
|
params.offline = cmd_params_defaults.offline;
|
||||||
params.longhaul_cache_bytes = cmd_params_defaults.longhaul_cache_bytes;
|
params.longhaul_cache_bytes = cmd_params_defaults.longhaul_cache_bytes;
|
||||||
|
params.tokenizer_longhaul = cmd_params_defaults.tokenizer_longhaul;
|
||||||
|
params.tokenizer_longhaul_cache_bytes = cmd_params_defaults.tokenizer_longhaul_cache_bytes;
|
||||||
|
|
||||||
if (const char * env = getenv("HF_TOKEN")) {
|
if (const char * env = getenv("HF_TOKEN")) {
|
||||||
params.hf_token = env;
|
params.hf_token = env;
|
||||||
@@ -801,6 +809,19 @@ static cmd_params parse_cmd_params(int argc, char ** argv) {
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
params.longhaul_cache_bytes = gib * 1024 * 1024 * 1024;
|
params.longhaul_cache_bytes = gib * 1024 * 1024 * 1024;
|
||||||
|
} else if (arg == "--tokenizer-longhaul") {
|
||||||
|
params.tokenizer_longhaul = true;
|
||||||
|
} else if (arg == "--tokenizer-longhaul-cache") {
|
||||||
|
if (++i >= argc) {
|
||||||
|
invalid_param = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
const uint64_t mib = std::stoull(argv[i]);
|
||||||
|
if (mib == 0 || mib > UINT64_MAX / 1024 / 1024) {
|
||||||
|
invalid_param = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
params.tokenizer_longhaul_cache_bytes = mib * 1024 * 1024;
|
||||||
} else if (arg == "-mg" || arg == "--main-gpu") {
|
} else if (arg == "-mg" || arg == "--main-gpu") {
|
||||||
if (++i >= argc) {
|
if (++i >= argc) {
|
||||||
invalid_param = true;
|
invalid_param = true;
|
||||||
@@ -1157,6 +1178,10 @@ static cmd_params parse_cmd_params(int argc, char ** argv) {
|
|||||||
fprintf(stderr, "error: longhaul load mode requires --longhaul-cache N\n");
|
fprintf(stderr, "error: longhaul load mode requires --longhaul-cache N\n");
|
||||||
exit(1);
|
exit(1);
|
||||||
}
|
}
|
||||||
|
if (params.tokenizer_longhaul != (params.tokenizer_longhaul_cache_bytes != 0)) {
|
||||||
|
fprintf(stderr, "error: --tokenizer-longhaul and --tokenizer-longhaul-cache must be used together\n");
|
||||||
|
exit(1);
|
||||||
|
}
|
||||||
if (params.main_gpu.empty()) {
|
if (params.main_gpu.empty()) {
|
||||||
params.main_gpu = cmd_params_defaults.main_gpu;
|
params.main_gpu = cmd_params_defaults.main_gpu;
|
||||||
}
|
}
|
||||||
@@ -1224,6 +1249,8 @@ struct cmd_params_instance {
|
|||||||
llama_split_mode split_mode;
|
llama_split_mode split_mode;
|
||||||
llama_load_mode load_mode;
|
llama_load_mode load_mode;
|
||||||
uint64_t longhaul_cache_bytes;
|
uint64_t longhaul_cache_bytes;
|
||||||
|
bool tokenizer_longhaul;
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes;
|
||||||
int main_gpu;
|
int main_gpu;
|
||||||
bool no_kv_offload;
|
bool no_kv_offload;
|
||||||
llama_flash_attn_type flash_attn;
|
llama_flash_attn_type flash_attn;
|
||||||
@@ -1246,6 +1273,8 @@ struct cmd_params_instance {
|
|||||||
mparams.split_mode = split_mode;
|
mparams.split_mode = split_mode;
|
||||||
mparams.load_mode = load_mode;
|
mparams.load_mode = load_mode;
|
||||||
mparams.longhaul_cache_bytes = longhaul_cache_bytes;
|
mparams.longhaul_cache_bytes = longhaul_cache_bytes;
|
||||||
|
mparams.tokenizer_longhaul = tokenizer_longhaul;
|
||||||
|
mparams.tokenizer_longhaul_cache_bytes = tokenizer_longhaul_cache_bytes;
|
||||||
mparams.main_gpu = main_gpu;
|
mparams.main_gpu = main_gpu;
|
||||||
mparams.tensor_split = tensor_split.data();
|
mparams.tensor_split = tensor_split.data();
|
||||||
mparams.no_host = no_host;
|
mparams.no_host = no_host;
|
||||||
@@ -1294,6 +1323,8 @@ struct cmd_params_instance {
|
|||||||
split_mode == other.split_mode &&
|
split_mode == other.split_mode &&
|
||||||
main_gpu == other.main_gpu && tensor_split == other.tensor_split &&
|
main_gpu == other.main_gpu && tensor_split == other.tensor_split &&
|
||||||
load_mode == other.load_mode && longhaul_cache_bytes == other.longhaul_cache_bytes &&
|
load_mode == other.load_mode && longhaul_cache_bytes == other.longhaul_cache_bytes &&
|
||||||
|
tokenizer_longhaul == other.tokenizer_longhaul &&
|
||||||
|
tokenizer_longhaul_cache_bytes == other.tokenizer_longhaul_cache_bytes &&
|
||||||
devices == other.devices && no_host == other.no_host &&
|
devices == other.devices && no_host == other.no_host &&
|
||||||
vec_tensor_buft_override_equal(tensor_buft_overrides, other.tensor_buft_overrides);
|
vec_tensor_buft_override_equal(tensor_buft_overrides, other.tensor_buft_overrides);
|
||||||
}
|
}
|
||||||
@@ -1368,6 +1399,8 @@ static std::vector<cmd_params_instance> get_cmd_params_instances(const cmd_param
|
|||||||
/* .split_mode = */ sm,
|
/* .split_mode = */ sm,
|
||||||
/* .load_mode = */ lm,
|
/* .load_mode = */ lm,
|
||||||
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
|
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
|
||||||
|
/* .tokenizer_longhaul = */ params.tokenizer_longhaul,
|
||||||
|
/* .tokenizer_longhaul_cache_bytes = */ params.tokenizer_longhaul_cache_bytes,
|
||||||
/* .main_gpu = */ mg,
|
/* .main_gpu = */ mg,
|
||||||
/* .no_kv_offload = */ nkvo,
|
/* .no_kv_offload = */ nkvo,
|
||||||
/* .flash_attn = */ fa,
|
/* .flash_attn = */ fa,
|
||||||
@@ -1405,6 +1438,8 @@ static std::vector<cmd_params_instance> get_cmd_params_instances(const cmd_param
|
|||||||
/* .split_mode = */ sm,
|
/* .split_mode = */ sm,
|
||||||
/* .load_mode = */ lm,
|
/* .load_mode = */ lm,
|
||||||
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
|
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
|
||||||
|
/* .tokenizer_longhaul = */ params.tokenizer_longhaul,
|
||||||
|
/* .tokenizer_longhaul_cache_bytes = */ params.tokenizer_longhaul_cache_bytes,
|
||||||
/* .main_gpu = */ mg,
|
/* .main_gpu = */ mg,
|
||||||
/* .no_kv_offload = */ nkvo,
|
/* .no_kv_offload = */ nkvo,
|
||||||
/* .flash_attn = */ fa,
|
/* .flash_attn = */ fa,
|
||||||
@@ -1442,6 +1477,8 @@ static std::vector<cmd_params_instance> get_cmd_params_instances(const cmd_param
|
|||||||
/* .split_mode = */ sm,
|
/* .split_mode = */ sm,
|
||||||
/* .load_mode = */ lm,
|
/* .load_mode = */ lm,
|
||||||
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
|
/* .longhaul_cache_bytes = */ params.longhaul_cache_bytes,
|
||||||
|
/* .tokenizer_longhaul = */ params.tokenizer_longhaul,
|
||||||
|
/* .tokenizer_longhaul_cache_bytes = */ params.tokenizer_longhaul_cache_bytes,
|
||||||
/* .main_gpu = */ mg,
|
/* .main_gpu = */ mg,
|
||||||
/* .no_kv_offload = */ nkvo,
|
/* .no_kv_offload = */ nkvo,
|
||||||
/* .flash_attn = */ fa,
|
/* .flash_attn = */ fa,
|
||||||
@@ -1484,6 +1521,8 @@ struct test {
|
|||||||
llama_split_mode split_mode;
|
llama_split_mode split_mode;
|
||||||
llama_load_mode load_mode;
|
llama_load_mode load_mode;
|
||||||
uint64_t longhaul_cache_bytes;
|
uint64_t longhaul_cache_bytes;
|
||||||
|
bool tokenizer_longhaul;
|
||||||
|
uint64_t tokenizer_longhaul_cache_bytes;
|
||||||
int main_gpu;
|
int main_gpu;
|
||||||
bool no_kv_offload;
|
bool no_kv_offload;
|
||||||
llama_flash_attn_type flash_attn;
|
llama_flash_attn_type flash_attn;
|
||||||
@@ -1524,6 +1563,8 @@ struct test {
|
|||||||
split_mode = inst.split_mode;
|
split_mode = inst.split_mode;
|
||||||
load_mode = inst.load_mode;
|
load_mode = inst.load_mode;
|
||||||
longhaul_cache_bytes = inst.longhaul_cache_bytes;
|
longhaul_cache_bytes = inst.longhaul_cache_bytes;
|
||||||
|
tokenizer_longhaul = inst.tokenizer_longhaul;
|
||||||
|
tokenizer_longhaul_cache_bytes = inst.tokenizer_longhaul_cache_bytes;
|
||||||
main_gpu = inst.main_gpu;
|
main_gpu = inst.main_gpu;
|
||||||
no_kv_offload = inst.no_kv_offload;
|
no_kv_offload = inst.no_kv_offload;
|
||||||
flash_attn = inst.flash_attn;
|
flash_attn = inst.flash_attn;
|
||||||
@@ -1591,7 +1632,8 @@ struct test {
|
|||||||
"n_ubatch", "n_threads", "cpu_mask", "cpu_strict", "poll",
|
"n_ubatch", "n_threads", "cpu_mask", "cpu_strict", "poll",
|
||||||
"type_k", "type_v", "n_gpu_layers", "n_cpu_moe", "split_mode",
|
"type_k", "type_v", "n_gpu_layers", "n_cpu_moe", "split_mode",
|
||||||
"main_gpu", "no_kv_offload", "flash_attn", "devices", "tensor_split",
|
"main_gpu", "no_kv_offload", "flash_attn", "devices", "tensor_split",
|
||||||
"tensor_buft_overrides", "load_mode", "longhaul_cache_bytes", "embeddings",
|
"tensor_buft_overrides", "load_mode", "longhaul_cache_bytes",
|
||||||
|
"tokenizer_longhaul", "tokenizer_longhaul_cache_bytes", "embeddings",
|
||||||
"no_op_offload", "no_host", "fit_target", "fit_min_ctx",
|
"no_op_offload", "no_host", "fit_target", "fit_min_ctx",
|
||||||
"n_prompt", "n_gen", "n_depth",
|
"n_prompt", "n_gen", "n_depth",
|
||||||
"test_time", "avg_ns", "stddev_ns", "avg_ts", "stddev_ts"
|
"test_time", "avg_ns", "stddev_ns", "avg_ts", "stddev_ts"
|
||||||
@@ -1607,11 +1649,11 @@ struct test {
|
|||||||
field == "main_gpu" || field == "n_prompt" || field == "n_gen" || field == "n_depth" || field == "avg_ns" ||
|
field == "main_gpu" || field == "n_prompt" || field == "n_gen" || field == "n_depth" || field == "avg_ns" ||
|
||||||
field == "stddev_ns" || field == "no_op_offload" || field == "n_cpu_moe" ||
|
field == "stddev_ns" || field == "no_op_offload" || field == "n_cpu_moe" ||
|
||||||
field == "fit_target" || field == "fit_min_ctx" || field == "flash_attn" ||
|
field == "fit_target" || field == "fit_min_ctx" || field == "flash_attn" ||
|
||||||
field == "longhaul_cache_bytes") {
|
field == "longhaul_cache_bytes" || field == "tokenizer_longhaul_cache_bytes") {
|
||||||
return INT;
|
return INT;
|
||||||
}
|
}
|
||||||
if (field == "f16_kv" || field == "no_kv_offload" || field == "cpu_strict" ||
|
if (field == "f16_kv" || field == "no_kv_offload" || field == "cpu_strict" ||
|
||||||
field == "embeddings" || field == "no_host") {
|
field == "embeddings" || field == "no_host" || field == "tokenizer_longhaul") {
|
||||||
return BOOL;
|
return BOOL;
|
||||||
}
|
}
|
||||||
if (field == "avg_ts" || field == "stddev_ts") {
|
if (field == "avg_ts" || field == "stddev_ts") {
|
||||||
@@ -1688,6 +1730,8 @@ struct test {
|
|||||||
tensor_buft_overrides_str,
|
tensor_buft_overrides_str,
|
||||||
llama_load_mode_name(load_mode),
|
llama_load_mode_name(load_mode),
|
||||||
std::to_string(longhaul_cache_bytes),
|
std::to_string(longhaul_cache_bytes),
|
||||||
|
std::to_string(tokenizer_longhaul),
|
||||||
|
std::to_string(tokenizer_longhaul_cache_bytes),
|
||||||
std::to_string(embeddings),
|
std::to_string(embeddings),
|
||||||
std::to_string(no_op_offload),
|
std::to_string(no_op_offload),
|
||||||
std::to_string(no_host),
|
std::to_string(no_host),
|
||||||
@@ -2005,6 +2049,12 @@ struct markdown_printer : public printer {
|
|||||||
if (params.longhaul_cache_bytes != cmd_params_defaults.longhaul_cache_bytes) {
|
if (params.longhaul_cache_bytes != cmd_params_defaults.longhaul_cache_bytes) {
|
||||||
fields.emplace_back("longhaul_cache_bytes");
|
fields.emplace_back("longhaul_cache_bytes");
|
||||||
}
|
}
|
||||||
|
if (params.tokenizer_longhaul != cmd_params_defaults.tokenizer_longhaul) {
|
||||||
|
fields.emplace_back("tokenizer_longhaul");
|
||||||
|
}
|
||||||
|
if (params.tokenizer_longhaul_cache_bytes != cmd_params_defaults.tokenizer_longhaul_cache_bytes) {
|
||||||
|
fields.emplace_back("tokenizer_longhaul_cache_bytes");
|
||||||
|
}
|
||||||
if (params.embeddings.size() > 1 || params.embeddings != cmd_params_defaults.embeddings) {
|
if (params.embeddings.size() > 1 || params.embeddings != cmd_params_defaults.embeddings) {
|
||||||
fields.emplace_back("embeddings");
|
fields.emplace_back("embeddings");
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -56,6 +56,8 @@ For the full list of features, please refer to [server's changelog](https://gith
|
|||||||
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
|
| `--token-cache-size N` | shared tokenizer and tokenized-text cache size in MiB (default: 5120, 0 = disable)<br/>(env: LLAMA_ARG_TOKEN_CACHE_SIZE) |
|
||||||
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
|
| `--token-cache-dir PATH` | path to the shared tokenizer and tokenized-text cache database<br/>(env: LLAMA_ARG_TOKEN_CACHE_DIR) |
|
||||||
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
|
| `--token-cache-clear` | clear the shared tokenizer and tokenized-text cache before loading the model<br/>(env: LLAMA_ARG_TOKEN_CACHE_CLEAR) |
|
||||||
|
| `--tokenizer-longhaul` | build and page the Qwen3.5/Laguna tokenizer through a bounded cache<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL) |
|
||||||
|
| `--tokenizer-longhaul-cache N` | tokenizer-longhaul RAM cache size in MiB<br/>(env: LLAMA_ARG_TOKENIZER_LONGHAUL_CACHE) |
|
||||||
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
|
| `-fa, --flash-attn [on\|off\|auto]` | set Flash Attention use ('on', 'off', or 'auto', default: 'auto')<br/>(env: LLAMA_ARG_FLASH_ATTN) |
|
||||||
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
|
| `--perf, --no-perf` | whether to enable internal libllama performance timings (default: false)<br/>(env: LLAMA_ARG_PERF) |
|
||||||
| `-e, --escape, --no-escape` | whether to process escapes sequences (\n, \r, \t, \', \", \\) (default: true) |
|
| `-e, --escape, --no-escape` | whether to process escapes sequences (\n, \r, \t, \', \", \\) (default: true) |
|
||||||
|
|||||||
Reference in New Issue
Block a user