Inkling longhaul support

This commit is contained in:
Owen Qwen
2026-08-03 20:50:22 -05:00
parent e3b2925150
commit ac9907e841
28 changed files with 1375 additions and 34 deletions
+1 -1
View File
@@ -2631,7 +2631,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
).set_env("LLAMA_ARG_LOAD_MODE"));
add_opt(common_arg(
{"--longhaul"},
"stream routed Qwen3.5 MoE experts through a bounded Metal cache",
"stream routed Qwen3.5 MoE, Laguna, or Inkling experts through a bounded CPU or Metal cache",
[](common_params & params) {
params.load_mode = LLAMA_LOAD_MODE_LONGHAUL;
}