llama: add lazy mode auto, fix iGPU regression

This commit is contained in:
Ruben Ortlam
2026-09-03 15:25:24 +02:00
parent de8656bd94
commit 38d20da312
5 changed files with 38 additions and 18 deletions
+4 -3
View File
@@ -215,9 +215,10 @@ extern "C" {
LLAMA_API enum llama_load_mode llama_load_mode_from_str(const char * str);
enum llama_lazy_mode {
LLAMA_LAZY_MODE_OFF = 0, // always read the whole tensor up front
LLAMA_LAZY_MODE_AUTO = 1, // lazy only for marked tensors larger than 4 GiB (requires mmap)
LLAMA_LAZY_MODE_ON = 2, // read the rows of tensors marked by the arch on demand (requires mmap)
LLAMA_LAZY_MODE_AUTO = 0, // pick a good default for the system: LARGE where mmap is supported, else OFF
LLAMA_LAZY_MODE_OFF = 1, // always read the whole tensor up front
LLAMA_LAZY_MODE_LARGE = 2, // read the rows on demand, but only for marked tensors larger than 4 GiB (requires mmap)
LLAMA_LAZY_MODE_ALL = 3, // read the rows of all tensors marked by the arch on demand (requires mmap)
};
enum llama_context_type {