mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
llama: add lazy mode auto, fix iGPU regression
This commit is contained in:
+4
-3
@@ -215,9 +215,10 @@ extern "C" {
|
||||
LLAMA_API enum llama_load_mode llama_load_mode_from_str(const char * str);
|
||||
|
||||
enum llama_lazy_mode {
|
||||
LLAMA_LAZY_MODE_OFF = 0, // always read the whole tensor up front
|
||||
LLAMA_LAZY_MODE_AUTO = 1, // lazy only for marked tensors larger than 4 GiB (requires mmap)
|
||||
LLAMA_LAZY_MODE_ON = 2, // read the rows of tensors marked by the arch on demand (requires mmap)
|
||||
LLAMA_LAZY_MODE_AUTO = 0, // pick a good default for the system: LARGE where mmap is supported, else OFF
|
||||
LLAMA_LAZY_MODE_OFF = 1, // always read the whole tensor up front
|
||||
LLAMA_LAZY_MODE_LARGE = 2, // read the rows on demand, but only for marked tensors larger than 4 GiB (requires mmap)
|
||||
LLAMA_LAZY_MODE_ALL = 3, // read the rows of all tensors marked by the arch on demand (requires mmap)
|
||||
};
|
||||
|
||||
enum llama_context_type {
|
||||
|
||||
Reference in New Issue
Block a user