You signed in with another tab or window. Reload to refresh your session.You signed out in another tab or window. Reload to refresh your session.You switched accounts on another tab or window. Reload to refresh your session.Dismiss alert
# LLAMA_LAZY_MODE_OFF = 0, // always read the whole tensor up front
541
+
# LLAMA_LAZY_MODE_AUTO = 1, // lazy only for marked tensors larger than 4 GiB (requires mmap)
542
+
# LLAMA_LAZY_MODE_ON = 2, // read the rows of tensors marked by the arch on demand (requires mmap)
543
+
# };
544
+
LLAMA_LAZY_MODE_OFF=0
545
+
LLAMA_LAZY_MODE_AUTO=1
546
+
LLAMA_LAZY_MODE_ON=2
547
+
548
+
539
549
# enum llama_context_type {
540
550
# LLAMA_CONTEXT_TYPE_DEFAULT = 0,
541
551
# LLAMA_CONTEXT_TYPE_MTP = 1,
@@ -809,6 +819,8 @@ class llama_model_imatrix_data(ctypes.Structure):
809
819
# enum llama_split_mode split_mode; // how to split the model across multiple GPUs
810
820
# enum llama_load_mode load_mode; // how to load the model
811
821
822
+
# enum llama_lazy_mode lazy_mode; // on-demand reading of tensors marked by the arch
823
+
812
824
# // the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
813
825
# int32_t main_gpu;
814
826
@@ -844,6 +856,7 @@ class llama_model_params(ctypes.Structure):
844
856
n_gpu_layers (int): number of layers to store in VRAM, a negative value means all layers
845
857
split_mode (int): how to split the model across multiple GPUs
846
858
load_mode (int): how to load the model
859
+
lazy_mode (int): on-demand reading of tensors marked by the arch
847
860
main_gpu (int): the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
848
861
tensor_split (ctypes.Array[ctypes.ctypes.c_float]): proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()
849
862
progress_callback (llama_progress_callback): called with a progress value between 0.0 and 1.0. Pass NULL to disable. If the provided progress_callback returns true, model loading continues. If it returns false, model loading is immediately aborted.
@@ -864,6 +877,7 @@ class llama_model_params(ctypes.Structure):
0 commit comments