Skip to content

Commit 0ee32f5

Browse files
committed
feat: update llama.cpp to v0.4.0
1 parent 3691546 commit 0ee32f5

6 files changed

Lines changed: 164 additions & 39 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
77

88
## [Unreleased]
99

10+
- feat: update llama.cpp to ggml-org/llama.cpp@v0.4.0
11+
1012
## [0.3.35]
1113

1214
- feat: update llama.cpp to ggml-org/llama.cpp@4df29be4f

‎CMakeLists.txt‎

Lines changed: 4 additions & 19 deletions
Original file line numberDiff line numberDiff line change
@@ -187,25 +187,10 @@ if (LLAMA_BUILD)
187187
add_compile_definitions(GGML_USE_METAL)
188188
endif()
189189

190-
# Upstream mtmd expects LLAMA_INSTALL_VERSION to be set by llama.cpp's
191-
# top-level CMakeLists.txt. When we include tools/mtmd directly from the
192-
# Python package build, that directory scope is skipped.
193-
if (NOT DEFINED LLAMA_INSTALL_VERSION OR "${LLAMA_INSTALL_VERSION}" STREQUAL "")
194-
set(LLAMA_INSTALL_VERSION 0.0.0)
195-
find_package(Git QUIET)
196-
if (Git_FOUND)
197-
execute_process(
198-
COMMAND ${GIT_EXECUTABLE} rev-list --count HEAD
199-
WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/vendor/llama.cpp
200-
OUTPUT_VARIABLE LLAMA_MTMD_BUILD_NUMBER
201-
OUTPUT_STRIP_TRAILING_WHITESPACE
202-
RESULT_VARIABLE LLAMA_MTMD_BUILD_NUMBER_RESULT
203-
)
204-
if (LLAMA_MTMD_BUILD_NUMBER_RESULT EQUAL 0)
205-
set(LLAMA_INSTALL_VERSION 0.0.${LLAMA_MTMD_BUILD_NUMBER})
206-
endif()
207-
endif()
208-
endif()
190+
# tools/mtmd is added outside llama.cpp's directory scope, so inherit
191+
# its library version variables explicitly.
192+
get_directory_property(LLAMA_VERSION_BASE DIRECTORY vendor/llama.cpp DEFINITION LLAMA_VERSION_BASE)
193+
get_directory_property(LLAMA_VERSION_MAJOR DIRECTORY vendor/llama.cpp DEFINITION LLAMA_VERSION_MAJOR)
209194

210195
# Building llava
211196
add_subdirectory(vendor/llama.cpp/tools/mtmd)

‎examples/server/server.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -10748,6 +10748,7 @@ def _create_loaded_media(
1074810748
buffer,
1074910749
len(media_bytes),
1075010750
False,
10751+
mtmd_cpp.mtmd_helper_init_opt_default(),
1075110752
)
1075210753
bitmap = wrapper.bitmap
1075310754
if bitmap is None:

‎llama_cpp/llama_cpp.py‎

Lines changed: 23 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -162,13 +162,13 @@ def _warn_deprecated(symbol: str, hint: str) -> None:
162162

163163
# define LLAMA_SESSION_MAGIC LLAMA_FILE_MAGIC_GGSN
164164
LLAMA_SESSION_MAGIC = LLAMA_FILE_MAGIC_GGSN
165-
# define LLAMA_SESSION_VERSION 9
166-
LLAMA_SESSION_VERSION = 9
165+
# define LLAMA_SESSION_VERSION 10
166+
LLAMA_SESSION_VERSION = 10
167167

168168
# define LLAMA_STATE_SEQ_MAGIC LLAMA_FILE_MAGIC_GGSQ
169169
LLAMA_STATE_SEQ_MAGIC = LLAMA_FILE_MAGIC_GGSQ
170-
# define LLAMA_STATE_SEQ_VERSION 2
171-
LLAMA_STATE_SEQ_VERSION = 2
170+
# define LLAMA_STATE_SEQ_VERSION 3
171+
LLAMA_STATE_SEQ_VERSION = 3
172172

173173
# struct llama_vocab;
174174
llama_vocab_p = NewType("llama_vocab_p", int)
@@ -536,6 +536,16 @@ def _warn_deprecated(symbol: str, hint: str) -> None:
536536
LLAMA_LOAD_MODE_DIRECT_IO = 4
537537

538538

539+
# enum llama_lazy_mode {
540+
# LLAMA_LAZY_MODE_OFF = 0, // always read the whole tensor up front
541+
# LLAMA_LAZY_MODE_AUTO = 1, // lazy only for marked tensors larger than 4 GiB (requires mmap)
542+
# LLAMA_LAZY_MODE_ON = 2, // read the rows of tensors marked by the arch on demand (requires mmap)
543+
# };
544+
LLAMA_LAZY_MODE_OFF = 0
545+
LLAMA_LAZY_MODE_AUTO = 1
546+
LLAMA_LAZY_MODE_ON = 2
547+
548+
539549
# enum llama_context_type {
540550
# LLAMA_CONTEXT_TYPE_DEFAULT = 0,
541551
# LLAMA_CONTEXT_TYPE_MTP = 1,
@@ -809,6 +819,8 @@ class llama_model_imatrix_data(ctypes.Structure):
809819
# enum llama_split_mode split_mode; // how to split the model across multiple GPUs
810820
# enum llama_load_mode load_mode; // how to load the model
811821

822+
# enum llama_lazy_mode lazy_mode; // on-demand reading of tensors marked by the arch
823+
812824
# // the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
813825
# int32_t main_gpu;
814826

@@ -844,6 +856,7 @@ class llama_model_params(ctypes.Structure):
844856
n_gpu_layers (int): number of layers to store in VRAM, a negative value means all layers
845857
split_mode (int): how to split the model across multiple GPUs
846858
load_mode (int): how to load the model
859+
lazy_mode (int): on-demand reading of tensors marked by the arch
847860
main_gpu (int): the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
848861
tensor_split (ctypes.Array[ctypes.ctypes.c_float]): proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()
849862
progress_callback (llama_progress_callback): called with a progress value between 0.0 and 1.0. Pass NULL to disable. If the provided progress_callback returns true, model loading continues. If it returns false, model loading is immediately aborted.
@@ -864,6 +877,7 @@ class llama_model_params(ctypes.Structure):
864877
n_gpu_layers: int
865878
split_mode: int
866879
load_mode: int
880+
lazy_mode: int
867881
main_gpu: int
868882
tensor_split: CtypesArray[ctypes.c_float]
869883
progress_callback: Callable[[float, ctypes.c_void_p], bool]
@@ -882,6 +896,7 @@ class llama_model_params(ctypes.Structure):
882896
("n_gpu_layers", ctypes.c_int32),
883897
("split_mode", ctypes.c_int),
884898
("load_mode", ctypes.c_int),
899+
("lazy_mode", ctypes.c_int),
885900
("main_gpu", ctypes.c_int32),
886901
("tensor_split", ctypes.POINTER(ctypes.c_float)),
887902
("progress_callback", llama_progress_callback),
@@ -1126,6 +1141,7 @@ class llama_context_params(ctypes.Structure):
11261141
# const struct llama_model_kv_override * kv_overrides; // pointer to kv overrides
11271142
# const struct llama_model_tensor_override * tt_overrides; // pointer to tensor overrides
11281143
# const int32_t * prune_layers; // pointer to layer indices to prune
1144+
# size_t max_buf_size; // max bytes of tensor rows kept in memory at once, 0 = default (8 GiB)
11291145
# } llama_model_quantize_params;
11301146
class llama_model_quantize_params(ctypes.Structure):
11311147
"""Parameters for llama_model_quantize
@@ -1145,6 +1161,7 @@ class llama_model_quantize_params(ctypes.Structure):
11451161
kv_overrides (ctypes.Array[llama_model_kv_override]): pointer to kv overrides
11461162
tt_overrides (ctypes.Array[llama_model_tensor_override]): pointer to tensor overrides
11471163
prune_layers (ctypes.Array[ctypes.c_int32]): pointer to layer indices to prune
1164+
max_buf_size (int): max bytes of tensor rows kept in memory at once, 0 = default (8 GiB)
11481165
"""
11491166

11501167
if TYPE_CHECKING:
@@ -1162,6 +1179,7 @@ class llama_model_quantize_params(ctypes.Structure):
11621179
kv_overrides: CtypesPointer[llama_model_kv_override]
11631180
tt_overrides: CtypesPointer[llama_model_tensor_override]
11641181
prune_layers: CtypesPointer[ctypes.c_int32]
1182+
max_buf_size: int
11651183

11661184
_fields_ = [
11671185
("nthread", ctypes.c_int32),
@@ -1178,6 +1196,7 @@ class llama_model_quantize_params(ctypes.Structure):
11781196
("kv_overrides", ctypes.POINTER(llama_model_kv_override)),
11791197
("tt_overrides", ctypes.POINTER(llama_model_tensor_override)),
11801198
("prune_layers", ctypes.POINTER(ctypes.c_int32)),
1199+
("max_buf_size", ctypes.c_size_t),
11811200
]
11821201

11831202

0 commit comments

Comments
 (0)