Skip to content

Commit 1de1569

Browse files
committed
feat: update llama.cpp to d2a818231
1 parent 629bd1b commit 1de1569

9 files changed

Lines changed: 82 additions & 30 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
77

88
## [Unreleased]
99

10+
- feat: update llama.cpp to ggml-org/llama.cpp@d2a818231
11+
1012
## [0.3.34]
1113

1214
- feat: update llama.cpp to ggml-org/llama.cpp@e3546c794

‎examples/low_level_api/low_level_api_chat_cpp.py‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -76,8 +76,12 @@ def __init__(self, params: GptParams) -> None:
7676
self.lparams.n_parts = self.params.n_parts
7777
self.lparams.seed = self.params.seed
7878
self.lparams.memory_f16 = self.params.memory_f16
79-
self.lparams.use_mlock = self.params.use_mlock
80-
self.lparams.use_mmap = self.params.use_mmap
79+
if self.params.use_mlock:
80+
self.lparams.load_mode = llama_cpp.LLAMA_LOAD_MODE_MLOCK
81+
elif self.params.use_mmap:
82+
self.lparams.load_mode = llama_cpp.LLAMA_LOAD_MODE_MMAP
83+
else:
84+
self.lparams.load_mode = llama_cpp.LLAMA_LOAD_MODE_NONE
8185

8286
self.model = llama_cpp.llama_model_load_from_file(
8387
self.params.model.encode("utf8"), self.lparams

‎examples/server/server.py‎

Lines changed: 11 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -11030,7 +11030,9 @@ def _build_prompt_plan_locked(
1103011030
"multiple videos require MTMD to report frame counts"
1103111031
)
1103211032
input_text = mtmd_cpp.mtmd_input_text()
11033-
input_text.text = prompt.encode("utf-8")
11033+
input_text_bytes = prompt.encode("utf-8")
11034+
input_text.text = input_text_bytes
11035+
input_text.text_len = len(input_text_bytes)
1103411036
input_text.add_special = False
1103511037
input_text.parse_special = True
1103611038
chunks = mtmd_cpp.mtmd_input_chunks_init()
@@ -11609,10 +11611,14 @@ def build_model_params(
1160911611
model_params.tensor_split = tensor_split_ref
1161011612
if vocab_only is not None:
1161111613
model_params.vocab_only = vocab_only
11612-
if use_mmap is not None:
11613-
model_params.use_mmap = use_mmap
11614-
if use_mlock is not None:
11615-
model_params.use_mlock = use_mlock
11614+
if use_mlock:
11615+
model_params.load_mode = llama_cpp.LLAMA_LOAD_MODE_MLOCK
11616+
elif use_mmap is not None:
11617+
model_params.load_mode = (
11618+
llama_cpp.LLAMA_LOAD_MODE_MMAP
11619+
if use_mmap
11620+
else llama_cpp.LLAMA_LOAD_MODE_NONE
11621+
)
1161611622

1161711623
kv_overrides_ref = None
1161811624
if kv_overrides is not None:

‎llama_cpp/llama.py‎

Lines changed: 11 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -244,8 +244,14 @@ def __init__(
244244
) # keep a reference to the array so it is not gc'd
245245
self.model_params.tensor_split = self._c_tensor_split
246246
self.model_params.vocab_only = vocab_only
247-
self.model_params.use_mmap = use_mmap if lora_path is None else False
248-
self.model_params.use_mlock = use_mlock
247+
if lora_path is not None:
248+
self.model_params.load_mode = llama_cpp.LLAMA_LOAD_MODE_NONE
249+
elif use_mlock:
250+
self.model_params.load_mode = llama_cpp.LLAMA_LOAD_MODE_MLOCK
251+
elif use_mmap:
252+
self.model_params.load_mode = llama_cpp.LLAMA_LOAD_MODE_MMAP
253+
else:
254+
self.model_params.load_mode = llama_cpp.LLAMA_LOAD_MODE_NONE
249255

250256
# kv_overrides is the original python dict
251257
self.kv_overrides = kv_overrides
@@ -2142,8 +2148,9 @@ def __getstate__(self):
21422148
main_gpu=self.model_params.main_gpu,
21432149
tensor_split=self.tensor_split,
21442150
vocab_only=self.model_params.vocab_only,
2145-
use_mmap=self.model_params.use_mmap,
2146-
use_mlock=self.model_params.use_mlock,
2151+
use_mmap=self.model_params.load_mode
2152+
in (llama_cpp.LLAMA_LOAD_MODE_MMAP, llama_cpp.LLAMA_LOAD_MODE_MLOCK),
2153+
use_mlock=self.model_params.load_mode == llama_cpp.LLAMA_LOAD_MODE_MLOCK,
21472154
kv_overrides=self.kv_overrides,
21482155
# Context Params
21492156
seed=self._seed,

‎llama_cpp/llama_chat_format.py‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2936,7 +2936,9 @@ def __call__(
29362936

29372937
# Create input text structure
29382938
input_text = self._mtmd_cpp.mtmd_input_text()
2939-
input_text.text = text.encode("utf-8")
2939+
input_text_bytes = text.encode("utf-8")
2940+
input_text.text = input_text_bytes
2941+
input_text.text_len = len(input_text_bytes)
29402942
input_text.add_special = True
29412943
input_text.parse_special = True
29422944

@@ -3485,7 +3487,9 @@ def raise_exception(message: str):
34853487
bitmap_cleanup.append(bitmap)
34863488

34873489
input_text = self._mtmd_cpp.mtmd_input_text()
3488-
input_text.text = text.encode("utf-8")
3490+
input_text_bytes = text.encode("utf-8")
3491+
input_text.text = input_text_bytes
3492+
input_text.text_len = len(input_text_bytes)
34893493
input_text.add_special = True
34903494
input_text.parse_special = True
34913495

‎llama_cpp/llama_cpp.py‎

Lines changed: 32 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -520,6 +520,18 @@ def _warn_deprecated(symbol: str, hint: str) -> None:
520520
LLAMA_SPLIT_MODE_TENSOR = 3
521521

522522

523+
# enum llama_load_mode {
524+
# LLAMA_LOAD_MODE_NONE = 0, // no special loading mode
525+
# LLAMA_LOAD_MODE_MMAP = 1, // memory map the model
526+
# LLAMA_LOAD_MODE_MLOCK = 2, // mmap + force system to keep model in RAM rather than swapping or compressing
527+
# LLAMA_LOAD_MODE_DIRECT_IO = 3, // use direct I/O if available
528+
# };
529+
LLAMA_LOAD_MODE_NONE = 0
530+
LLAMA_LOAD_MODE_MMAP = 1
531+
LLAMA_LOAD_MODE_MLOCK = 2
532+
LLAMA_LOAD_MODE_DIRECT_IO = 3
533+
534+
523535
# enum llama_context_type {
524536
# LLAMA_CONTEXT_TYPE_DEFAULT = 0,
525537
# LLAMA_CONTEXT_TYPE_MTP = 1,
@@ -789,8 +801,9 @@ class llama_model_imatrix_data(ctypes.Structure):
789801
# // NULL-terminated list of buffer types to use for tensors that match a pattern
790802
# const struct llama_model_tensor_buft_override * tensor_buft_overrides;
791803

792-
# int32_t n_gpu_layers; // number of layers to store in VRAM
804+
# int32_t n_gpu_layers; // number of layers to store in VRAM, a negative value means all layers
793805
# enum llama_split_mode split_mode; // how to split the model across multiple GPUs
806+
# enum llama_load_mode load_mode; // how to load the model
794807

795808
# // the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
796809
# int32_t main_gpu;
@@ -812,9 +825,6 @@ class llama_model_imatrix_data(ctypes.Structure):
812825

813826
# // Keep the booleans together to avoid misalignment during copy-by-value.
814827
# bool vocab_only; // only load the vocabulary, no weights
815-
# bool use_mmap; // use mmap if possible
816-
# bool use_direct_io; // use direct io, takes precedence over use_mmap when supported
817-
# bool use_mlock; // force system to keep model in RAM
818828
# bool check_tensors; // validate model tensor data
819829
# bool use_extra_bufts; // use extra buffer types (used for weight repacking)
820830
# bool no_host; // bypass host buffer allowing extra buffers to be used
@@ -826,17 +836,15 @@ class llama_model_params(ctypes.Structure):
826836
Attributes:
827837
devices (ctypes.Array[ggml_backend_dev_t]): NULL-terminated list of devices to use for offloading (if NULL, all available devices are used)
828838
tensor_buft_overrides (ctypes.Array[llama_model_tensor_buft_override]): NULL-terminated list of buffer types to use for tensors that match a pattern
829-
n_gpu_layers (int): number of layers to store in VRAM
839+
n_gpu_layers (int): number of layers to store in VRAM, a negative value means all layers
830840
split_mode (int): how to split the model across multiple GPUs
841+
load_mode (int): how to load the model
831842
main_gpu (int): the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE
832843
tensor_split (ctypes.Array[ctypes.ctypes.c_float]): proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()
833844
progress_callback (llama_progress_callback): called with a progress value between 0.0 and 1.0. Pass NULL to disable. If the provided progress_callback returns true, model loading continues. If it returns false, model loading is immediately aborted.
834845
progress_callback_user_data (ctypes.ctypes.c_void_p): context pointer passed to the progress callback
835846
kv_overrides (ctypes.Array[llama_model_kv_override]): override key-value pairs of the model meta data
836847
vocab_only (bool): only load the vocabulary, no weights
837-
use_mmap (bool): use mmap if possible
838-
use_direct_io (bool): use direct io, takes precedence over use_mmap when supported
839-
use_mlock (bool): force system to keep model in RAM
840848
check_tensors (bool): validate model tensor data
841849
use_extra_bufts (bool): use extra buffer types (used for weight repacking)
842850
no_host (bool): bypass host buffer allowing extra buffers to be used
@@ -849,15 +857,13 @@ class llama_model_params(ctypes.Structure):
849857
] # NOTE: unused
850858
n_gpu_layers: int
851859
split_mode: int
860+
load_mode: int
852861
main_gpu: int
853862
tensor_split: CtypesArray[ctypes.c_float]
854863
progress_callback: Callable[[float, ctypes.c_void_p], bool]
855864
progress_callback_user_data: ctypes.c_void_p
856865
kv_overrides: CtypesArray[llama_model_kv_override]
857866
vocab_only: bool
858-
use_mmap: bool
859-
use_direct_io: bool
860-
use_mlock: bool
861867
check_tensors: bool
862868
use_extra_bufts: bool
863869
no_host: bool
@@ -868,15 +874,13 @@ class llama_model_params(ctypes.Structure):
868874
("tensor_buft_overrides", ctypes.c_void_p), # NOTE: unused
869875
("n_gpu_layers", ctypes.c_int32),
870876
("split_mode", ctypes.c_int),
877+
("load_mode", ctypes.c_int),
871878
("main_gpu", ctypes.c_int32),
872879
("tensor_split", ctypes.POINTER(ctypes.c_float)),
873880
("progress_callback", llama_progress_callback),
874881
("progress_callback_user_data", ctypes.c_void_p),
875882
("kv_overrides", ctypes.POINTER(llama_model_kv_override)),
876883
("vocab_only", ctypes.c_bool),
877-
("use_mmap", ctypes.c_bool),
878-
("use_direct_io", ctypes.c_bool),
879-
("use_mlock", ctypes.c_bool),
880884
("check_tensors", ctypes.c_bool),
881885
("use_extra_bufts", ctypes.c_bool),
882886
("no_host", ctypes.c_bool),
@@ -1276,6 +1280,20 @@ def llama_flash_attn_type_name(flash_attn_type: int, /) -> Optional[bytes]:
12761280
...
12771281

12781282

1283+
# LLAMA_API const char * llama_load_mode_name(enum llama_load_mode load_mode);
1284+
@ctypes_function("llama_load_mode_name", [ctypes.c_int], ctypes.c_char_p)
1285+
def llama_load_mode_name(load_mode: int, /) -> Optional[bytes]:
1286+
"""Get the model load mode name."""
1287+
...
1288+
1289+
1290+
# LLAMA_API enum llama_load_mode llama_load_mode_from_str(const char * str);
1291+
@ctypes_function("llama_load_mode_from_str", [ctypes.c_char_p], ctypes.c_int)
1292+
def llama_load_mode_from_str(value: bytes, /) -> int:
1293+
"""Get the model load mode from a string."""
1294+
...
1295+
1296+
12791297
# // Get the model file type (quantization) as a string, e.g. "Q8_0" or "Q4_K - Medium"
12801298
# LLAMA_API const char * llama_ftype_name(enum llama_ftype ftype);
12811299
@ctypes_function("llama_ftype_name", [ctypes.c_int], ctypes.c_char_p)

‎llama_cpp/mtmd_cpp.py‎

Lines changed: 7 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -133,8 +133,15 @@ class mtmd_context_params(Structure):
133133
class mtmd_input_text(Structure):
134134
"""Text input passed to `mtmd_tokenize`."""
135135

136+
if TYPE_CHECKING:
137+
text: Optional[bytes]
138+
text_len: int
139+
add_special: bool
140+
parse_special: bool
141+
136142
_fields_ = [
137143
("text", c_char_p),
144+
("text_len", c_size_t),
138145
("add_special", c_bool),
139146
("parse_special", c_bool),
140147
]

‎tests/test_llama.py‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -103,8 +103,12 @@ def test_real_model(llama_cpp_model_path):
103103
assert os.path.exists(llama_cpp_model_path)
104104

105105
params = llama_cpp.llama_model_default_params()
106-
params.use_mmap = llama_cpp.llama_supports_mmap()
107-
params.use_mlock = llama_cpp.llama_supports_mlock()
106+
if llama_cpp.llama_supports_mlock():
107+
params.load_mode = llama_cpp.LLAMA_LOAD_MODE_MLOCK
108+
elif llama_cpp.llama_supports_mmap():
109+
params.load_mode = llama_cpp.LLAMA_LOAD_MODE_MMAP
110+
else:
111+
params.load_mode = llama_cpp.LLAMA_LOAD_MODE_NONE
108112
params.check_tensors = False
109113

110114
model = internals.LlamaModel(path_model=llama_cpp_model_path, params=params)

‎vendor/llama.cpp‎

Submodule llama.cpp updated 590 files

0 commit comments

Comments
 (0)