Skip to content

Commit ea3b56b

Browse files
authored
feat: update llama.cpp to fb34fc262 (#2369)
1 parent d736646 commit ea3b56b

3 files changed

Lines changed: 30 additions & 5 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -7,7 +7,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
77

88
## [Unreleased]
99

10-
- feat: update llama.cpp to ggml-org/llama.cpp@v0.4.0
10+
- feat: update llama.cpp to ggml-org/llama.cpp@fb34fc262
1111

1212
## [0.3.35]
1313

‎llama_cpp/llama_cpp.py‎

Lines changed: 28 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -213,6 +213,7 @@ def _warn_deprecated(symbol: str, hint: str) -> None:
213213
# LLAMA_VOCAB_TYPE_UGM = 4, // T5 tokenizer based on Unigram
214214
# LLAMA_VOCAB_TYPE_RWKV = 5, // RWKV tokenizer based on greedy tokenization
215215
# LLAMA_VOCAB_TYPE_PLAMO2 = 6, // PLaMo-2 tokenizer based on Aho-Corasick with dynamic programming
216+
# LLAMA_VOCAB_TYPE_TEST = 7, // Dummy tokenizer for testing: rolling hash of fixed-size chunks -> tokens, tokens -> hex
216217
# };
217218
LLAMA_VOCAB_TYPE_NONE = 0
218219
"""For models without vocab"""
@@ -228,6 +229,8 @@ def _warn_deprecated(symbol: str, hint: str) -> None:
228229
"""RWKV tokenizer based on greedy tokenization"""
229230
LLAMA_VOCAB_TYPE_PLAMO2 = 6
230231
"""PLaMo-2 tokenizer based on Aho-Corasick with dynamic programming"""
232+
LLAMA_VOCAB_TYPE_TEST = 7
233+
"""Dummy tokenizer for testing: rolling hash of fixed-size chunks -> tokens, tokens -> hex"""
231234

232235

233236
# NOTE: Deprecated and will be removed in the future. (already gone in llama.cpp)
@@ -1476,6 +1479,8 @@ def llama_model_load_from_splits(
14761479

14771480

14781481
# // Load a model from an open FILE pointer
1482+
# // The GGUF is read from the current position, so it can be embedded in a larger file
1483+
# // mmap needs the GGUF data section at a file offset to be aligned to the CPU tensor alignment (32 bytes)
14791484
# LLAMA_API struct llama_model * llama_model_load_from_file_ptr(
14801485
# FILE * file,
14811486
# struct llama_model_params params);
@@ -1487,7 +1492,11 @@ def llama_model_load_from_splits(
14871492
def llama_model_load_from_file_ptr(
14881493
file: ctypes.c_void_p, params: llama_model_params, /
14891494
) -> Optional[llama_model_p]:
1490-
"""Load a model from an open FILE pointer."""
1495+
"""Load a model from an open FILE pointer
1496+
1497+
The GGUF is read from the current position, so it can be embedded in a larger file
1498+
mmap needs the GGUF data section at a file offset to be aligned to the CPU tensor alignment (32 bytes)
1499+
"""
14911500
...
14921501

14931502

@@ -2115,6 +2124,22 @@ def llama_adapter_lora_init(
21152124
) -> Optional[llama_adapter_lora_p]: ...
21162125

21172126

2127+
# // Load a LoRA adapter from an open FILE pointer, reading from its current position
2128+
# LLAMA_API struct llama_adapter_lora * llama_adapter_lora_init_from_file_ptr(
2129+
# struct llama_model * model,
2130+
# FILE * file);
2131+
@ctypes_function(
2132+
"llama_adapter_lora_init_from_file_ptr",
2133+
[llama_model_p_ctypes, ctypes.c_void_p],
2134+
llama_adapter_lora_p_ctypes,
2135+
)
2136+
def llama_adapter_lora_init_from_file_ptr(
2137+
model: llama_model_p, file: ctypes.c_void_p, /
2138+
) -> Optional[llama_adapter_lora_p]:
2139+
"""Load a LoRA adapter from an open FILE pointer, reading from its current position"""
2140+
...
2141+
2142+
21182143
# // Get metadata value as a string by key name
21192144
# LLAMA_API int32_t llama_adapter_meta_val_str(const struct llama_adapter_lora * adapter, const char * key, char * buf, size_t buf_size);
21202145
@ctypes_function(
@@ -4519,11 +4544,11 @@ def llama_sampler_chain_get(
45194544
) -> llama_sampler_p: ...
45204545

45214546

4522-
# LLAMA_API int llama_sampler_chain_n (const struct llama_sampler * chain);
4547+
# LLAMA_API int32_t llama_sampler_chain_n (const struct llama_sampler * chain);
45234548
@ctypes_function(
45244549
"llama_sampler_chain_n",
45254550
[llama_sampler_p_ctypes],
4526-
ctypes.c_int,
4551+
ctypes.c_int32,
45274552
)
45284553
def llama_sampler_chain_n(chain: llama_sampler_p, /) -> int: ...
45294554

‎vendor/llama.cpp‎

Submodule llama.cpp updated 622 files

0 commit comments

Comments
 (0)