diff --git a/CHANGELOG.md b/CHANGELOG.md index be1130f95..064844d5d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] -- feat: update llama.cpp to ggml-org/llama.cpp@fb34fc262 +- feat: update llama.cpp to ggml-org/llama.cpp@0c1e57098 ## [0.3.35] diff --git a/llama_cpp/llama_cpp.py b/llama_cpp/llama_cpp.py index c7af8b9a7..94e5aede3 100644 --- a/llama_cpp/llama_cpp.py +++ b/llama_cpp/llama_cpp.py @@ -727,6 +727,14 @@ class llama_batch(ctypes.Structure): LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT_ETA = 11 +# enum llama_process_type { +# LLAMA_PROCESS_TYPE_ENCODE, +# LLAMA_PROCESS_TYPE_DECODE, +# }; +LLAMA_PROCESS_TYPE_ENCODE = 0 +LLAMA_PROCESS_TYPE_DECODE = 1 + + # struct llama_model_kv_override { # enum llama_model_kv_override_type tag; @@ -1344,7 +1352,6 @@ def llama_ftype_name(ftype: int, /) -> Optional[bytes]: # // Initialize the llama + ggml backend -# // If numa is true, use NUMA optimizations # // Call once at the start of the program # LLAMA_API void llama_backend_init(void); @ctypes_function( @@ -1387,7 +1394,8 @@ def llama_backend_free(): ... -# //optional: +# // Optional: enable numa optimizations +# // TODO: deprecate and make part of llama_backend_init() # LLAMA_API void llama_numa_init(enum ggml_numa_strategy numa); @ctypes_function( "llama_numa_init", @@ -3192,6 +3200,207 @@ def llama_decode(ctx: llama_context_p, batch: llama_batch, /) -> int: ... +# // +# // Extended batch API +# // + +# struct llama_batch_ext; +llama_batch_ext_p = NewType("llama_batch_ext_p", int) +llama_batch_ext_p_ctypes = ctypes.c_void_p + + +# struct llama_embd { +# const float * data; +# size_t n_rows; // number of embedding rows in data +# size_t n_embd; // size of one row +# }; +class llama_embd(ctypes.Structure): + if TYPE_CHECKING: + data: CtypesPointer[ctypes.c_float] + n_rows: int + n_embd: int + + _fields_ = [ + ("data", ctypes.POINTER(ctypes.c_float)), + ("n_rows", ctypes.c_size_t), + ("n_embd", ctypes.c_size_t), + ] + + +# LLAMA_API struct llama_batch_ext * llama_batch_ext_init (struct llama_context * ctx); +@ctypes_function( + "llama_batch_ext_init", [llama_context_p_ctypes], llama_batch_ext_p_ctypes +) +def llama_batch_ext_init(ctx: llama_context_p, /) -> Optional[llama_batch_ext_p]: ... + + +# LLAMA_API void llama_batch_ext_free (struct llama_batch_ext * batch); +@ctypes_function("llama_batch_ext_free", [llama_batch_ext_p_ctypes], None) +def llama_batch_ext_free(batch: llama_batch_ext_p, /): ... + + +# LLAMA_API void llama_batch_ext_clear(struct llama_batch_ext * batch); +@ctypes_function("llama_batch_ext_clear", [llama_batch_ext_p_ctypes], None) +def llama_batch_ext_clear(batch: llama_batch_ext_p, /): ... + + +# // Add an input token to the batch, with default values: +# // id = LLAMA_TOKEN_NULL +# // embd = None +# // pos = not set, the caller must set it with llama_batch_ext_set_pos() +# // Returns the batch index (>= 0) +# // On error: +# // -1: batch is full +# // -2: token is invalid (id == LLAMA_TOKEN_NULL or invalid embd) +# // -3: invalid sequence id +# LLAMA_API int32_t llama_batch_ext_add(struct llama_batch_ext * batch, llama_seq_id seq_id); +@ctypes_function( + "llama_batch_ext_add", [llama_batch_ext_p_ctypes, llama_seq_id], ctypes.c_int32 +) +def llama_batch_ext_add(batch: llama_batch_ext_p, seq_id: int, /) -> int: ... + + +# // Add an input token to the batch, with a specified token ID or token embedding +# LLAMA_API int32_t llama_batch_ext_add_token(struct llama_batch_ext * batch, llama_seq_id seq_id, llama_token id); +@ctypes_function( + "llama_batch_ext_add_token", + [llama_batch_ext_p_ctypes, llama_seq_id, llama_token], + ctypes.c_int32, +) +def llama_batch_ext_add_token( + batch: llama_batch_ext_p, seq_id: int, id: int, / +) -> int: ... + + +# LLAMA_API int32_t llama_batch_ext_add_embd(struct llama_batch_ext * batch, llama_seq_id seq_id, struct llama_embd embd); +@ctypes_function( + "llama_batch_ext_add_embd", + [llama_batch_ext_p_ctypes, llama_seq_id, llama_embd], + ctypes.c_int32, +) +def llama_batch_ext_add_embd( + batch: llama_batch_ext_p, seq_id: int, embd: llama_embd, / +) -> int: ... + + +# // Add the token at index idx in the batch to another sequence id. The position will stays the same. +# // Note: this should be called before other _set() functions +# LLAMA_API bool llama_batch_ext_add_seq( +# struct llama_batch_ext * batch, +# int32_t idx, +# llama_seq_id seq_id); +@ctypes_function( + "llama_batch_ext_add_seq", + [llama_batch_ext_p_ctypes, ctypes.c_int32, llama_seq_id], + ctypes.c_bool, +) +def llama_batch_ext_add_seq( + batch: llama_batch_ext_p, idx: int, seq_id: int, / +) -> bool: ... + + +# // Set the token embedding for the token at index idx in the batch +# // use it after llama_batch_ext_add_token() to have an entry with both a token id and an embedding +# LLAMA_API bool llama_batch_ext_set_embd_token( +# struct llama_batch_ext * batch, +# int32_t idx, +# struct llama_embd embd); +@ctypes_function( + "llama_batch_ext_set_embd_token", + [llama_batch_ext_p_ctypes, ctypes.c_int32, llama_embd], + ctypes.c_bool, +) +def llama_batch_ext_set_embd_token( + batch: llama_batch_ext_p, idx: int, embd: llama_embd, / +) -> bool: ... + + +# // Set the "state" embedding for the token at index idx in the batch +# // "state" here means extra hidden state carried over from a previous stage, e.g.: +# // - MTP: state from N layers of the target model +# // - Qwen3 VL (deepstack): state from N layers of the vision encoder +# LLAMA_API bool llama_batch_ext_set_embd_state( +# struct llama_batch_ext * batch, +# int32_t idx, +# struct llama_embd embd); +@ctypes_function( + "llama_batch_ext_set_embd_state", + [llama_batch_ext_p_ctypes, ctypes.c_int32, llama_embd], + ctypes.c_bool, +) +def llama_batch_ext_set_embd_state( + batch: llama_batch_ext_p, idx: int, embd: llama_embd, / +) -> bool: ... + + +# // Set if output embeddings should be available for the token at index idx in the batch +# // Note: for now, this is equivalent to setting the output logits +# LLAMA_API bool llama_batch_ext_set_output_embd( +# struct llama_batch_ext * batch, +# int32_t idx, +# bool value); +@ctypes_function( + "llama_batch_ext_set_output_embd", + [llama_batch_ext_p_ctypes, ctypes.c_int32, ctypes.c_bool], + ctypes.c_bool, +) +def llama_batch_ext_set_output_embd( + batch: llama_batch_ext_p, idx: int, value: bool, / +) -> bool: ... + + +# // Set output logits for the token at index idx in the batch +# // Note: for now, this is equivalent to setting the output embd +# LLAMA_API bool llama_batch_ext_set_output_logits( +# struct llama_batch_ext * batch, +# int32_t idx, +# bool value); +@ctypes_function( + "llama_batch_ext_set_output_logits", + [llama_batch_ext_p_ctypes, ctypes.c_int32, ctypes.c_bool], + ctypes.c_bool, +) +def llama_batch_ext_set_output_logits( + batch: llama_batch_ext_p, idx: int, value: bool, / +) -> bool: ... + + +# // Set custom position for the token at index idx in the batch +# // For M-RoPE models: +# // - Embedding tokens must have multiple positions per token +# // - Text token only requires one single position per token +# LLAMA_API bool llama_batch_ext_set_pos( +# struct llama_batch_ext * batch, +# int32_t idx, +# const llama_pos * pos); +@ctypes_function( + "llama_batch_ext_set_pos", + [llama_batch_ext_p_ctypes, ctypes.c_int32, ctypes.POINTER(llama_pos)], + ctypes.c_bool, +) +def llama_batch_ext_set_pos( + batch: llama_batch_ext_p, idx: int, pos: CtypesPointerOrRef[llama_pos], / +) -> bool: ... + + +# // TODO: implement get_embeddings() and get_logits() for llama_batch_ext + + +# // Return values are the same as llama_decode() +# LLAMA_API int32_t llama_process( +# struct llama_context * ctx, +# enum llama_process_type type, +# struct llama_batch_ext * batch); +@ctypes_function( + "llama_process", + [llama_context_p_ctypes, ctypes.c_int, llama_batch_ext_p_ctypes], + ctypes.c_int32, +) +def llama_process(ctx: llama_context_p, type: int, batch: llama_batch_ext_p, /) -> int: + """Return values are the same as llama_decode()""" + ... + + # // Set the number of threads used for decoding # // n_threads is the number of threads used for generation (single token) # // n_threads_batch is the number of threads used for prompt and batch processing (multiple tokens) @@ -3254,6 +3463,14 @@ def llama_set_causal_attn(ctx: llama_context_p, causal_attn: bool, /): ... +# // Returns whether the context is currently using causal attention +# LLAMA_API bool llama_get_causal_attn(const struct llama_context * ctx); +@ctypes_function("llama_get_causal_attn", [llama_context_p_ctypes], ctypes.c_bool) +def llama_get_causal_attn(ctx: llama_context_p, /) -> bool: + """Returns whether the context is currently using causal attention""" + ... + + # // Set whether the model is in warmup mode or not # // If true, all model tensors are activated during llama_decode() to load and cache their weights. # LLAMA_API void llama_set_warmup(struct llama_context * ctx, bool warmup); diff --git a/llama_cpp/mtmd_cpp.py b/llama_cpp/mtmd_cpp.py index a6fb187f8..646f1f719 100644 --- a/llama_cpp/mtmd_cpp.py +++ b/llama_cpp/mtmd_cpp.py @@ -335,9 +335,41 @@ class mtmd_gen_out(Structure): POINTER(c_char_p), ) + +# // one decoded sub-batch of embeddings, passed to mtmd_helper_post_decode_callback +# struct mtmd_helper_embd_batch { +# int32_t n_tokens; +# const float * embd; // [n_tokens, n_embd] +# int32_t n_embd; +# const llama_pos * pos; // [n_pos, n_tokens], section-major +# int32_t n_pos; // 4 for M-RoPE models, 1 otherwise +# llama_seq_id seq_id; +# }; +class mtmd_helper_embd_batch(Structure): + """One decoded sub-batch of embeddings, passed to mtmd_helper_post_decode_callback.""" + + if TYPE_CHECKING: + n_tokens: int + embd: "_Pointer[c_float]" + n_embd: int + pos: "_Pointer[llama_cpp.llama_pos]" + n_pos: int + seq_id: int + + _fields_ = [ + ("n_tokens", c_int32), + ("embd", POINTER(c_float)), + ("n_embd", c_int32), + ("pos", POINTER(llama_cpp.llama_pos)), + ("n_pos", c_int32), + ("seq_id", llama_cpp.llama_seq_id), + ] + + +# typedef int32_t (*mtmd_helper_post_decode_callback)(const struct mtmd_helper_embd_batch * batch, void * user_data); mtmd_helper_post_decode_callback = CFUNCTYPE( - c_int, - llama_cpp.llama_batch, + c_int32, + POINTER(mtmd_helper_embd_batch), c_void_p, ) diff --git a/vendor/llama.cpp b/vendor/llama.cpp index fb34fc262..0c1e57098 160000 --- a/vendor/llama.cpp +++ b/vendor/llama.cpp @@ -1 +1 @@ -Subproject commit fb34fc262c1b43f1832c7472429fb2247d650493 +Subproject commit 0c1e57098bba43ac29e6e3b677cdceebdd22334f