Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

- feat: update llama.cpp to ggml-org/llama.cpp@fb34fc262
- feat: update llama.cpp to ggml-org/llama.cpp@0c1e57098

## [0.3.35]

Expand Down
221 changes: 219 additions & 2 deletions llama_cpp/llama_cpp.py
Original file line number Diff line number Diff line change
Expand Up @@ -727,6 +727,14 @@ class llama_batch(ctypes.Structure):
LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT_ETA = 11


# enum llama_process_type {
# LLAMA_PROCESS_TYPE_ENCODE,
# LLAMA_PROCESS_TYPE_DECODE,
# };
LLAMA_PROCESS_TYPE_ENCODE = 0
LLAMA_PROCESS_TYPE_DECODE = 1


# struct llama_model_kv_override {
# enum llama_model_kv_override_type tag;

Expand Down Expand Up @@ -1344,7 +1352,6 @@ def llama_ftype_name(ftype: int, /) -> Optional[bytes]:


# // Initialize the llama + ggml backend
# // If numa is true, use NUMA optimizations
# // Call once at the start of the program
# LLAMA_API void llama_backend_init(void);
@ctypes_function(
Expand Down Expand Up @@ -1387,7 +1394,8 @@ def llama_backend_free():
...


# //optional:
# // Optional: enable numa optimizations
# // TODO: deprecate and make part of llama_backend_init()
# LLAMA_API void llama_numa_init(enum ggml_numa_strategy numa);
@ctypes_function(
"llama_numa_init",
Expand Down Expand Up @@ -3192,6 +3200,207 @@ def llama_decode(ctx: llama_context_p, batch: llama_batch, /) -> int:
...


# //
# // Extended batch API
# //

# struct llama_batch_ext;
llama_batch_ext_p = NewType("llama_batch_ext_p", int)
llama_batch_ext_p_ctypes = ctypes.c_void_p


# struct llama_embd {
# const float * data;
# size_t n_rows; // number of embedding rows in data
# size_t n_embd; // size of one row
# };
class llama_embd(ctypes.Structure):
if TYPE_CHECKING:
data: CtypesPointer[ctypes.c_float]
n_rows: int
n_embd: int

_fields_ = [
("data", ctypes.POINTER(ctypes.c_float)),
("n_rows", ctypes.c_size_t),
("n_embd", ctypes.c_size_t),
]


# LLAMA_API struct llama_batch_ext * llama_batch_ext_init (struct llama_context * ctx);
@ctypes_function(
"llama_batch_ext_init", [llama_context_p_ctypes], llama_batch_ext_p_ctypes
)
def llama_batch_ext_init(ctx: llama_context_p, /) -> Optional[llama_batch_ext_p]: ...


# LLAMA_API void llama_batch_ext_free (struct llama_batch_ext * batch);
@ctypes_function("llama_batch_ext_free", [llama_batch_ext_p_ctypes], None)
def llama_batch_ext_free(batch: llama_batch_ext_p, /): ...


# LLAMA_API void llama_batch_ext_clear(struct llama_batch_ext * batch);
@ctypes_function("llama_batch_ext_clear", [llama_batch_ext_p_ctypes], None)
def llama_batch_ext_clear(batch: llama_batch_ext_p, /): ...


# // Add an input token to the batch, with default values:
# // id = LLAMA_TOKEN_NULL
# // embd = None
# // pos = not set, the caller must set it with llama_batch_ext_set_pos()
# // Returns the batch index (>= 0)
# // On error:
# // -1: batch is full
# // -2: token is invalid (id == LLAMA_TOKEN_NULL or invalid embd)
# // -3: invalid sequence id
# LLAMA_API int32_t llama_batch_ext_add(struct llama_batch_ext * batch, llama_seq_id seq_id);
@ctypes_function(
"llama_batch_ext_add", [llama_batch_ext_p_ctypes, llama_seq_id], ctypes.c_int32
)
def llama_batch_ext_add(batch: llama_batch_ext_p, seq_id: int, /) -> int: ...


# // Add an input token to the batch, with a specified token ID or token embedding
# LLAMA_API int32_t llama_batch_ext_add_token(struct llama_batch_ext * batch, llama_seq_id seq_id, llama_token id);
@ctypes_function(
"llama_batch_ext_add_token",
[llama_batch_ext_p_ctypes, llama_seq_id, llama_token],
ctypes.c_int32,
)
def llama_batch_ext_add_token(
batch: llama_batch_ext_p, seq_id: int, id: int, /
) -> int: ...


# LLAMA_API int32_t llama_batch_ext_add_embd(struct llama_batch_ext * batch, llama_seq_id seq_id, struct llama_embd embd);
@ctypes_function(
"llama_batch_ext_add_embd",
[llama_batch_ext_p_ctypes, llama_seq_id, llama_embd],
ctypes.c_int32,
)
def llama_batch_ext_add_embd(
batch: llama_batch_ext_p, seq_id: int, embd: llama_embd, /
) -> int: ...


# // Add the token at index idx in the batch to another sequence id. The position will stays the same.
# // Note: this should be called before other _set() functions
# LLAMA_API bool llama_batch_ext_add_seq(
# struct llama_batch_ext * batch,
# int32_t idx,
# llama_seq_id seq_id);
@ctypes_function(
"llama_batch_ext_add_seq",
[llama_batch_ext_p_ctypes, ctypes.c_int32, llama_seq_id],
ctypes.c_bool,
)
def llama_batch_ext_add_seq(
batch: llama_batch_ext_p, idx: int, seq_id: int, /
) -> bool: ...


# // Set the token embedding for the token at index idx in the batch
# // use it after llama_batch_ext_add_token() to have an entry with both a token id and an embedding
# LLAMA_API bool llama_batch_ext_set_embd_token(
# struct llama_batch_ext * batch,
# int32_t idx,
# struct llama_embd embd);
@ctypes_function(
"llama_batch_ext_set_embd_token",
[llama_batch_ext_p_ctypes, ctypes.c_int32, llama_embd],
ctypes.c_bool,
)
def llama_batch_ext_set_embd_token(
batch: llama_batch_ext_p, idx: int, embd: llama_embd, /
) -> bool: ...


# // Set the "state" embedding for the token at index idx in the batch
# // "state" here means extra hidden state carried over from a previous stage, e.g.:
# // - MTP: state from N layers of the target model
# // - Qwen3 VL (deepstack): state from N layers of the vision encoder
# LLAMA_API bool llama_batch_ext_set_embd_state(
# struct llama_batch_ext * batch,
# int32_t idx,
# struct llama_embd embd);
@ctypes_function(
"llama_batch_ext_set_embd_state",
[llama_batch_ext_p_ctypes, ctypes.c_int32, llama_embd],
ctypes.c_bool,
)
def llama_batch_ext_set_embd_state(
batch: llama_batch_ext_p, idx: int, embd: llama_embd, /
) -> bool: ...


# // Set if output embeddings should be available for the token at index idx in the batch
# // Note: for now, this is equivalent to setting the output logits
# LLAMA_API bool llama_batch_ext_set_output_embd(
# struct llama_batch_ext * batch,
# int32_t idx,
# bool value);
@ctypes_function(
"llama_batch_ext_set_output_embd",
[llama_batch_ext_p_ctypes, ctypes.c_int32, ctypes.c_bool],
ctypes.c_bool,
)
def llama_batch_ext_set_output_embd(
batch: llama_batch_ext_p, idx: int, value: bool, /
) -> bool: ...


# // Set output logits for the token at index idx in the batch
# // Note: for now, this is equivalent to setting the output embd
# LLAMA_API bool llama_batch_ext_set_output_logits(
# struct llama_batch_ext * batch,
# int32_t idx,
# bool value);
@ctypes_function(
"llama_batch_ext_set_output_logits",
[llama_batch_ext_p_ctypes, ctypes.c_int32, ctypes.c_bool],
ctypes.c_bool,
)
def llama_batch_ext_set_output_logits(
batch: llama_batch_ext_p, idx: int, value: bool, /
) -> bool: ...


# // Set custom position for the token at index idx in the batch
# // For M-RoPE models:
# // - Embedding tokens must have multiple positions per token
# // - Text token only requires one single position per token
# LLAMA_API bool llama_batch_ext_set_pos(
# struct llama_batch_ext * batch,
# int32_t idx,
# const llama_pos * pos);
@ctypes_function(
"llama_batch_ext_set_pos",
[llama_batch_ext_p_ctypes, ctypes.c_int32, ctypes.POINTER(llama_pos)],
ctypes.c_bool,
)
def llama_batch_ext_set_pos(
batch: llama_batch_ext_p, idx: int, pos: CtypesPointerOrRef[llama_pos], /
) -> bool: ...


# // TODO: implement get_embeddings() and get_logits() for llama_batch_ext


# // Return values are the same as llama_decode()
# LLAMA_API int32_t llama_process(
# struct llama_context * ctx,
# enum llama_process_type type,
# struct llama_batch_ext * batch);
@ctypes_function(
"llama_process",
[llama_context_p_ctypes, ctypes.c_int, llama_batch_ext_p_ctypes],
ctypes.c_int32,
)
def llama_process(ctx: llama_context_p, type: int, batch: llama_batch_ext_p, /) -> int:
"""Return values are the same as llama_decode()"""
...


# // Set the number of threads used for decoding
# // n_threads is the number of threads used for generation (single token)
# // n_threads_batch is the number of threads used for prompt and batch processing (multiple tokens)
Expand Down Expand Up @@ -3254,6 +3463,14 @@ def llama_set_causal_attn(ctx: llama_context_p, causal_attn: bool, /):
...


# // Returns whether the context is currently using causal attention
# LLAMA_API bool llama_get_causal_attn(const struct llama_context * ctx);
@ctypes_function("llama_get_causal_attn", [llama_context_p_ctypes], ctypes.c_bool)
def llama_get_causal_attn(ctx: llama_context_p, /) -> bool:
"""Returns whether the context is currently using causal attention"""
...


# // Set whether the model is in warmup mode or not
# // If true, all model tensors are activated during llama_decode() to load and cache their weights.
# LLAMA_API void llama_set_warmup(struct llama_context * ctx, bool warmup);
Expand Down
36 changes: 34 additions & 2 deletions llama_cpp/mtmd_cpp.py
Original file line number Diff line number Diff line change
Expand Up @@ -335,9 +335,41 @@ class mtmd_gen_out(Structure):
POINTER(c_char_p),
)


# // one decoded sub-batch of embeddings, passed to mtmd_helper_post_decode_callback
# struct mtmd_helper_embd_batch {
# int32_t n_tokens;
# const float * embd; // [n_tokens, n_embd]
# int32_t n_embd;
# const llama_pos * pos; // [n_pos, n_tokens], section-major
# int32_t n_pos; // 4 for M-RoPE models, 1 otherwise
# llama_seq_id seq_id;
# };
class mtmd_helper_embd_batch(Structure):
"""One decoded sub-batch of embeddings, passed to mtmd_helper_post_decode_callback."""

if TYPE_CHECKING:
n_tokens: int
embd: "_Pointer[c_float]"
n_embd: int
pos: "_Pointer[llama_cpp.llama_pos]"
n_pos: int
seq_id: int

_fields_ = [
("n_tokens", c_int32),
("embd", POINTER(c_float)),
("n_embd", c_int32),
("pos", POINTER(llama_cpp.llama_pos)),
("n_pos", c_int32),
("seq_id", llama_cpp.llama_seq_id),
]


# typedef int32_t (*mtmd_helper_post_decode_callback)(const struct mtmd_helper_embd_batch * batch, void * user_data);
mtmd_helper_post_decode_callback = CFUNCTYPE(
c_int,
llama_cpp.llama_batch,
c_int32,
POINTER(mtmd_helper_embd_batch),
c_void_p,
)

Expand Down
2 changes: 1 addition & 1 deletion vendor/llama.cpp
Submodule llama.cpp updated 445 files
Loading