Expand description
Low-level FFI bindings to ik_llama.cpp (ikawrakow’s SOTA-quant fork of llama.cpp).
The bindings are generated by bindgen in build.rs against the vendored
ik_llama.cpp/include/llama.h + ggml/include/gguf.h and included below.
Structs§
- _IO_
FILE - _IO_
codecvt - _IO_
marker - _IO_
wide_ data - ggml_
backend - ggml_
backend_ buffer - ggml_
backend_ buffer_ type - ggml_
backend_ event - ggml_
backend_ graph_ copy - ggml_
backend_ sched - ggml_
bf16_ t - ggml_
cgraph - ggml_
context - ggml_
cplan - ggml_
gallocr - ggml_
hash_ set - ggml_
init_ params - ggml_
object - ggml_
opt_ context - ggml_
opt_ context__ bindgen_ ty_ 1 - ggml_
opt_ context__ bindgen_ ty_ 2 - ggml_
opt_ params - ggml_
opt_ params__ bindgen_ ty_ 1 - ggml_
opt_ params__ bindgen_ ty_ 2 - ggml_
scratch - ggml_
split_ tensor_ t - ggml_
tallocr - ggml_
tensor - ggml_
type_ traits_ t - gguf_
context - gguf_
init_ params - ik_
llama_ rs_ mtp - llama_
batch - llama_
chat_ message - llama_
context - llama_
context_ params - llama_
grammar - llama_
kv_ cache_ view - llama_
kv_ cache_ view_ cell - llama_
lora_ adapter - llama_
model - llama_
model_ kv_ override - llama_
model_ params - llama_
model_ quantize_ params - llama_
model_ tensor_ buft_ override - llama_
sampler - llama_
sampler_ adaptive_ p - llama_
sampler_ dry - llama_
sampler_ grammar - llama_
sampler_ i - llama_
timings - llama_
token_ data - llama_
token_ data_ array - llama_
vocab - quantize_
user_ data
Constants§
- GGML_
BACKEND_ BUFFER_ USAGE_ ANY - GGML_
BACKEND_ BUFFER_ USAGE_ COMPUTE - GGML_
BACKEND_ BUFFER_ USAGE_ WEIGHTS - GGML_
BACKEND_ TYPE_ CPU - GGML_
BACKEND_ TYPE_ GPU - GGML_
BACKEND_ TYPE_ GPU_ SPLIT - GGML_
CGRAPH_ EVAL_ ORDER_ COUNT - GGML_
CGRAPH_ EVAL_ ORDER_ LEFT_ TO_ RIGHT - GGML_
CGRAPH_ EVAL_ ORDER_ RIGHT_ TO_ LEFT - GGML_
FTYPE_ ALL_ F32 - GGML_
FTYPE_ MOSTLY_ BF16 - GGML_
FTYPE_ MOSTLY_ BF16_ R16 - GGML_
FTYPE_ MOSTLY_ F16 - GGML_
FTYPE_ MOSTLY_ IQ1_ BN - GGML_
FTYPE_ MOSTLY_ IQ1_ KT - GGML_
FTYPE_ MOSTLY_ IQ1_ M - GGML_
FTYPE_ MOSTLY_ IQ1_ M_ R4 - GGML_
FTYPE_ MOSTLY_ IQ1_ S - GGML_
FTYPE_ MOSTLY_ IQ1_ S_ R4 - GGML_
FTYPE_ MOSTLY_ IQ2_ BN - GGML_
FTYPE_ MOSTLY_ IQ2_ BN_ R4 - GGML_
FTYPE_ MOSTLY_ IQ2_ K - GGML_
FTYPE_ MOSTLY_ IQ2_ KL - GGML_
FTYPE_ MOSTLY_ IQ2_ KS - GGML_
FTYPE_ MOSTLY_ IQ2_ KT - GGML_
FTYPE_ MOSTLY_ IQ2_ K_ R4 - GGML_
FTYPE_ MOSTLY_ IQ2_ S - GGML_
FTYPE_ MOSTLY_ IQ2_ S_ R4 - GGML_
FTYPE_ MOSTLY_ IQ2_ XS - GGML_
FTYPE_ MOSTLY_ IQ2_ XS_ R4 - GGML_
FTYPE_ MOSTLY_ IQ2_ XXS - GGML_
FTYPE_ MOSTLY_ IQ2_ XXS_ R4 - GGML_
FTYPE_ MOSTLY_ IQ3_ K - GGML_
FTYPE_ MOSTLY_ IQ3_ KS - GGML_
FTYPE_ MOSTLY_ IQ3_ KT - GGML_
FTYPE_ MOSTLY_ IQ3_ K_ R4 - GGML_
FTYPE_ MOSTLY_ IQ3_ S - GGML_
FTYPE_ MOSTLY_ IQ3_ S_ R4 - GGML_
FTYPE_ MOSTLY_ IQ3_ XXS - GGML_
FTYPE_ MOSTLY_ IQ3_ XXS_ R4 - GGML_
FTYPE_ MOSTLY_ IQ4_ K - GGML_
FTYPE_ MOSTLY_ IQ4_ KS - GGML_
FTYPE_ MOSTLY_ IQ4_ KSS - GGML_
FTYPE_ MOSTLY_ IQ4_ KS_ R4 - GGML_
FTYPE_ MOSTLY_ IQ4_ KT - GGML_
FTYPE_ MOSTLY_ IQ4_ K_ R4 - GGML_
FTYPE_ MOSTLY_ IQ4_ NL - GGML_
FTYPE_ MOSTLY_ IQ4_ NL_ R4 - GGML_
FTYPE_ MOSTLY_ IQ4_ XS - GGML_
FTYPE_ MOSTLY_ IQ4_ XS_ R8 - GGML_
FTYPE_ MOSTLY_ IQ5_ K - GGML_
FTYPE_ MOSTLY_ IQ5_ KS - GGML_
FTYPE_ MOSTLY_ IQ5_ KS_ R4 - GGML_
FTYPE_ MOSTLY_ IQ5_ K_ R4 - GGML_
FTYPE_ MOSTLY_ IQ6_ K - GGML_
FTYPE_ MOSTLY_ MXFP4 - GGML_
FTYPE_ MOSTLY_ Q1_ 0_ 128 - GGML_
FTYPE_ MOSTLY_ Q2_ K - GGML_
FTYPE_ MOSTLY_ Q2_ K_ R4 - GGML_
FTYPE_ MOSTLY_ Q3_ K - GGML_
FTYPE_ MOSTLY_ Q3_ K_ R4 - GGML_
FTYPE_ MOSTLY_ Q4_ 0 - GGML_
FTYPE_ MOSTLY_ Q4_ 0_ 4_ 4 - GGML_
FTYPE_ MOSTLY_ Q4_ 0_ 4_ 8 - GGML_
FTYPE_ MOSTLY_ Q4_ 0_ 8_ 8 - GGML_
FTYPE_ MOSTLY_ Q4_ 0_ R8 - GGML_
FTYPE_ MOSTLY_ Q4_ 1 - GGML_
FTYPE_ MOSTLY_ Q4_ 1_ SOME_ F16 - GGML_
FTYPE_ MOSTLY_ Q4_ K - GGML_
FTYPE_ MOSTLY_ Q4_ K_ R4 - GGML_
FTYPE_ MOSTLY_ Q5_ 0 - GGML_
FTYPE_ MOSTLY_ Q5_ 0_ R4 - GGML_
FTYPE_ MOSTLY_ Q5_ 1 - GGML_
FTYPE_ MOSTLY_ Q5_ K - GGML_
FTYPE_ MOSTLY_ Q5_ K_ R4 - GGML_
FTYPE_ MOSTLY_ Q6_ 0 - GGML_
FTYPE_ MOSTLY_ Q6_ 0_ R4 - GGML_
FTYPE_ MOSTLY_ Q6_ K - GGML_
FTYPE_ MOSTLY_ Q6_ K_ R4 - GGML_
FTYPE_ MOSTLY_ Q8_ 0 - GGML_
FTYPE_ MOSTLY_ Q8_ 0_ R8 - GGML_
FTYPE_ MOSTLY_ Q8_ KV - GGML_
FTYPE_ MOSTLY_ Q8_ KV_ R8 - GGML_
FTYPE_ MOSTLY_ Q8_ K_ R8 - GGML_
FTYPE_ MOSTLY_ Q8_ K_ R16 - GGML_
FTYPE_ UNKNOWN - GGML_
GLU_ OP_ COUNT - GGML_
GLU_ OP_ GEGLU - GGML_
GLU_ OP_ GEGLU_ ERF - GGML_
GLU_ OP_ GEGLU_ QUICK - GGML_
GLU_ OP_ REGLU - GGML_
GLU_ OP_ SWIGLU - GGML_
GLU_ OP_ SWIGLU_ OAI - GGML_
LINESEARCH_ BACKTRACKING_ ARMIJO - GGML_
LINESEARCH_ BACKTRACKING_ STRONG_ WOLFE - GGML_
LINESEARCH_ BACKTRACKING_ WOLFE - GGML_
LINESEARCH_ DEFAULT - GGML_
LINESEARCH_ FAIL - GGML_
LINESEARCH_ INVALID_ PARAMETERS - GGML_
LINESEARCH_ MAXIMUM_ ITERATIONS - GGML_
LINESEARCH_ MAXIMUM_ STEP - GGML_
LINESEARCH_ MINIMUM_ STEP - GGML_
LOG_ LEVEL_ CONT - GGML_
LOG_ LEVEL_ DEBUG - GGML_
LOG_ LEVEL_ ERROR - GGML_
LOG_ LEVEL_ INFO - GGML_
LOG_ LEVEL_ NONE - GGML_
LOG_ LEVEL_ WARN - GGML_
NUMA_ STRATEGY_ COUNT - GGML_
NUMA_ STRATEGY_ DISABLED - GGML_
NUMA_ STRATEGY_ DISTRIBUTE - GGML_
NUMA_ STRATEGY_ ISOLATE - GGML_
NUMA_ STRATEGY_ MIRROR - GGML_
NUMA_ STRATEGY_ NUMACTL - GGML_
OBJECT_ TYPE_ GRAPH - GGML_
OBJECT_ TYPE_ TENSOR - GGML_
OBJECT_ TYPE_ WORK_ BUFFER - GGML_
OPT_ RESULT_ CANCEL - GGML_
OPT_ RESULT_ DID_ NOT_ CONVERGE - GGML_
OPT_ RESULT_ FAIL - GGML_
OPT_ RESULT_ INVALID_ WOLFE - GGML_
OPT_ RESULT_ NO_ CONTEXT - GGML_
OPT_ RESULT_ OK - GGML_
OPT_ TYPE_ ADAM - GGML_
OPT_ TYPE_ LBFGS - GGML_
OP_ ACC - GGML_
OP_ ADD - GGML_
OP_ ADD1 - GGML_
OP_ ADD_ ID - GGML_
OP_ ADD_ REL_ POS - GGML_
OP_ ARANGE - GGML_
OP_ ARGMAX - GGML_
OP_ ARGSORT - GGML_
OP_ ARGSORT_ THRESH - GGML_
OP_ BLEND - GGML_
OP_ CLAMP - GGML_
OP_ CONCAT - GGML_
OP_ CONT - GGML_
OP_ CONV_ 2D - GGML_
OP_ CONV_ 2D_ DW - GGML_
OP_ CONV_ TRANSPOSE_ 1D - GGML_
OP_ CONV_ TRANSPOSE_ 2D - GGML_
OP_ COUNT - GGML_
OP_ CPY - GGML_
OP_ CROSS_ ENTROPY_ LOSS - GGML_
OP_ CROSS_ ENTROPY_ LOSS_ BACK - GGML_
OP_ CUMSUM - GGML_
OP_ DELTA_ NET - GGML_
OP_ DIAG - GGML_
OP_ DIAG_ MASK_ INF - GGML_
OP_ DIAG_ MASK_ ZERO - GGML_
OP_ DIV - GGML_
OP_ DUP - GGML_
OP_ FAKE_ CPY - GGML_
OP_ FILL - GGML_
OP_ FLASH_ ATTN_ BACK - GGML_
OP_ FLASH_ ATTN_ EXT - GGML_
OP_ FUSED_ MUL_ UNARY - GGML_
OP_ FUSED_ NORM - GGML_
OP_ FUSED_ RMS_ NORM - GGML_
OP_ FUSED_ RMS_ RMS_ ADD - GGML_
OP_ FUSED_ UP_ GATE - GGML_
OP_ GET_ REL_ POS - GGML_
OP_ GET_ ROWS - GGML_
OP_ GET_ ROWS_ BACK - GGML_
OP_ GLU - GGML_
OP_ GROUPED_ TOPK - GGML_
OP_ GROUP_ NORM - GGML_
OP_ HADAMARD - GGML_
OP_ IM2COL - GGML_
OP_ INDEXER_ TOPK - GGML_
OP_ L2_ NORM - GGML_
OP_ LEAKY_ RELU - GGML_
OP_ LOG - GGML_
OP_ MAP_ BINARY - GGML_
OP_ MAP_ CUSTO M1 - GGML_
OP_ MAP_ CUSTO M2 - GGML_
OP_ MAP_ CUSTO M3 - GGML_
OP_ MAP_ CUSTO M1_ F32 - GGML_
OP_ MAP_ CUSTO M2_ F32 - GGML_
OP_ MAP_ CUSTO M3_ F32 - GGML_
OP_ MAP_ UNARY - GGML_
OP_ MASK_ TOPK - GGML_
OP_ MEAN - GGML_
OP_ MOE_ FUSED_ UP_ GATE - GGML_
OP_ MUL - GGML_
OP_ MULTI_ ADD - GGML_
OP_ MUL_ MAT - GGML_
OP_ MUL_ MAT_ ID - GGML_
OP_ MUL_ MULTI_ ADD - GGML_
OP_ NONE - GGML_
OP_ NORM - GGML_
OP_ OUT_ PROD - GGML_
OP_ PAD - GGML_
OP_ PERMUTE - GGML_
OP_ POOL_ 1D - GGML_
OP_ POOL_ 2D - GGML_
OP_ POOL_ AVG - GGML_
OP_ POOL_ COUNT - GGML_
OP_ POOL_ MAX - GGML_
OP_ REDUCE - GGML_
OP_ REPEAT - GGML_
OP_ REPEAT_ BACK - GGML_
OP_ RESHAPE - GGML_
OP_ RMS_ NORM - GGML_
OP_ RMS_ NORM_ BACK - GGML_
OP_ ROPE - GGML_
OP_ ROPE_ BACK - GGML_
OP_ ROPE_ CACHE - GGML_
OP_ ROPE_ FAST - GGML_
OP_ SCALE - GGML_
OP_ SET - GGML_
OP_ SET_ ROWS - GGML_
OP_ SILU_ BACK - GGML_
OP_ SINKHORN - GGML_
OP_ SOFTCAP - GGML_
OP_ SOFT_ CAP_ MAX - GGML_
OP_ SOFT_ MAX - GGML_
OP_ SOFT_ MAX_ BACK - GGML_
OP_ SOLVE_ TRI - GGML_
OP_ SQR - GGML_
OP_ SQRT - GGML_
OP_ SSM_ CONV - GGML_
OP_ SSM_ SCAN - GGML_
OP_ SUB - GGML_
OP_ SUM - GGML_
OP_ SUM_ ROWS - GGML_
OP_ TIMESTEP_ EMBEDDING - GGML_
OP_ TRANSPOSE - GGML_
OP_ TRI - GGML_
OP_ UNARY - GGML_
OP_ UPSCALE - GGML_
OP_ VIEW - GGML_
OP_ WIN_ PART - GGML_
OP_ WIN_ UNPART - GGML_
PREC_ DEFAULT - GGML_
PREC_ F32 - GGML_
SCALE_ FLAG_ ALIGN_ CORNERS - GGML_
SCALE_ MODE_ BICUBIC - GGML_
SCALE_ MODE_ BILINEAR - GGML_
SCALE_ MODE_ COUNT - GGML_
SCALE_ MODE_ NEAREST - GGML_
SORT_ ORDER_ ASC - GGML_
SORT_ ORDER_ DESC - GGML_
STATUS_ ABORTED - GGML_
STATUS_ ALLOC_ FAILED - GGML_
STATUS_ FAILED - GGML_
STATUS_ SUCCESS - GGML_
TENSOR_ FLAG_ INPUT - GGML_
TENSOR_ FLAG_ LOSS - GGML_
TENSOR_ FLAG_ OUTPUT - GGML_
TENSOR_ FLAG_ PARAM - GGML_
TRI_ TYPE_ LOWER - GGML_
TRI_ TYPE_ LOWER_ DIAG - GGML_
TRI_ TYPE_ UPPER - GGML_
TRI_ TYPE_ UPPER_ DIAG - GGML_
TYPE_ BF16 - GGML_
TYPE_ BF16_ R16 - GGML_
TYPE_ COUNT - GGML_
TYPE_ F16 - GGML_
TYPE_ F32 - GGML_
TYPE_ F64 - GGML_
TYPE_ I8 - GGML_
TYPE_ I2_ S - GGML_
TYPE_ I16 - GGML_
TYPE_ I32 - GGML_
TYPE_ I64 - GGML_
TYPE_ IQ1_ BN - GGML_
TYPE_ IQ1_ KT - GGML_
TYPE_ IQ1_ M - GGML_
TYPE_ IQ1_ M_ R4 - GGML_
TYPE_ IQ1_ S - GGML_
TYPE_ IQ1_ S_ R4 - GGML_
TYPE_ IQ2_ BN - GGML_
TYPE_ IQ2_ BN_ R4 - GGML_
TYPE_ IQ2_ K - GGML_
TYPE_ IQ2_ KL - GGML_
TYPE_ IQ2_ KS - GGML_
TYPE_ IQ2_ KT - GGML_
TYPE_ IQ2_ K_ R4 - GGML_
TYPE_ IQ2_ S - GGML_
TYPE_ IQ2_ S_ R4 - GGML_
TYPE_ IQ2_ XS - GGML_
TYPE_ IQ2_ XS_ R4 - GGML_
TYPE_ IQ2_ XXS - GGML_
TYPE_ IQ2_ XXS_ R4 - GGML_
TYPE_ IQ3_ K - GGML_
TYPE_ IQ3_ KS - GGML_
TYPE_ IQ3_ KT - GGML_
TYPE_ IQ3_ K_ R4 - GGML_
TYPE_ IQ3_ S - GGML_
TYPE_ IQ3_ S_ R4 - GGML_
TYPE_ IQ3_ XXS - GGML_
TYPE_ IQ3_ XXS_ R4 - GGML_
TYPE_ IQ4_ K - GGML_
TYPE_ IQ4_ KS - GGML_
TYPE_ IQ4_ KSS - GGML_
TYPE_ IQ4_ KS_ R4 - GGML_
TYPE_ IQ4_ KT - GGML_
TYPE_ IQ4_ K_ R4 - GGML_
TYPE_ IQ4_ NL - GGML_
TYPE_ IQ4_ NL_ R4 - GGML_
TYPE_ IQ4_ XS - GGML_
TYPE_ IQ4_ XS_ R8 - GGML_
TYPE_ IQ5_ K - GGML_
TYPE_ IQ5_ KS - GGML_
TYPE_ IQ5_ KS_ R4 - GGML_
TYPE_ IQ5_ K_ R4 - GGML_
TYPE_ IQ6_ K - GGML_
TYPE_ MXFP4 - GGML_
TYPE_ Q1_ 0_ G128 - GGML_
TYPE_ Q2_ K - GGML_
TYPE_ Q2_ K_ R4 - GGML_
TYPE_ Q3_ K - GGML_
TYPE_ Q3_ K_ R4 - GGML_
TYPE_ Q4_ 0 - GGML_
TYPE_ Q4_ 0_ 4_ 4 - GGML_
TYPE_ Q4_ 0_ 4_ 8 - GGML_
TYPE_ Q4_ 0_ 8_ 8 - GGML_
TYPE_ Q4_ 0_ R8 - GGML_
TYPE_ Q4_ 1 - GGML_
TYPE_ Q4_ K - GGML_
TYPE_ Q4_ K_ R4 - GGML_
TYPE_ Q5_ 0 - GGML_
TYPE_ Q5_ 0_ R4 - GGML_
TYPE_ Q5_ 1 - GGML_
TYPE_ Q5_ K - GGML_
TYPE_ Q5_ K_ R4 - GGML_
TYPE_ Q6_ 0 - GGML_
TYPE_ Q6_ 0_ R4 - GGML_
TYPE_ Q6_ K - GGML_
TYPE_ Q6_ K_ R4 - GGML_
TYPE_ Q8_ 0 - GGML_
TYPE_ Q8_ 0_ R8 - GGML_
TYPE_ Q8_ 0_ X4 - GGML_
TYPE_ Q8_ 1 - GGML_
TYPE_ Q8_ 1_ X4 - GGML_
TYPE_ Q8_ 2_ X4 - GGML_
TYPE_ Q8_ K - GGML_
TYPE_ Q8_ K16 - GGML_
TYPE_ Q8_ K32 - GGML_
TYPE_ Q8_ K64 - GGML_
TYPE_ Q8_ K128 - GGML_
TYPE_ Q8_ KR8 - GGML_
TYPE_ Q8_ KV - GGML_
TYPE_ Q8_ KV_ R8 - GGML_
TYPE_ Q8_ K_ R8 - GGML_
TYPE_ Q8_ K_ R16 - GGML_
UNARY_ OP_ ABS - GGML_
UNARY_ OP_ COUNT - GGML_
UNARY_ OP_ ELU - GGML_
UNARY_ OP_ EXP - GGML_
UNARY_ OP_ GELU - GGML_
UNARY_ OP_ GELU_ ERF - GGML_
UNARY_ OP_ GELU_ QUICK - GGML_
UNARY_ OP_ HARDSIGMOID - GGML_
UNARY_ OP_ HARDSWISH - GGML_
UNARY_ OP_ NEG - GGML_
UNARY_ OP_ RELU - GGML_
UNARY_ OP_ SGN - GGML_
UNARY_ OP_ SIGMOID - GGML_
UNARY_ OP_ SILU - GGML_
UNARY_ OP_ SOFTPLUS - GGML_
UNARY_ OP_ STEP - GGML_
UNARY_ OP_ SWIGLU - GGML_
UNARY_ OP_ SWIGLU_ OAI - GGML_
UNARY_ OP_ TANH - GGUF_
TYPE_ ARRAY - GGUF_
TYPE_ BOOL - GGUF_
TYPE_ COUNT - GGUF_
TYPE_ FLOA T32 - GGUF_
TYPE_ FLOA T64 - GGUF_
TYPE_ INT8 - GGUF_
TYPE_ INT16 - GGUF_
TYPE_ INT32 - GGUF_
TYPE_ INT64 - GGUF_
TYPE_ STRING - GGUF_
TYPE_ UINT8 - GGUF_
TYPE_ UINT16 - GGUF_
TYPE_ UINT32 - GGUF_
TYPE_ UINT64 - LLAMA_
ATTENTION_ TYPE_ CAUSAL - LLAMA_
ATTENTION_ TYPE_ NON_ CAUSAL - LLAMA_
ATTENTION_ TYPE_ UNSPECIFIED - LLAMA_
FLASH_ ATTN_ TYPE_ AUTO - LLAMA_
FLASH_ ATTN_ TYPE_ DISABLED - LLAMA_
FLASH_ ATTN_ TYPE_ ENABLED - LLAMA_
FTYPE_ ALL_ F32 - LLAMA_
FTYPE_ GUESSED - LLAMA_
FTYPE_ MOSTLY_ BF16 - LLAMA_
FTYPE_ MOSTLY_ BF16_ R16 - LLAMA_
FTYPE_ MOSTLY_ F16 - LLAMA_
FTYPE_ MOSTLY_ IQ1_ BN - LLAMA_
FTYPE_ MOSTLY_ IQ1_ KT - LLAMA_
FTYPE_ MOSTLY_ IQ1_ M - LLAMA_
FTYPE_ MOSTLY_ IQ1_ M_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ1_ S - LLAMA_
FTYPE_ MOSTLY_ IQ1_ S_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ2_ BN - LLAMA_
FTYPE_ MOSTLY_ IQ2_ BN_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ2_ K - LLAMA_
FTYPE_ MOSTLY_ IQ2_ KL - LLAMA_
FTYPE_ MOSTLY_ IQ2_ KS - LLAMA_
FTYPE_ MOSTLY_ IQ2_ KT - LLAMA_
FTYPE_ MOSTLY_ IQ2_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ2_ M - LLAMA_
FTYPE_ MOSTLY_ IQ2_ M_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ2_ S - LLAMA_
FTYPE_ MOSTLY_ IQ2_ XS - LLAMA_
FTYPE_ MOSTLY_ IQ2_ XS_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ2_ XXS - LLAMA_
FTYPE_ MOSTLY_ IQ2_ XXS_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ3_ K - LLAMA_
FTYPE_ MOSTLY_ IQ3_ KL - LLAMA_
FTYPE_ MOSTLY_ IQ3_ KS - LLAMA_
FTYPE_ MOSTLY_ IQ3_ KT - LLAMA_
FTYPE_ MOSTLY_ IQ3_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ3_ M - LLAMA_
FTYPE_ MOSTLY_ IQ3_ S - LLAMA_
FTYPE_ MOSTLY_ IQ3_ S_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ3_ XS - LLAMA_
FTYPE_ MOSTLY_ IQ3_ XXS - LLAMA_
FTYPE_ MOSTLY_ IQ3_ XXS_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ4_ K - LLAMA_
FTYPE_ MOSTLY_ IQ4_ KS - LLAMA_
FTYPE_ MOSTLY_ IQ4_ KSS - LLAMA_
FTYPE_ MOSTLY_ IQ4_ KS_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ4_ KT - LLAMA_
FTYPE_ MOSTLY_ IQ4_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ4_ NL - LLAMA_
FTYPE_ MOSTLY_ IQ4_ NL_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ4_ XS - LLAMA_
FTYPE_ MOSTLY_ IQ4_ XS_ R8 - LLAMA_
FTYPE_ MOSTLY_ IQ5_ K - LLAMA_
FTYPE_ MOSTLY_ IQ5_ KS - LLAMA_
FTYPE_ MOSTLY_ IQ5_ KS_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ5_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ IQ6_ K - LLAMA_
FTYPE_ MOSTLY_ MXFP4 - LLAMA_
FTYPE_ MOSTLY_ Q1_ 0_ G128 - LLAMA_
FTYPE_ MOSTLY_ Q2_ K - LLAMA_
FTYPE_ MOSTLY_ Q2_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q2_ K_ S - LLAMA_
FTYPE_ MOSTLY_ Q3_ K_ L - LLAMA_
FTYPE_ MOSTLY_ Q3_ K_ M - LLAMA_
FTYPE_ MOSTLY_ Q3_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q3_ K_ S - LLAMA_
FTYPE_ MOSTLY_ Q4_ 0 - LLAMA_
FTYPE_ MOSTLY_ Q4_ 0_ 4_ 4 - LLAMA_
FTYPE_ MOSTLY_ Q4_ 0_ 4_ 8 - LLAMA_
FTYPE_ MOSTLY_ Q4_ 0_ 8_ 8 - LLAMA_
FTYPE_ MOSTLY_ Q4_ 0_ R8 - LLAMA_
FTYPE_ MOSTLY_ Q4_ 1 - LLAMA_
FTYPE_ MOSTLY_ Q4_ K_ M - LLAMA_
FTYPE_ MOSTLY_ Q4_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q4_ K_ S - LLAMA_
FTYPE_ MOSTLY_ Q5_ 0 - LLAMA_
FTYPE_ MOSTLY_ Q5_ 0_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q5_ 1 - LLAMA_
FTYPE_ MOSTLY_ Q5_ K_ M - LLAMA_
FTYPE_ MOSTLY_ Q5_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q5_ K_ S - LLAMA_
FTYPE_ MOSTLY_ Q6_ 0 - LLAMA_
FTYPE_ MOSTLY_ Q6_ 0_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q6_ K - LLAMA_
FTYPE_ MOSTLY_ Q6_ K_ R4 - LLAMA_
FTYPE_ MOSTLY_ Q8_ 0 - LLAMA_
FTYPE_ MOSTLY_ Q8_ 0_ R8 - LLAMA_
FTYPE_ MOSTLY_ Q8_ KV - LLAMA_
FTYPE_ MOSTLY_ Q8_ KV_ R8 - LLAMA_
FTYPE_ MOSTLY_ Q8_ K_ R8 - LLAMA_
KV_ OVERRIDE_ TYPE_ BOOL - LLAMA_
KV_ OVERRIDE_ TYPE_ FLOAT - LLAMA_
KV_ OVERRIDE_ TYPE_ INT - LLAMA_
KV_ OVERRIDE_ TYPE_ STR - LLAMA_
POOLING_ TYPE_ CLS - LLAMA_
POOLING_ TYPE_ LAST - LLAMA_
POOLING_ TYPE_ MEAN - LLAMA_
POOLING_ TYPE_ NONE - LLAMA_
POOLING_ TYPE_ UNSPECIFIED - LLAMA_
ROPE_ SCALING_ TYPE_ LINEAR - LLAMA_
ROPE_ SCALING_ TYPE_ LONGROPE - LLAMA_
ROPE_ SCALING_ TYPE_ MAX_ VALUE - LLAMA_
ROPE_ SCALING_ TYPE_ NONE - LLAMA_
ROPE_ SCALING_ TYPE_ UNSPECIFIED - LLAMA_
ROPE_ SCALING_ TYPE_ YARN - LLAMA_
ROPE_ TYPE_ IMROPE - LLAMA_
ROPE_ TYPE_ MROPE - LLAMA_
ROPE_ TYPE_ NEOX - LLAMA_
ROPE_ TYPE_ NONE - LLAMA_
ROPE_ TYPE_ NORM - LLAMA_
ROPE_ TYPE_ VISION - LLAMA_
RS_ STATUS_ ALLOCATION_ FAILED - LLAMA_
RS_ STATUS_ EXCEPTION - LLAMA_
RS_ STATUS_ INVALID_ ARGUMENT - LLAMA_
RS_ STATUS_ OK - LLAMA_
SPEC_ CKPT_ AUTO - LLAMA_
SPEC_ CKPT_ CPU - LLAMA_
SPEC_ CKPT_ GPU_ FALLBACK - LLAMA_
SPEC_ CKPT_ NONE - LLAMA_
SPEC_ CKPT_ PER_ STEP - LLAMA_
SPLIT_ MODE_ ATTN - LLAMA_
SPLIT_ MODE_ GRAPH - LLAMA_
SPLIT_ MODE_ LAYER - LLAMA_
SPLIT_ MODE_ NONE - LLAMA_
TOKEN_ ATTR_ BYTE - LLAMA_
TOKEN_ ATTR_ CONTROL - LLAMA_
TOKEN_ ATTR_ LSTRIP - LLAMA_
TOKEN_ ATTR_ NORMAL - LLAMA_
TOKEN_ ATTR_ NORMALIZED - LLAMA_
TOKEN_ ATTR_ RSTRIP - LLAMA_
TOKEN_ ATTR_ SINGLE_ WORD - LLAMA_
TOKEN_ ATTR_ UNDEFINED - LLAMA_
TOKEN_ ATTR_ UNKNOWN - LLAMA_
TOKEN_ ATTR_ UNUSED - LLAMA_
TOKEN_ ATTR_ USER_ DEFINED - LLAMA_
TOKEN_ TYPE_ BYTE - LLAMA_
TOKEN_ TYPE_ CONTROL - LLAMA_
TOKEN_ TYPE_ NORMAL - LLAMA_
TOKEN_ TYPE_ UNDEFINED - LLAMA_
TOKEN_ TYPE_ UNKNOWN - LLAMA_
TOKEN_ TYPE_ UNUSED - LLAMA_
TOKEN_ TYPE_ USER_ DEFINED - LLAMA_
VOCAB_ TYPE_ BPE - LLAMA_
VOCAB_ TYPE_ NONE - LLAMA_
VOCAB_ TYPE_ PLAM O2 - LLAMA_
VOCAB_ TYPE_ RWKV - LLAMA_
VOCAB_ TYPE_ SPM - LLAMA_
VOCAB_ TYPE_ UGM - LLAMA_
VOCAB_ TYPE_ WPM - MTP_
OP_ DRAFT_ GEN - MTP_
OP_ NONE - MTP_
OP_ UPDATE_ ACCEPTED - MTP_
OP_ WARMUP
Functions§
- ggml_
abort ⚠ - ggml_
abs ⚠ - ggml_
abs_ ⚠inplace - ggml_
acc ⚠ - ggml_
acc_ ⚠inplace - ggml_
add ⚠ - ggml_
add1 ⚠ - ggml_
add1_ ⚠inplace - ggml_
add_ ⚠cast - ggml_
add_ ⚠id - ggml_
add_ ⚠inplace - ggml_
add_ ⚠rel_ pos - ggml_
add_ ⚠rel_ pos_ inplace - ggml_
arange ⚠ - ggml_
are_ ⚠same_ shape - ggml_
are_ ⚠same_ stride - ggml_
argmax ⚠ - ggml_
argsort ⚠ - ggml_
argsort_ ⚠thresh - ggml_
backend_ ⚠alloc_ buffer - ggml_
backend_ ⚠alloc_ ctx_ tensors - ggml_
backend_ ⚠alloc_ ctx_ tensors_ from_ buft - ggml_
backend_ ⚠buffer_ clear - ggml_
backend_ ⚠buffer_ free - ggml_
backend_ ⚠buffer_ get_ alignment - ggml_
backend_ ⚠buffer_ get_ alloc_ size - ggml_
backend_ ⚠buffer_ get_ base - ggml_
backend_ ⚠buffer_ get_ max_ size - ggml_
backend_ ⚠buffer_ get_ size - ggml_
backend_ ⚠buffer_ get_ type - ggml_
backend_ ⚠buffer_ get_ usage - ggml_
backend_ ⚠buffer_ init_ tensor - ggml_
backend_ ⚠buffer_ is_ host - ggml_
backend_ ⚠buffer_ name - ggml_
backend_ ⚠buffer_ reset - ggml_
backend_ ⚠buffer_ set_ usage - ggml_
backend_ ⚠buft_ alloc_ buffer - ggml_
backend_ ⚠buft_ get_ alignment - ggml_
backend_ ⚠buft_ get_ alloc_ size - ggml_
backend_ ⚠buft_ get_ max_ size - ggml_
backend_ ⚠buft_ is_ host - ggml_
backend_ ⚠buft_ name - ggml_
backend_ ⚠compare_ graph_ backend - ggml_
backend_ ⚠cpu_ buffer_ from_ ptr - ggml_
backend_ ⚠cpu_ buffer_ type - ggml_
backend_ ⚠cpu_ init - ggml_
backend_ ⚠cpu_ set_ abort_ callback - ggml_
backend_ ⚠cpu_ set_ moe_ expert_ prefetch - ggml_
backend_ ⚠cpu_ set_ n_ threads - ggml_
backend_ ⚠event_ free - ggml_
backend_ ⚠event_ new - ggml_
backend_ ⚠event_ record - ggml_
backend_ ⚠event_ synchronize - ggml_
backend_ ⚠event_ wait - ggml_
backend_ ⚠free - ggml_
backend_ ⚠get_ alignment - ggml_
backend_ ⚠get_ default_ buffer_ type - ggml_
backend_ ⚠get_ max_ size - ggml_
backend_ ⚠graph_ compute - ggml_
backend_ ⚠graph_ compute_ async - ggml_
backend_ ⚠graph_ copy - ggml_
backend_ ⚠graph_ copy_ free - ggml_
backend_ ⚠graph_ plan_ compute - ggml_
backend_ ⚠graph_ plan_ create - ggml_
backend_ ⚠graph_ plan_ free - ggml_
backend_ ⚠guid - ggml_
backend_ ⚠is_ cpu - ggml_
backend_ ⚠name - ggml_
backend_ ⚠offload_ op - ggml_
backend_ ⚠prefetch_ init - ggml_
backend_ ⚠prefetch_ register_ mapping - ggml_
backend_ ⚠prefetch_ unregister_ mapping - ggml_
backend_ ⚠reg_ alloc_ buffer - ggml_
backend_ ⚠reg_ find_ by_ name - ggml_
backend_ ⚠reg_ get_ count - ggml_
backend_ ⚠reg_ get_ default_ buffer_ type - ggml_
backend_ ⚠reg_ get_ name - ggml_
backend_ ⚠reg_ init_ backend - ggml_
backend_ ⚠reg_ init_ backend_ from_ str - ggml_
backend_ ⚠sched_ alloc_ graph - ggml_
backend_ ⚠sched_ free - ggml_
backend_ ⚠sched_ get_ backend - ggml_
backend_ ⚠sched_ get_ backend_ idx - ggml_
backend_ ⚠sched_ get_ buffer_ size - ggml_
backend_ ⚠sched_ get_ n_ backends - ggml_
backend_ ⚠sched_ get_ n_ copies - ggml_
backend_ ⚠sched_ get_ n_ splits - ggml_
backend_ ⚠sched_ get_ tensor_ backend - ggml_
backend_ ⚠sched_ graph_ compute - ggml_
backend_ ⚠sched_ graph_ compute_ async - ggml_
backend_ ⚠sched_ new - ggml_
backend_ ⚠sched_ reserve - ggml_
backend_ ⚠sched_ reset - ggml_
backend_ ⚠sched_ set_ eval_ callback - ggml_
backend_ ⚠sched_ set_ max_ extra_ alloc - ggml_
backend_ ⚠sched_ set_ only_ active_ experts - ggml_
backend_ ⚠sched_ set_ op_ offload - ggml_
backend_ ⚠sched_ set_ split_ mode_ graph - ggml_
backend_ ⚠sched_ set_ tensor_ backend - ggml_
backend_ ⚠sched_ synchronize - ggml_
backend_ ⚠supports_ buft - ggml_
backend_ ⚠supports_ op - ggml_
backend_ ⚠synchronize - ggml_
backend_ ⚠tensor_ alloc - ggml_
backend_ ⚠tensor_ copy - ggml_
backend_ ⚠tensor_ copy_ async - ggml_
backend_ ⚠tensor_ get - ggml_
backend_ ⚠tensor_ get_ async - ggml_
backend_ ⚠tensor_ set - ggml_
backend_ ⚠tensor_ set_ async - ggml_
backend_ ⚠view_ init - ggml_
bf16_ ⚠to_ fp32 - ggml_
bf16_ ⚠to_ fp32_ row - ggml_
blck_ ⚠size - ggml_
blend ⚠ - ggml_
build_ ⚠backward_ expand - ggml_
build_ ⚠backward_ gradient_ checkpointing - ggml_
build_ ⚠forward_ expand - ggml_
can_ ⚠repeat - ggml_
cast ⚠ - ggml_
clamp ⚠ - ggml_
concat ⚠ - ggml_
concat_ ⚠inplace - ggml_
cont ⚠ - ggml_
cont_ ⚠1d - ggml_
cont_ ⚠2d - ggml_
cont_ ⚠3d - ggml_
cont_ ⚠4d - ggml_
conv_ ⚠1d - ggml_
conv_ ⚠1d_ ph - ggml_
conv_ ⚠2d - ggml_
conv_ ⚠2d_ dw - ggml_
conv_ ⚠2d_ dw_ direct - ggml_
conv_ ⚠2d_ s1_ ph - ggml_
conv_ ⚠2d_ sk_ p0 - ggml_
conv_ ⚠depthwise_ 2d - ggml_
conv_ ⚠transpose_ 1d - ggml_
conv_ ⚠transpose_ 2d_ p0 - ggml_
cpu_ ⚠has_ arm_ fma - ggml_
cpu_ ⚠has_ avx - ggml_
cpu_ ⚠has_ avx2 - ggml_
cpu_ ⚠has_ avx512 - ggml_
cpu_ ⚠has_ avx512_ bf16 - ggml_
cpu_ ⚠has_ avx512_ vbmi - ggml_
cpu_ ⚠has_ avx512_ vnni - ggml_
cpu_ ⚠has_ avx_ vnni - ggml_
cpu_ ⚠has_ blas - ggml_
cpu_ ⚠has_ cann - ggml_
cpu_ ⚠has_ cuda - ggml_
cpu_ ⚠has_ f16c - ggml_
cpu_ ⚠has_ fma - ggml_
cpu_ ⚠has_ fp16_ va - ggml_
cpu_ ⚠has_ gpublas - ggml_
cpu_ ⚠has_ matmul_ int8 - ggml_
cpu_ ⚠has_ metal - ggml_
cpu_ ⚠has_ neon - ggml_
cpu_ ⚠has_ rpc - ggml_
cpu_ ⚠has_ sse3 - ggml_
cpu_ ⚠has_ ssse3 - ggml_
cpu_ ⚠has_ sve - ggml_
cpu_ ⚠has_ sycl - ggml_
cpu_ ⚠has_ vsx - ggml_
cpu_ ⚠has_ vulkan - ggml_
cpu_ ⚠has_ wasm_ simd - ggml_
cpy ⚠ - ggml_
cross_ ⚠entropy_ loss - ggml_
cross_ ⚠entropy_ loss_ back - ggml_
cumsum ⚠ - ggml_
cycles ⚠ - ggml_
cycles_ ⚠per_ ms - ggml_
delta_ ⚠net - ggml_
diag ⚠ - ggml_
diag_ ⚠mask_ inf - ggml_
diag_ ⚠mask_ inf_ inplace - ggml_
diag_ ⚠mask_ zero - ggml_
diag_ ⚠mask_ zero_ inplace - ggml_
div ⚠ - ggml_
div_ ⚠inplace - ggml_
dup ⚠ - ggml_
dup_ ⚠inplace - ggml_
dup_ ⚠tensor - ggml_
element_ ⚠size - ggml_
elu ⚠ - ggml_
elu_ ⚠inplace - ggml_
exp ⚠ - ggml_
exp_ ⚠inplace - ggml_
fake_ ⚠cpy - ggml_
fill ⚠ - ggml_
fill_ ⚠inplace - ggml_
flash_ ⚠attn_ back - ggml_
flash_ ⚠attn_ ext - ggml_
flash_ ⚠attn_ ext_ add_ sinks - ggml_
flash_ ⚠attn_ ext_ set_ prec - ggml_
fopen ⚠ - ggml_
format_ ⚠name - ggml_
fp16_ ⚠to_ fp32 - ggml_
fp16_ ⚠to_ fp32_ row - ggml_
fp32_ ⚠to_ bf16 - ggml_
fp32_ ⚠to_ bf16_ row - ggml_
fp32_ ⚠to_ bf16_ row_ ref - ggml_
fp32_ ⚠to_ fp16 - ggml_
fp32_ ⚠to_ fp16_ row - ggml_
free ⚠ - ggml_
ftype_ ⚠to_ ggml_ type - ggml_
fused_ ⚠mul_ unary - ggml_
fused_ ⚠mul_ unary_ inplace - ggml_
fused_ ⚠norm - ggml_
fused_ ⚠norm_ inplace - ggml_
fused_ ⚠rms_ norm - ggml_
fused_ ⚠rms_ norm_ inplace - ggml_
fused_ ⚠rms_ rms_ add - ggml_
fused_ ⚠up_ gate - ggml_
gallocr_ ⚠alloc_ graph - ggml_
gallocr_ ⚠free - ggml_
gallocr_ ⚠get_ buffer_ size - ggml_
gallocr_ ⚠new - ggml_
gallocr_ ⚠new_ n - ggml_
gallocr_ ⚠reserve - ggml_
gallocr_ ⚠reserve_ n - ggml_
geglu ⚠ - ggml_
geglu_ ⚠erf - ggml_
geglu_ ⚠erf_ split - ggml_
geglu_ ⚠erf_ swapped - ggml_
geglu_ ⚠quick - ggml_
geglu_ ⚠quick_ split - ggml_
geglu_ ⚠quick_ swapped - ggml_
geglu_ ⚠split - ggml_
geglu_ ⚠swapped - ggml_
gelu ⚠ - ggml_
gelu_ ⚠erf - ggml_
gelu_ ⚠erf_ inplace - ggml_
gelu_ ⚠inplace - ggml_
gelu_ ⚠quick - ggml_
gelu_ ⚠quick_ inplace - ggml_
get_ ⚠data - ggml_
get_ ⚠data_ f32 - ggml_
get_ ⚠f32_ 1d - ggml_
get_ ⚠f32_ nd - ggml_
get_ ⚠first_ tensor - ggml_
get_ ⚠glu_ op - ggml_
get_ ⚠i32_ 1d - ggml_
get_ ⚠i32_ nd - ggml_
get_ ⚠max_ tensor_ size - ggml_
get_ ⚠mem_ buffer - ggml_
get_ ⚠mem_ size - ggml_
get_ ⚠name - ggml_
get_ ⚠next_ tensor - ggml_
get_ ⚠no_ alloc - ggml_
get_ ⚠rel_ pos - ggml_
get_ ⚠rows - ggml_
get_ ⚠rows_ back - ggml_
get_ ⚠tensor - ggml_
get_ ⚠unary_ op - ggml_
glu ⚠ - ggml_
glu_ ⚠op_ name - ggml_
glu_ ⚠split - ggml_
graph_ ⚠clear - ggml_
graph_ ⚠compute - ggml_
graph_ ⚠compute_ with_ ctx - ggml_
graph_ ⚠cpy - ggml_
graph_ ⚠dump_ dot - ggml_
graph_ ⚠dup - ggml_
graph_ ⚠export - ggml_
graph_ ⚠get_ tensor - ggml_
graph_ ⚠import - ggml_
graph_ ⚠n_ nodes - ggml_
graph_ ⚠node - ggml_
graph_ ⚠nodes - ggml_
graph_ ⚠overhead - ggml_
graph_ ⚠overhead_ custom - ggml_
graph_ ⚠plan - ggml_
graph_ ⚠print - ggml_
graph_ ⚠reset - ggml_
graph_ ⚠size - ggml_
graph_ ⚠view - ggml_
group_ ⚠norm - ggml_
group_ ⚠norm_ inplace - ggml_
grouped_ ⚠topk - ggml_
guid_ ⚠matches - ggml_
hadamard ⚠ - ggml_
hardsigmoid ⚠ - ggml_
hardswish ⚠ - ggml_
im2col ⚠ - ggml_
indexer_ ⚠mask - ggml_
indexer_ ⚠topk - ggml_
init ⚠ - ggml_
internal_ ⚠get_ type_ traits - ggml_
interpolate ⚠ - ggml_
is_ ⚠3d - ggml_
is_ ⚠contiguous - ggml_
is_ ⚠contiguous_ 0 - ggml_
is_ ⚠contiguous_ 1 - ggml_
is_ ⚠contiguous_ 2 - ggml_
is_ ⚠contiguous_ channels - ggml_
is_ ⚠contiguous_ rows - ggml_
is_ ⚠contiguously_ allocated - ggml_
is_ ⚠empty - ggml_
is_ ⚠matrix - ggml_
is_ ⚠noop - ggml_
is_ ⚠numa - ggml_
is_ ⚠permuted - ggml_
is_ ⚠quantized - ggml_
is_ ⚠scalar - ggml_
is_ ⚠transposed - ggml_
is_ ⚠vector - ggml_
l2_ ⚠norm - ggml_
l2_ ⚠norm_ inplace - ggml_
leaky_ ⚠relu - ggml_
log ⚠ - ggml_
log_ ⚠inplace - ggml_
map_ ⚠binary_ f32 - ggml_
map_ ⚠binary_ inplace_ f32 - ggml_
map_ ⚠custom1 - ggml_
map_ ⚠custom2 - ggml_
map_ ⚠custom3 - ggml_
map_ ⚠custom1_ f32 - ggml_
map_ ⚠custom1_ inplace - ggml_
map_ ⚠custom1_ inplace_ f32 - ggml_
map_ ⚠custom2_ f32 - ggml_
map_ ⚠custom2_ inplace - ggml_
map_ ⚠custom2_ inplace_ f32 - ggml_
map_ ⚠custom3_ f32 - ggml_
map_ ⚠custom3_ inplace - ggml_
map_ ⚠custom3_ inplace_ f32 - ggml_
map_ ⚠unary_ f32 - ggml_
map_ ⚠unary_ inplace_ f32 - ggml_
mean ⚠ - ggml_
moe_ ⚠up_ gate - ggml_
moe_ ⚠up_ gate_ ext - ggml_
mul ⚠ - ggml_
mul_ ⚠inplace - ggml_
mul_ ⚠mat - ggml_
mul_ ⚠mat_ id - ggml_
mul_ ⚠mat_ inplace - ggml_
mul_ ⚠mat_ set_ prec - ggml_
mul_ ⚠multi_ add - ggml_
multi_ ⚠add - ggml_
n_ ⚠dims - ggml_
nbytes ⚠ - ggml_
nbytes_ ⚠pad - ggml_
neg ⚠ - ggml_
neg_ ⚠inplace - ggml_
nelements ⚠ - ggml_
new_ ⚠f32 - ggml_
new_ ⚠graph - ggml_
new_ ⚠graph_ custom - ggml_
new_ ⚠i32 - ggml_
new_ ⚠tensor - ggml_
new_ ⚠tensor_ 1d - ggml_
new_ ⚠tensor_ 2d - ggml_
new_ ⚠tensor_ 3d - ggml_
new_ ⚠tensor_ 4d - ggml_
norm ⚠ - ggml_
norm_ ⚠inplace - ggml_
nrows ⚠ - ggml_
numa_ ⚠init - ggml_
op_ ⚠desc - ggml_
op_ ⚠name - ggml_
op_ ⚠symbol - ggml_
opt ⚠ - ggml_
opt_ ⚠default_ params - ggml_
opt_ ⚠init - ggml_
opt_ ⚠resume - ggml_
opt_ ⚠resume_ g - ggml_
out_ ⚠prod - ggml_
pad ⚠ - ggml_
permute ⚠ - ggml_
pool_ ⚠1d - ggml_
pool_ ⚠2d - ggml_
print_ ⚠object - ggml_
print_ ⚠objects - ggml_
quantize_ ⚠chunk - ggml_
quantize_ ⚠free - ggml_
quantize_ ⚠init - ggml_
quantize_ ⚠requires_ imatrix - ggml_
reduce ⚠ - ggml_
reglu ⚠ - ggml_
reglu_ ⚠split - ggml_
reglu_ ⚠swapped - ggml_
relu ⚠ - ggml_
relu_ ⚠inplace - ggml_
repeat ⚠ - ggml_
repeat_ ⚠4d - ggml_
repeat_ ⚠back - ggml_
reshape ⚠ - ggml_
reshape_ ⚠1d - ggml_
reshape_ ⚠2d - ggml_
reshape_ ⚠3d - ggml_
reshape_ ⚠4d - ggml_
reshape_ ⚠4d_ ext - ggml_
rms_ ⚠norm - ggml_
rms_ ⚠norm_ back - ggml_
rms_ ⚠norm_ inplace - ggml_
rope ⚠ - ggml_
rope_ ⚠back - ggml_
rope_ ⚠cache - ggml_
rope_ ⚠custom - ggml_
rope_ ⚠custom_ inplace - ggml_
rope_ ⚠ext - ggml_
rope_ ⚠ext_ inplace - ggml_
rope_ ⚠fast - ggml_
rope_ ⚠inplace - ggml_
rope_ ⚠multi - ggml_
rope_ ⚠multi_ inplace - ggml_
rope_ ⚠yarn_ corr_ dims - ggml_
row_ ⚠size - ggml_
scale ⚠ - ggml_
scale_ ⚠bias - ggml_
scale_ ⚠bias_ inplace - ggml_
scale_ ⚠inplace - ggml_
set ⚠ - ggml_
set_ ⚠1d - ggml_
set_ ⚠1d_ inplace - ggml_
set_ ⚠2d - ggml_
set_ ⚠2d_ inplace - ggml_
set_ ⚠f32 - ggml_
set_ ⚠f32_ 1d - ggml_
set_ ⚠f32_ nd - ggml_
set_ ⚠i32 - ggml_
set_ ⚠i32_ 1d - ggml_
set_ ⚠i32_ nd - ggml_
set_ ⚠inplace - ggml_
set_ ⚠input - ggml_
set_ ⚠name - ggml_
set_ ⚠no_ alloc - ggml_
set_ ⚠output - ggml_
set_ ⚠param - ggml_
set_ ⚠rows - ggml_
set_ ⚠scratch - ggml_
set_ ⚠zero - ggml_
sgn ⚠ - ggml_
sgn_ ⚠inplace - ggml_
sigmoid ⚠ - ggml_
sigmoid_ ⚠inplace - ggml_
silu ⚠ - ggml_
silu_ ⚠back - ggml_
silu_ ⚠inplace - ggml_
sinkhorn ⚠ - ggml_
soft_ ⚠max - ggml_
soft_ ⚠max_ add_ sinks - ggml_
soft_ ⚠max_ back - ggml_
soft_ ⚠max_ back_ inplace - ggml_
soft_ ⚠max_ ext - ggml_
soft_ ⚠max_ inplace - ggml_
softcap ⚠ - ggml_
softcap_ ⚠inplace - ggml_
softcap_ ⚠max - ggml_
softcap_ ⚠max_ inplace - ggml_
softplus ⚠ - ggml_
softplus_ ⚠inplace - ggml_
solve_ ⚠tri - ggml_
sqr ⚠ - ggml_
sqr_ ⚠inplace - ggml_
sqrt ⚠ - ggml_
sqrt_ ⚠inplace - ggml_
ssm_ ⚠conv - ggml_
ssm_ ⚠scan - ggml_
status_ ⚠to_ string - ggml_
step ⚠ - ggml_
step_ ⚠inplace - ggml_
sub ⚠ - ggml_
sub_ ⚠inplace - ggml_
sum ⚠ - ggml_
sum_ ⚠rows - ggml_
sum_ ⚠rows_ ext - ggml_
swiglu ⚠ - ggml_
swiglu_ ⚠oai - ggml_
swiglu_ ⚠split - ggml_
swiglu_ ⚠swapped - ggml_
tallocr_ ⚠alloc - ggml_
tallocr_ ⚠new - ggml_
tanh ⚠ - ggml_
tanh_ ⚠inplace - ggml_
tensor_ ⚠overhead - ggml_
time_ ⚠init - ggml_
time_ ⚠ms - ggml_
time_ ⚠us - ggml_
timestep_ ⚠embedding - ggml_
top_ ⚠k - ggml_
top_ ⚠k_ thresh - ggml_
transpose ⚠ - ggml_
tri ⚠ - ggml_
type_ ⚠name - ggml_
type_ ⚠size - ggml_
type_ ⚠sizef - ggml_
unary ⚠ - ggml_
unary_ ⚠inplace - ggml_
unary_ ⚠op_ name - ggml_
unravel_ ⚠index - ggml_
upscale ⚠ - ggml_
upscale_ ⚠ext - ggml_
used_ ⚠mem - ggml_
validate_ ⚠row_ data - ggml_
view_ ⚠1d - ggml_
view_ ⚠2d - ggml_
view_ ⚠3d - ggml_
view_ ⚠4d - ggml_
view_ ⚠tensor - ggml_
win_ ⚠part - ggml_
win_ ⚠unpart - gguf_
add_ ⚠tensor - gguf_
find_ ⚠key - gguf_
find_ ⚠tensor - gguf_
free ⚠ - gguf_
get_ ⚠alignment - gguf_
get_ ⚠arr_ data - gguf_
get_ ⚠arr_ n - gguf_
get_ ⚠arr_ str - gguf_
get_ ⚠arr_ type - gguf_
get_ ⚠data - gguf_
get_ ⚠data_ offset - gguf_
get_ ⚠key - gguf_
get_ ⚠kv_ type - gguf_
get_ ⚠meta_ data - gguf_
get_ ⚠meta_ size - gguf_
get_ ⚠n_ kv - gguf_
get_ ⚠n_ tensors - gguf_
get_ ⚠tensor_ name - gguf_
get_ ⚠tensor_ offset - gguf_
get_ ⚠tensor_ type - gguf_
get_ ⚠val_ bool - gguf_
get_ ⚠val_ data - gguf_
get_ ⚠val_ f32 - gguf_
get_ ⚠val_ f64 - gguf_
get_ ⚠val_ i8 - gguf_
get_ ⚠val_ i16 - gguf_
get_ ⚠val_ i32 - gguf_
get_ ⚠val_ i64 - gguf_
get_ ⚠val_ str - gguf_
get_ ⚠val_ u8 - gguf_
get_ ⚠val_ u16 - gguf_
get_ ⚠val_ u32 - gguf_
get_ ⚠val_ u64 - gguf_
get_ ⚠version - gguf_
init_ ⚠empty - gguf_
init_ ⚠from_ file - gguf_
remove_ ⚠key - gguf_
set_ ⚠arr_ data - gguf_
set_ ⚠arr_ str - gguf_
set_ ⚠kv - gguf_
set_ ⚠tensor_ data - gguf_
set_ ⚠tensor_ type - gguf_
set_ ⚠val_ bool - gguf_
set_ ⚠val_ f32 - gguf_
set_ ⚠val_ f64 - gguf_
set_ ⚠val_ i8 - gguf_
set_ ⚠val_ i16 - gguf_
set_ ⚠val_ i32 - gguf_
set_ ⚠val_ i64 - gguf_
set_ ⚠val_ str - gguf_
set_ ⚠val_ u8 - gguf_
set_ ⚠val_ u16 - gguf_
set_ ⚠val_ u32 - gguf_
set_ ⚠val_ u64 - gguf_
type_ ⚠name - gguf_
write_ ⚠to_ file - ik_
llama_ ⚠rs_ grammar_ accept - ik_
llama_ ⚠rs_ grammar_ apply - ik_
llama_ ⚠rs_ json_ schema_ to_ grammar - ik_
llama_ ⚠rs_ mtp_ begin - ik_
llama_ ⚠rs_ mtp_ free - ik_
llama_ ⚠rs_ mtp_ init - ik_
llama_ ⚠rs_ mtp_ step - ik_
llama_ ⚠rs_ string_ free - llama_
add_ ⚠bos_ token - llama_
add_ ⚠eos_ token - llama_
backend_ ⚠free - llama_
backend_ ⚠init - llama_
batch_ ⚠free - llama_
batch_ ⚠get_ one - llama_
batch_ ⚠init - llama_
chat_ ⚠apply_ template - Apply chat template. Inspired by hf apply_chat_template() on python. Both “model” and “custom_template” are optional, but at least one is required. “custom_template” has higher precedence than “model” NOTE: This function does not use a jinja parser. It only support a pre-defined list of template. See more: https://github.com/ggerganov/llama.cpp/wiki/Templates-supported-by-llama_chat_apply_template @param tmpl A Jinja template to use for this chat. If this is nullptr, the model’s default chat template will be used instead. @param chat Pointer to a list of multiple llama_chat_message @param n_msg Number of llama_chat_message in this chat @param add_ass Whether to end the prompt with the token(s) that indicate the start of an assistant message. @param buf A buffer to hold the output formatted prompt. The recommended alloc size is 2 * (total number of characters of all messages) @param length The size of the allocated buffer @return The total number of bytes of the formatted prompt. If is it larger than the size of buffer, you may need to re-alloc it and then re-apply the template.
- llama_
chat_ ⚠builtin_ templates - llama_
context_ ⚠default_ params - llama_
control_ ⚠vector_ apply - llama_
copy_ ⚠state_ data - llama_
decode ⚠ - llama_
detokenize ⚠ - llama_
dump_ ⚠timing_ info_ yaml - llama_
encode ⚠ - llama_
fill_ ⚠from_ utf8 - llama_
free ⚠ - llama_
free_ ⚠model - llama_
get_ ⚠dflash_ draft_ token_ ith - llama_
get_ ⚠embeddings - llama_
get_ ⚠embeddings_ ith - llama_
get_ ⚠embeddings_ seq - llama_
get_ ⚠kv_ cache_ token_ count - llama_
get_ ⚠kv_ cache_ used_ cells - llama_
get_ ⚠logits - llama_
get_ ⚠logits_ ith - llama_
get_ ⚠model - llama_
get_ ⚠model_ tensor - llama_
get_ ⚠model_ vocab - llama_
get_ ⚠state_ size - llama_
get_ ⚠timings - llama_
grammar_ ⚠accept_ token - @details Accepts the sampled token into the grammar
- llama_
grammar_ ⚠apply - @details Apply constraints from grammar
- llama_
grammar_ ⚠copy - llama_
grammar_ ⚠free - llama_
grammar_ ⚠init_ lazy - llama_
init_ ⚠adaptive_ p - @details Adaptive p sampler initializer @param target Select tokens near this probability (valid range 0.0 to 1.0; <0 = disabled) @param decay Decay rate for target adaptation over time. lower values -> faster but less stable adaptation. (valid range 0.0 to 1.0; ≤0 = no adaptation)
- llama_
init_ ⚠from_ model - llama_
is_ ⚠gemma4_ mtp_ file - llama_
kv_ ⚠cache_ clear - llama_
kv_ ⚠cache_ defrag - llama_
kv_ ⚠cache_ seq_ add - llama_
kv_ ⚠cache_ seq_ cp - llama_
kv_ ⚠cache_ seq_ div - llama_
kv_ ⚠cache_ seq_ keep - llama_
kv_ ⚠cache_ seq_ pos_ max - llama_
kv_ ⚠cache_ seq_ pos_ min - llama_
kv_ ⚠cache_ seq_ rm - llama_
kv_ ⚠cache_ update - llama_
kv_ ⚠cache_ view_ free - llama_
kv_ ⚠cache_ view_ init - llama_
kv_ ⚠cache_ view_ update - llama_
load_ ⚠session_ file - llama_
log_ ⚠set - llama_
lora_ ⚠adapter_ clear - llama_
lora_ ⚠adapter_ free - llama_
lora_ ⚠adapter_ init - llama_
lora_ ⚠adapter_ remove - llama_
lora_ ⚠adapter_ set - llama_
max_ ⚠devices - llama_
model_ ⚠arch_ string - llama_
model_ ⚠chat_ template - llama_
model_ ⚠decoder_ start_ token - llama_
model_ ⚠default_ params - llama_
model_ ⚠desc - llama_
model_ ⚠get_ vocab - llama_
model_ ⚠has_ decoder - llama_
model_ ⚠has_ encoder - llama_
model_ ⚠has_ recurrent - llama_
model_ ⚠is_ gemma4_ mtp_ assistant - llama_
model_ ⚠is_ hybrid - llama_
model_ ⚠is_ openpangu - llama_
model_ ⚠is_ recurrent - llama_
model_ ⚠is_ split_ mode_ graph - llama_
model_ ⚠load_ from_ file - llama_
model_ ⚠meta_ count - llama_
model_ ⚠meta_ key_ by_ index - llama_
model_ ⚠meta_ val_ str - llama_
model_ ⚠meta_ val_ str_ by_ index - llama_
model_ ⚠n_ embd - llama_
model_ ⚠n_ embd_ inp - llama_
model_ ⚠n_ nextn_ layer - llama_
model_ ⚠n_ params - llama_
model_ ⚠quantize - llama_
model_ ⚠quantize_ default_ params - llama_
model_ ⚠size - llama_
model_ ⚠supports_ ctx_ shift - llama_
model_ ⚠supports_ partial_ kv_ reuse - llama_
n_ ⚠batch - llama_
n_ ⚠ctx - llama_
n_ ⚠ctx_ train - llama_
n_ ⚠layer - llama_
n_ ⚠seq_ max - llama_
n_ ⚠threads - llama_
n_ ⚠threads_ batch - llama_
n_ ⚠ubatch - llama_
n_ ⚠vocab - llama_
numa_ ⚠init - llama_
pooling_ ⚠type - llama_
prep_ ⚠adaptive_ p - llama_
print_ ⚠system_ info - llama_
print_ ⚠timings - llama_
reload_ ⚠changed_ tensors - llama_
reset_ ⚠timings - llama_
review_ ⚠adaptive_ p - llama_
rope_ ⚠freq_ scale_ train - llama_
rope_ ⚠type - llama_
sample_ ⚠adaptive_ p - @details Adaptive p sampler described in https://github.com/MrJackSpade/adaptive-p-docs/blob/main/README.md
- llama_
sample_ ⚠apply_ guidance - @details Apply classifier-free guidance to the logits as described in academic paper “Stay on topic with Classifier-Free Guidance” https://arxiv.org/abs/2306.17806 @param logits Logits extracted from the original generation context. @param logits_guidance Logits extracted from a separate context from the same model. Other than a negative prompt at the beginning, it should have all generated and user input tokens copied from the main context. @param scale Guidance strength. 1.0f means no guidance. Higher values mean stronger guidance.
- llama_
sample_ ⚠dist - llama_
sample_ ⚠dry - llama_
sample_ ⚠entropy - @details Dynamic temperature implementation described in the paper https://arxiv.org/abs/2309.02772.
- llama_
sample_ ⚠grammar - llama_
sample_ ⚠min_ p - @details Minimum P sampling as described in https://github.com/ggerganov/llama.cpp/pull/3841
- llama_
sample_ ⚠repetition_ penalties - @details Repetition penalty described in CTRL academic paper https://arxiv.org/abs/1909.05858, with negative logit fix. @details Frequency and presence penalties described in OpenAI API https://platform.openai.com/docs/api-reference/parameter-details.
- llama_
sample_ ⚠softmax - @details Sorts candidate tokens by their logits in descending order and calculate probabilities based on logits.
- llama_
sample_ ⚠tail_ free - @details Tail Free Sampling described in https://www.trentonbricken.com/Tail-Free-Sampling/.
- llama_
sample_ ⚠temp - llama_
sample_ ⚠token - @details Randomly selects a token from the candidates based on their probabilities using the RNG of ctx.
- llama_
sample_ ⚠token_ adaptive_ p - @details Randonly selects a token from the candidates following adaptive p sampler.
- llama_
sample_ ⚠token_ greedy - @details Selects the token with the highest probability. Does not compute the token probabilities. Use llama_sample_softmax() instead.
- llama_
sample_ ⚠token_ mirostat - @details Mirostat 1.0 algorithm described in the paper https://arxiv.org/abs/2007.14966. Uses tokens instead of words.
@param candidates A vector of
llama_token_datacontaining the candidate tokens, their probabilities (p), and log-odds (logit) for the current position in the generated text. @param tau The target cross-entropy (or surprise) value you want to achieve for the generated text. A higher value corresponds to more surprising or less predictable text, while a lower value corresponds to less surprising or more predictable text. @param eta The learning rate used to updatemubased on the error between the target and observed surprisal of the sampled word. A larger learning rate will causemuto be updated more quickly, while a smaller learning rate will result in slower updates. @param m The number of tokens considered in the estimation ofs_hat. This is an arbitrary value that is used to calculates_hat, which in turn helps to calculate the value ofk. In the paper, they usem = 100, but you can experiment with different values to see how it affects the performance of the algorithm. @param mu Maximum cross-entropy. This value is initialized to be twice the target cross-entropy (2 * tau) and is updated in the algorithm based on the error between the target and observed surprisal. - llama_
sample_ ⚠token_ mirostat_ v2 - @details Mirostat 2.0 algorithm described in the paper https://arxiv.org/abs/2007.14966. Uses tokens instead of words.
@param candidates A vector of
llama_token_datacontaining the candidate tokens, their probabilities (p), and log-odds (logit) for the current position in the generated text. @param tau The target cross-entropy (or surprise) value you want to achieve for the generated text. A higher value corresponds to more surprising or less predictable text, while a lower value corresponds to less surprising or more predictable text. @param eta The learning rate used to updatemubased on the error between the target and observed surprisal of the sampled word. A larger learning rate will causemuto be updated more quickly, while a smaller learning rate will result in slower updates. @param mu Maximum cross-entropy. This value is initialized to be twice the target cross-entropy (2 * tau) and is updated in the algorithm based on the error between the target and observed surprisal. - llama_
sample_ ⚠top_ k - @details Top-K sampling described in academic paper “The Curious Case of Neural Text Degeneration” https://arxiv.org/abs/1904.09751
- llama_
sample_ ⚠top_ n_ sigma - @details Top n sigma sampling as described in academic paper “Top-nσ: Not All Logits Are You Need” https://arxiv.org/pdf/2411.07641
- llama_
sample_ ⚠top_ p - @details Nucleus sampling described in academic paper “The Curious Case of Neural Text Degeneration” https://arxiv.org/abs/1904.09751
- llama_
sample_ ⚠typical - @details Locally Typical Sampling implementation described in the paper https://arxiv.org/abs/2202.00666.
- llama_
sample_ ⚠xtc - @details XTC sampler as described in https://github.com/oobabooga/text-generation-webui/pull/6335
- llama_
sampler_ ⚠dry_ accept - llama_
sampler_ ⚠dry_ clone - llama_
sampler_ ⚠dry_ free - llama_
sampler_ ⚠dry_ reset - llama_
sampler_ ⚠init_ dry - @details DRY sampler, designed by p-e-w, as described in: https://github.com/oobabooga/text-generation-webui/pull/5677, porting Koboldcpp implementation authored by pi6am: https://github.com/LostRuins/koboldcpp/pull/982
- llama_
sampler_ ⚠init_ grammar - @details Intializes a GBNF grammar, see grammars/README.md for details. @param vocab The vocabulary that this grammar will be used with. @param grammar_str The production rules for the grammar, encoded as a string. Returns an empty grammar if empty. Returns NULL if parsing of grammar_str fails. @param grammar_root The name of the start symbol for the grammar.
- llama_
sampler_ ⚠init_ grammar_ lazy - @details Lazy grammar sampler, introduced in https://github.com/ggerganov/llama.cpp/pull/9639 @param trigger_words A list of words that will trigger the grammar sampler. This may be updated to a loose regex syntax (w/ ^) in a near future. @param trigger_tokens A list of tokens that will trigger the grammar sampler.
- llama_
sampler_ ⚠init_ grammar_ lazy_ patterns - @details Lazy grammar sampler, introduced in https://github.com/ggml-org/llama.cpp/pull/9639 @param trigger_patterns A list of patterns that will trigger the grammar sampler. Pattern will be matched from the start of the generation output, and grammar sampler will be fed content starting from its first match group. @param trigger_tokens A list of tokens that will trigger the grammar sampler. Grammar sampler will be fed content starting from the trigger token included.
- llama_
sampler_ ⚠reset - llama_
save_ ⚠session_ file - llama_
set_ ⚠abort_ callback - llama_
set_ ⚠causal_ attn - llama_
set_ ⚠draft_ input_ hidden_ state - llama_
set_ ⚠embeddings - llama_
set_ ⚠mtp_ op_ type - llama_
set_ ⚠n_ threads - llama_
set_ ⚠offload_ policy - llama_
set_ ⚠rng_ seed - llama_
set_ ⚠state_ data - llama_
spec_ ⚠ckpt_ discard - llama_
spec_ ⚠ckpt_ init - llama_
spec_ ⚠ckpt_ restore - llama_
spec_ ⚠ckpt_ save - llama_
split_ ⚠path - @details Build a split GGUF final path for this chunk. llama_split_path(split_path, sizeof(split_path), “/models/ggml-model-q4_0”, 2, 4) => split_path = “/models/ggml-model-q4_0-00002-of-00004.gguf”
- llama_
split_ ⚠prefix - @details Extract the path prefix from the split_path if and only if the split_no and split_count match. llama_split_prefix(split_prefix, 64, “/models/ggml-model-q4_0-00002-of-00004.gguf”, 2, 4) => split_prefix = “/models/ggml-model-q4_0”
- llama_
state_ ⚠get_ data - llama_
state_ ⚠get_ size - llama_
state_ ⚠load_ file - llama_
state_ ⚠save_ file - llama_
state_ ⚠seq_ get_ data - llama_
state_ ⚠seq_ get_ size - llama_
state_ ⚠seq_ load_ file - llama_
state_ ⚠seq_ save_ file - llama_
state_ ⚠seq_ set_ data - llama_
state_ ⚠set_ data - llama_
supports_ ⚠gpu_ offload - llama_
supports_ ⚠mlock - llama_
supports_ ⚠mmap - llama_
synchronize ⚠ - llama_
time_ ⚠us - llama_
token_ ⚠bos - llama_
token_ ⚠cls - llama_
token_ ⚠eos - llama_
token_ ⚠eot - llama_
token_ ⚠get_ attr - llama_
token_ ⚠get_ score - llama_
token_ ⚠get_ text - llama_
token_ ⚠is_ control - llama_
token_ ⚠is_ eog - llama_
token_ ⚠middle - llama_
token_ ⚠nl - llama_
token_ ⚠pad - llama_
token_ ⚠prefix - llama_
token_ ⚠sep - llama_
token_ ⚠suffix - llama_
token_ ⚠to_ piece - llama_
token_ ⚠to_ piece_ vocab - llama_
tokenize ⚠ - @details Convert the provided text into tokens. @param tokens The tokens pointer must be large enough to hold the resulting tokens. @return Returns the number of tokens on success, no more than n_tokens_max @return Returns a negative number on failure - the number of tokens that would have been returned @param add_special Allow to add BOS and EOS tokens if model is configured to do so. @param parse_special Allow tokenizing special and/or control tokens which otherwise are not exposed and treated as plaintext. Does not insert a leading space.
- llama_
vocab_ ⚠bos - llama_
vocab_ ⚠eos - llama_
vocab_ ⚠get_ add_ bos - llama_
vocab_ ⚠get_ add_ eos - llama_
vocab_ ⚠get_ text - llama_
vocab_ ⚠is_ eog - llama_
vocab_ ⚠n_ tokens - llama_
vocab_ ⚠tokenize - llama_
vocab_ ⚠type
Type Aliases§
- FILE
- _IO_
lock_ t - __
off64_ t - __off_t
- ggml_
abort_ callback - ggml_
backend_ buffer_ t - ggml_
backend_ buffer_ type_ t - ggml_
backend_ buffer_ usage - ggml_
backend_ eval_ callback - ggml_
backend_ event_ t - ggml_
backend_ graph_ plan_ t - ggml_
backend_ sched_ eval_ callback - ggml_
backend_ sched_ t - ggml_
backend_ t - ggml_
backend_ type - ggml_
binary_ op_ f32_ t - ggml_
bitset_ t - ggml_
cgraph_ eval_ order - ggml_
custom1_ op_ f32_ t - ggml_
custom1_ op_ t - ggml_
custom2_ op_ f32_ t - ggml_
custom2_ op_ t - ggml_
custom3_ op_ f32_ t - ggml_
custom3_ op_ t - ggml_
fp16_ t - ggml_
from_ float_ t - ggml_
from_ float_ to_ mat_ t - ggml_
ftype - ggml_
gallocr_ t - ggml_
gemm_ t - ggml_
gemv_ t - ggml_
glu_ op - ggml_
guid - ggml_
guid_ t - ggml_
linesearch - ggml_
log_ callback - ggml_
log_ level - ggml_
numa_ strategy - ggml_
object_ type - ggml_op
- ggml_
op_ pool - ggml_
opt_ callback - ggml_
opt_ result - ggml_
opt_ type - ggml_
prec - ggml_
scale_ flag - ggml_
scale_ mode - ggml_
sort_ order - ggml_
status - ggml_
tensor_ flag - ggml_
to_ float_ t - ggml_
tri_ type - ggml_
type - ggml_
unary_ op - ggml_
unary_ op_ f32_ t - ggml_
vec_ dot_ t - gguf_
type - llama_
attention_ type - llama_
flash_ attn_ type - llama_
ftype - llama_
model_ kv_ override_ type - llama_
mtp_ op_ type - llama_
pooling_ type - llama_
pos - llama_
progress_ callback - llama_
rope_ scaling_ type - llama_
rope_ type - llama_
rs_ status - llama_
sampler_ context_ t - llama_
seq_ id - llama_
spec_ ckpt_ mode - llama_
split_ mode - llama_
state_ seq_ flags - llama_
token - llama_
token_ attr - llama_
token_ type - llama_
vocab_ type