Packages
llama_cpp_ex
0.6.3
0.8.36
0.8.35
0.8.34
0.8.33
0.8.32
0.8.31
0.8.28
0.8.27
0.8.26
0.8.25
0.8.24
0.8.23
0.8.22
0.8.21
0.8.20
0.8.19
0.8.18
0.8.17
0.8.16
0.8.15
0.8.14
0.8.13
0.8.12
0.8.11
0.8.10
0.8.9
0.8.8
0.8.7
0.8.6
0.8.5
0.8.4
0.8.3
0.8.2
0.8.1
0.8.0
0.7.9
0.7.8
0.7.7
0.7.6
0.7.5
0.7.4
0.7.3
0.7.2
0.7.0
0.6.14
0.6.13
0.6.12
0.6.11
0.6.10
0.6.9
0.6.8
0.6.7
0.6.6
0.6.5
0.6.4
0.6.3
0.6.1
0.6.0
0.5.0
0.4.4
0.4.3
0.4.2
0.4.1
0.3.0
0.2.0
Elixir bindings for llama.cpp — run LLMs locally with Metal, CUDA, Vulkan, or CPU acceleration.
Current section
Files
Jump to
Current section
Files
lib/llama_cpp_ex/sampler.ex
defmodule LlamaCppEx.Sampler do
@moduledoc """
Token sampling configuration.
Builds a sampler chain with the common sampling parameters.
The samplers are applied in order: grammar -> penalties -> top_k -> top_p -> min_p -> temp -> dist/greedy.
"""
@enforce_keys [:ref]
defstruct [:ref]
@type t :: %__MODULE__{ref: reference()}
@doc """
Creates a new sampler chain.
Requires a model reference (needed for grammar-constrained sampling).
## Options
* `:seed` - Random seed for sampling. Defaults to a random value.
* `:temp` - Temperature. `0.0` for greedy sampling. Defaults to `0.8`.
* `:top_k` - Top-K filtering. `0` to disable. Defaults to `40`.
* `:top_p` - Top-P (nucleus) filtering. `1.0` to disable. Defaults to `0.95`.
* `:min_p` - Min-P filtering. `0.0` to disable. Defaults to `0.05`.
* `:penalty_repeat` - Repetition penalty. `1.0` to disable. Defaults to `1.0`.
* `:penalty_freq` - Frequency penalty (0.0–2.0). `0.0` to disable. Defaults to `0.0`.
* `:penalty_present` - Presence penalty (0.0–2.0). `0.0` to disable. Defaults to `0.0`.
* `:grammar` - GBNF grammar string for constrained generation. Defaults to `""` (none).
* `:grammar_root` - Root rule name for grammar. Defaults to `"root"`.
"""
@spec create(LlamaCppEx.Model.t(), keyword()) :: {:ok, t()}
def create(%LlamaCppEx.Model{ref: model_ref}, opts \\ []) do
seed = Keyword.get(opts, :seed, :rand.uniform(1_000_000_000))
temp = Keyword.get(opts, :temp, 0.8)
top_k = Keyword.get(opts, :top_k, 40)
top_p = Keyword.get(opts, :top_p, 0.95)
min_p = Keyword.get(opts, :min_p, 0.05)
penalty_repeat = Keyword.get(opts, :penalty_repeat, 1.0)
penalty_freq = Keyword.get(opts, :penalty_freq, 0.0)
penalty_present = Keyword.get(opts, :penalty_present, 0.0)
grammar = Keyword.get(opts, :grammar, "")
grammar_root = Keyword.get(opts, :grammar_root, "root")
ref =
LlamaCppEx.NIF.sampler_init(
model_ref,
seed,
temp / 1,
top_k,
top_p / 1,
min_p / 1,
penalty_repeat / 1,
penalty_freq / 1,
penalty_present / 1,
grammar,
grammar_root
)
{:ok, %__MODULE__{ref: ref}}
end
@doc "Resets the sampler state."
@spec reset(t()) :: :ok
def reset(%__MODULE__{ref: ref}), do: LlamaCppEx.NIF.sampler_reset(ref)
@doc "Accepts a token (updates sampler internal state)."
@spec accept(t(), integer()) :: :ok
def accept(%__MODULE__{ref: ref}, token), do: LlamaCppEx.NIF.sampler_accept(ref, token)
@doc "Samples the next token from the context's logits."
@spec sample(t(), LlamaCppEx.Context.t()) :: integer()
def sample(%__MODULE__{ref: ref}, %LlamaCppEx.Context{ref: ctx_ref}) do
LlamaCppEx.NIF.sampler_sample(ref, ctx_ref)
end
end