Packages
llama_cpp_ex
0.6.10
0.8.36
0.8.35
0.8.34
0.8.33
0.8.32
0.8.31
0.8.28
0.8.27
0.8.26
0.8.25
0.8.24
0.8.23
0.8.22
0.8.21
0.8.20
0.8.19
0.8.18
0.8.17
0.8.16
0.8.15
0.8.14
0.8.13
0.8.12
0.8.11
0.8.10
0.8.9
0.8.8
0.8.7
0.8.6
0.8.5
0.8.4
0.8.3
0.8.2
0.8.1
0.8.0
0.7.9
0.7.8
0.7.7
0.7.6
0.7.5
0.7.4
0.7.3
0.7.2
0.7.0
0.6.14
0.6.13
0.6.12
0.6.11
0.6.10
0.6.9
0.6.8
0.6.7
0.6.6
0.6.5
0.6.4
0.6.3
0.6.1
0.6.0
0.5.0
0.4.4
0.4.3
0.4.2
0.4.1
0.3.0
0.2.0
Elixir bindings for llama.cpp — run LLMs locally with Metal, CUDA, Vulkan, or CPU acceleration.
Current section
Files
Jump to
Current section
Files
lib/llama_cpp_ex/chat.ex
defmodule LlamaCppEx.Chat do
@moduledoc """
Chat template formatting using llama.cpp's Jinja template engine.
Converts a list of chat messages into a formatted prompt string
using the model's embedded chat template. Uses the full Jinja engine
from llama.cpp's common library, which supports `enable_thinking` and
arbitrary `chat_template_kwargs`.
## Examples
{:ok, prompt} = LlamaCppEx.Chat.apply_template(model, [
%{role: "system", content: "You are helpful."},
%{role: "user", content: "Hi!"}
])
# Disable thinking (for Qwen3 and similar models)
{:ok, prompt} = LlamaCppEx.Chat.apply_template(model, messages,
enable_thinking: false
)
"""
@type message :: %{role: String.t(), content: String.t()} | {String.t(), String.t()}
@doc """
Applies the model's chat template to a list of messages using the Jinja engine.
## Options
* `:add_assistant` - Whether to add the assistant turn prefix. Defaults to `true`.
* `:enable_thinking` - Whether to enable thinking/reasoning mode. Defaults to `true`.
* `:chat_template_kwargs` - Extra template variables as a list of `{key, value}` string tuples.
Defaults to `[]`.
"""
@spec apply_template(LlamaCppEx.Model.t(), [message()], keyword()) ::
{:ok, String.t()} | {:error, String.t()}
def apply_template(%LlamaCppEx.Model{} = model, messages, opts \\ []) when is_list(messages) do
add_assistant = Keyword.get(opts, :add_assistant, true)
enable_thinking = Keyword.get(opts, :enable_thinking, true)
extra_kwargs = Keyword.get(opts, :chat_template_kwargs, [])
msg_tuples =
Enum.map(messages, fn
%{role: role, content: content} -> {to_string(role), to_string(content)}
%{"role" => role, "content" => content} -> {to_string(role), to_string(content)}
{role, content} -> {to_string(role), to_string(content)}
end)
kwargs_tuples =
Enum.map(extra_kwargs, fn {k, v} -> {to_string(k), to_string(v)} end)
try do
result =
LlamaCppEx.NIF.chat_apply_template_jinja(
model.ref,
msg_tuples,
add_assistant,
enable_thinking,
kwargs_tuples
)
{:ok, result}
rescue
e in ErlangError -> {:error, "chat template failed: #{inspect(e.original)}"}
end
end
end