Packages
llama_cpp_ex
0.8.36
0.8.36
0.8.35
0.8.34
0.8.33
0.8.32
0.8.31
0.8.28
0.8.27
0.8.26
0.8.25
0.8.24
0.8.23
0.8.22
0.8.21
0.8.20
0.8.19
0.8.18
0.8.17
0.8.16
0.8.15
0.8.14
0.8.13
0.8.12
0.8.11
0.8.10
0.8.9
0.8.8
0.8.7
0.8.6
0.8.5
0.8.4
0.8.3
0.8.2
0.8.1
0.8.0
0.7.9
0.7.8
0.7.7
0.7.6
0.7.5
0.7.4
0.7.3
0.7.2
0.7.0
0.6.14
0.6.13
0.6.12
0.6.11
0.6.10
0.6.9
0.6.8
0.6.7
0.6.6
0.6.5
0.6.4
0.6.3
0.6.1
0.6.0
0.5.0
0.4.4
0.4.3
0.4.2
0.4.1
0.3.0
0.2.0
Elixir bindings for llama.cpp — run LLMs locally with Metal, CUDA, Vulkan, or CPU acceleration.
Current section
Files
Jump to
Current section
Files
lib/llama_cpp_ex/server/strategy/balanced.ex
defmodule LlamaCppEx.Server.Strategy.Balanced do
@moduledoc """
Balanced batching strategy.
Splits the token budget equally between decode and prefill operations.
Decode tokens always use 1 token per slot, so the decode half is capped
at the number of generating slots. The prefill half gets the remainder.
Fair under mixed workloads where both generation latency and prefill
throughput matter equally.
"""
@behaviour LlamaCppEx.Server.BatchStrategy
alias LlamaCppEx.Server.Strategy.Batch
@impl true
def build_batch(slots, budget, chunk_size, _opts) do
n_generating =
Enum.count(slots, fn {_id, slot} ->
slot.state == :generating and slot.pending_token != nil
end)
# Split budget: decode gets half (capped at generating count), prefill the rest.
decode_budget = min(div(budget, 2), n_generating)
prefill_budget = budget - decode_budget
{entries, n_entries, slots, decode_remaining} =
Batch.add_decode_tokens(slots, [], 0, decode_budget)
# Prefill gets its half plus any unused decode budget.
effective_prefill_budget = prefill_budget + decode_remaining
{entries, _n_entries, slots, _budget} =
Batch.add_prefill_chunks(slots, entries, n_entries, effective_prefill_budget, chunk_size)
{Enum.reverse(entries), slots}
end
end