Packages
localize
0.15.0
1.0.0-rc.4
1.0.0-rc.3
1.0.0-rc.2
1.0.0-rc.1
1.0.0-rc.0
0.50.0
0.49.0
0.48.0
0.47.0
0.46.0
0.45.0
0.44.0
0.41.3
0.41.2
0.41.1
0.41.0
0.40.0
0.39.0
0.38.0
0.37.0
0.36.0
0.35.0
0.34.0
0.33.0
0.32.0
0.31.0
0.30.1
0.30.0
retired
0.29.0
0.28.0
0.27.0
0.26.0
0.25.0
0.24.0
0.23.0
0.22.0
0.21.0
0.20.0
0.19.0
0.18.0
0.16.0
0.15.0
0.14.0
0.13.0
0.12.0
0.11.0
0.10.0
0.9.0
0.8.0
0.7.0
0.6.0
0.5.0
0.4.0
0.3.0
0.2.0
0.1.0
0.1.0-alpha.1
Localization (parsing, formatting) of numbers, dates/time/calendar, units of measure, messages and lists. Includes localized collation.
Current section
Files
Jump to
Current section
Files
lib/localize/collation/normalizer.ex
defmodule Localize.Collation.Normalizer do
@moduledoc """
Unicode NFD normalization for collation.
Delegates to Erlang's `:unicode` module.
"""
@doc """
Normalize a string to NFD (Canonical Decomposition) form.
Uses Erlang's `:unicode.characters_to_nfd_binary/1` followed by a canonical
reordering pass using the `unicode` package's CCC data to correct ordering
for newer Unicode codepoints.
### Arguments
* `string` - a UTF-8 binary string.
### Returns
The NFD-normalized string as a UTF-8 binary.
### Examples
iex> "café" |> Localize.Collation.Normalizer.nfd() |> String.to_charlist() |> length()
5
iex> Localize.Collation.Normalizer.nfd("e\u0301")
"e\u0301"
"""
@spec nfd(String.t()) :: String.t()
def nfd(string) when is_binary(string) do
string
|> :unicode.characters_to_nfd_binary()
|> String.to_charlist()
|> canonical_reorder()
|> List.to_string()
end
@doc """
Convert a string to a list of integer codepoints.
### Arguments
* `string` - a UTF-8 binary string.
### Returns
A list of integer codepoints.
### Examples
iex> Localize.Collation.Normalizer.to_codepoints("abc")
[97, 98, 99]
"""
@spec to_codepoints(String.t()) :: [non_neg_integer(), ...]
def to_codepoints(string) when is_binary(string) do
String.to_charlist(string)
end
@doc """
Optionally normalize a string and convert it to a list of integer codepoints.
### Arguments
* `string` - a UTF-8 binary string.
* `normalize?` - whether to apply NFD normalization first (default: `false`).
### Returns
A list of integer codepoints, optionally NFD-normalized.
### Examples
iex> Localize.Collation.Normalizer.normalize_to_codepoints("abc")
[97, 98, 99]
iex> Localize.Collation.Normalizer.normalize_to_codepoints("café", true)
[99, 97, 102, 101, 769]
"""
@spec normalize_to_codepoints(String.t(), boolean()) :: [non_neg_integer()]
def normalize_to_codepoints(string, normalize? \\ false) do
string
|> then(fn s -> if normalize?, do: nfd(s), else: s end)
|> to_codepoints()
end
defp canonical_reorder(codepoints) do
case reorder_pass(codepoints, false) do
{result, true} -> canonical_reorder(result)
{result, false} -> result
end
end
defp reorder_pass([a, b | rest], swapped) do
ccc_a = combining_class(a)
ccc_b = combining_class(b)
if ccc_a > ccc_b and ccc_b > 0 do
{tail, s} = reorder_pass([a | rest], true)
{[b | tail], s}
else
{tail, s} = reorder_pass([b | rest], swapped)
{[a | tail], s}
end
end
defp reorder_pass([cp], swapped), do: {[cp], swapped}
defp reorder_pass([], swapped), do: {[], swapped}
defp combining_class(cp) do
Localize.Collation.Unicode.combining_class(cp)
end
end