Packages
caffeine_lang
6.3.0
6.3.1
6.3.0
6.2.2
6.2.1
6.2.0
6.1.2
6.1.1
6.1.0
6.0.0
5.6.0
5.5.0
5.4.4
5.4.3
5.4.2
5.4.1
5.4.0
5.3.0
5.2.0
5.1.1
5.1.0
5.0.12
5.0.11
5.0.10
5.0.8
5.0.7
5.0.6
5.0.5
5.0.4
5.0.1
5.0.0
4.10.0
4.9.0
4.8.3
4.8.2
4.8.1
4.8.0
4.7.9
4.7.8
4.7.7
4.7.6
4.7.5
4.6.7
4.6.6
4.6.5
4.6.4
4.6.3
4.6.2
4.6.0
4.5.1
4.5.0
4.4.4
4.4.3
4.4.1
4.4.0
4.3.7
4.3.6
3.0.6
3.0.5
3.0.4
3.0.3
3.0.2
3.0.1
3.0.0
2.0.5
2.0.4
2.0.3
2.0.2
2.0.1
2.0.0
1.0.2
1.0.1
0.1.0
0.0.24
0.0.23
0.0.22
0.0.21
0.0.20
0.0.19
0.0.18
0.0.17
0.0.16
0.0.15
0.0.14
0.0.13
0.0.12
0.0.11
0.0.10
0.0.9
0.0.8
0.0.7
0.0.6
0.0.5
0.0.4
0.0.2
0.0.1
A compiler for generating reliability artifacts from service expectation definitions.
Current section
Files
Jump to
Current section
Files
src/caffeine_lang/frontend/tokenizer_ffi.mjs
// Fast tokenizer primitives. gleam_stdlib's string.pop_grapheme constructs
// a fresh Intl.Segmenter().segment(s)[Symbol.iterator]() on every call and
// walks it from index 0, turning the tokenizer's per-char loop into O(N^2).
// We split by codepoint instead — which only differs from grapheme for
// combining-mark / ZWJ sequences inside string-literal or comment bodies,
// neither of which the tokenizer needs to slice grapheme-correctly (it just
// scans for ASCII terminators).
// Returns [first_codepoint_as_string, rest]. Returns ["", ""] when source is
// empty — the Gleam wrapper turns that back into Error(Nil).
export function pop_codepoint(s) {
if (s.length === 0) return ["", ""];
const cp = s.codePointAt(0);
const charLen = cp > 0xffff ? 2 : 1;
return [String.fromCodePoint(cp), s.substring(charLen)];
}
// Returns the UTF-16 code unit at index i, or -1 if i is out of bounds.
// Used by is_digit / is_letter on single-codepoint strings.
export function code_unit_at(s, i) {
if (i < 0 || i >= s.length) return -1;
return s.charCodeAt(i);
}
// Codeunit length — used to advance the column counter without re-walking
// the just-read token via Intl.Segmenter.
export function code_unit_length(s) {
return s.length;
}