Packages

A terminal UI framework for Elixir with a high-performance Zig NIF backend.

Current section

Files

Jump to
elixir_opentui zig opentui text-buffer-segment.zig
Raw

zig/opentui/text-buffer-segment.zig

const std = @import("std");
const Allocator = std.mem.Allocator;
const rope_mod = @import("rope.zig");
const buffer = @import("buffer.zig");
const mem_registry_mod = @import("mem-registry.zig");
const gp = @import("grapheme.zig");
const utf8 = @import("utf8.zig");
pub const RGBA = buffer.RGBA;
pub const TextSelection = buffer.TextSelection;
pub const TextBufferError = error{
OutOfMemory,
InvalidDimensions,
InvalidIndex,
InvalidId,
InvalidMemId,
};
const MemRegistry = mem_registry_mod.MemRegistry;
pub const WrapMode = enum {
none,
char,
word,
};
pub const ChunkFitResult = struct {
char_count: u32,
width: u32,
};
pub const GraphemeInfo = utf8.GraphemeInfo;
/// A chunk represents a contiguous sequence of UTF-8 bytes from a specific memory buffer
pub const TextChunk = struct {
mem_id: u8,
byte_start: u32,
byte_end: u32,
width: u16,
flags: u8 = 0,
graphemes: ?[]GraphemeInfo = null,
wrap_offsets: ?[]utf8.WrapBreak = null,
pub const Flags = struct {
pub const ASCII_ONLY: u8 = 0b00000001; // Printable ASCII only (32..126).
};
pub fn isAsciiOnly(self: *const TextChunk) bool {
return (self.flags & Flags.ASCII_ONLY) != 0;
}
pub fn empty() TextChunk {
return .{
.mem_id = 0,
.byte_start = 0,
.byte_end = 0,
.width = 0,
};
}
pub fn is_empty(self: *const TextChunk) bool {
return self.width == 0;
}
pub fn getBytes(self: *const TextChunk, mem_registry: *const MemRegistry) []const u8 {
const mem_buf = mem_registry.get(self.mem_id) orelse return &[_]u8{};
return mem_buf[self.byte_start..self.byte_end];
}
/// Lazily compute and cache grapheme info for this chunk
/// Returns a slice that is valid until the buffer is reset
/// For ASCII-only chunks, returns an empty slice (sentinel)
/// For mixed chunks, returns only multibyte (non-ASCII) graphemes and tabs with their column offsets
pub fn getGraphemes(
self: *const TextChunk,
mem_registry: *const MemRegistry,
allocator: Allocator,
tabwidth: u8,
width_method: utf8.WidthMethod,
) TextBufferError![]const GraphemeInfo {
const mut_self = @constCast(self);
if (self.graphemes) |cached| {
return cached;
}
if (self.isAsciiOnly()) {
const empty_slice = try allocator.alloc(GraphemeInfo, 0);
mut_self.graphemes = empty_slice;
return empty_slice;
}
const chunk_bytes = self.getBytes(mem_registry);
var grapheme_list: std.ArrayListUnmanaged(GraphemeInfo) = .{};
errdefer grapheme_list.deinit(allocator);
try utf8.findGraphemeInfo(chunk_bytes, tabwidth, self.isAsciiOnly(), width_method, allocator, &grapheme_list);
// TODO: Calling this with an arena allocator will just double the memory usage?
const graphemes = try grapheme_list.toOwnedSlice(allocator);
mut_self.graphemes = graphemes;
return graphemes;
}
/// Lazily compute and cache wrap offsets for this chunk
/// Returns a slice that is valid until the buffer is reset
pub fn getWrapOffsets(
self: *const TextChunk,
mem_registry: *const MemRegistry,
allocator: Allocator,
width_method: utf8.WidthMethod,
) TextBufferError![]const utf8.WrapBreak {
const mut_self = @constCast(self);
if (self.wrap_offsets) |cached| {
return cached;
}
const chunk_bytes = self.getBytes(mem_registry);
var wrap_result = utf8.WrapBreakResult.init(allocator);
errdefer wrap_result.deinit();
try utf8.findWrapBreaks(chunk_bytes, &wrap_result, width_method);
// TODO: Do not cache for chunks < 64 bytes, as it does not profit from the cache
// Use toOwnedSlice to transfer ownership without copying
const wrap_offsets = try wrap_result.breaks.toOwnedSlice(allocator);
mut_self.wrap_offsets = wrap_offsets;
return wrap_offsets;
}
};
/// A highlight represents a styled region on a line
pub const Highlight = struct {
col_start: u32,
col_end: u32,
style_id: u32,
priority: u8,
hl_ref: u16 = 0,
};
/// Pre-computed style span for efficient rendering
/// Represents a contiguous region with a single style
pub const StyleSpan = struct {
col: u32,
style_id: u32,
next_col: u32,
};
/// A segment in the unified rope - either text content or a line break marker
pub const Segment = union(enum) {
text: TextChunk,
brk: void,
linestart: void,
/// Define which union tags are markers (for O(1) line lookup)
pub const MarkerTypes = &[_]std.meta.Tag(Segment){ .brk, .linestart };
/// Metrics for aggregation in the rope tree
/// These enable O(log n) row/col coordinate mapping and efficient line queries
pub const Metrics = struct {
total_width: u32 = 0,
total_bytes: u32 = 0,
linestart_count: u32 = 0,
newline_count: u32 = 0,
max_line_width: u32 = 0,
/// Whether all text segments in subtree are ASCII-only (for fast wrapping paths)
ascii_only: bool = true,
pub fn add(self: *Metrics, other: Metrics) void {
self.total_width += other.total_width;
self.total_bytes += other.total_bytes;
self.linestart_count += other.linestart_count;
self.newline_count += other.newline_count;
self.max_line_width = @max(self.max_line_width, other.max_line_width);
self.ascii_only = self.ascii_only and other.ascii_only;
}
/// Get the balancing weight for the rope
/// We use total_width + newline_count to give each break a weight of 1
/// This eliminates boundary ambiguity in coordinate/offset conversions
pub fn weight(self: *const Metrics) u32 {
return self.total_width + self.newline_count;
}
};
/// Measure this segment to produce its metrics
pub fn measure(self: *const Segment) Metrics {
return switch (self.*) {
.text => |chunk| blk: {
const is_ascii = (chunk.flags & TextChunk.Flags.ASCII_ONLY) != 0;
const byte_len = chunk.byte_end - chunk.byte_start;
break :blk Metrics{
.total_width = chunk.width,
.total_bytes = byte_len,
.linestart_count = 0,
.newline_count = 0,
.max_line_width = chunk.width,
.ascii_only = is_ascii,
};
},
.brk => Metrics{
.total_width = 0,
.total_bytes = 0,
.linestart_count = 0,
.newline_count = 1,
.max_line_width = 0,
.ascii_only = true,
},
.linestart => Metrics{
.total_width = 0,
.total_bytes = 0,
.linestart_count = 1,
.newline_count = 0,
.max_line_width = 0,
.ascii_only = true,
},
};
}
pub fn empty() Segment {
return .{ .text = TextChunk.empty() };
}
pub fn is_empty(self: *const Segment) bool {
return switch (self.*) {
.text => |chunk| chunk.is_empty(),
.brk => false,
.linestart => false,
};
}
pub fn getBytes(self: *const Segment, mem_registry: *const MemRegistry) []const u8 {
return switch (self.*) {
.text => |chunk| chunk.getBytes(mem_registry),
.brk => &[_]u8{},
.linestart => &[_]u8{},
};
}
pub fn isBreak(self: *const Segment) bool {
return switch (self.*) {
.brk => true,
else => false,
};
}
pub fn isLineStart(self: *const Segment) bool {
return switch (self.*) {
.linestart => true,
else => false,
};
}
pub fn isText(self: *const Segment) bool {
return switch (self.*) {
.text => true,
else => false,
};
}
pub fn asText(self: *const Segment) ?*const TextChunk {
return switch (self.*) {
.text => |*chunk| chunk,
else => null,
};
}
/// Two text chunks can be merged if they reference contiguous memory in the same buffer
pub fn canMerge(left: *const Segment, right: *const Segment) bool {
if (!left.isText() or !right.isText()) return false;
const left_chunk = left.asText() orelse return false;
const right_chunk = right.asText() orelse return false;
if (left_chunk.mem_id != right_chunk.mem_id) return false;
if (left_chunk.byte_end != right_chunk.byte_start) return false;
if (left_chunk.flags != right_chunk.flags) return false;
return true;
}
pub fn merge(allocator: Allocator, left: *const Segment, right: *const Segment) Segment {
_ = allocator;
const left_chunk = left.asText().?;
const right_chunk = right.asText().?;
// TODO: could clear the caches on the original chunks,
// as the original chunks are only kept for history purposes.
return Segment{
.text = TextChunk{
.mem_id = left_chunk.mem_id,
.byte_start = left_chunk.byte_start,
.byte_end = right_chunk.byte_end,
.width = left_chunk.width + right_chunk.width,
.flags = left_chunk.flags,
.graphemes = null,
.wrap_offsets = null,
},
};
}
/// Boundary normalization action
pub const BoundaryAction = struct {
delete_left: bool = false,
delete_right: bool = false,
insert_between: []const Segment = &[_]Segment{},
};
/// Rewrite boundary between two adjacent segments to enforce invariants
///
/// Document invariants enforced at join boundaries:
/// - Every line starts with a linestart marker
/// - Line breaks must be followed by linestart markers
/// - No duplicate linestart markers (deduplicated automatically)
/// - When joining lines, orphaned linestart markers are removed
/// - Empty lines are represented as [linestart, brk] with no text, or [linestart] if final
/// - Consecutive breaks [brk, brk] get a linestart inserted between (empty line)
///
/// Rules applied locally at O(log n) join points:
/// - [linestart, linestart] → delete right (dedup)
/// - [brk, text] → insert linestart between (ensure line starts with marker)
/// - [brk, brk] → insert linestart between (represents empty line)
/// - [text, linestart] → delete right (remove orphaned linestart when joining lines)
///
/// Valid patterns (no action needed):
/// - [text, brk] (line content followed by break)
/// - [linestart, text] (line marker followed by content)
/// - [linestart, brk] (empty line before another line)
/// - [linestart] alone (empty final line or empty buffer)
/// - [brk, linestart, brk] (empty line between two lines, normalized from [brk, brk])
///
/// These rules preserve linestart markers when deleting at col=0 within a line,
/// since the deletion splits around the marker, and [text, linestart] only triggers
/// when actually joining lines (deleting the break between them).
pub fn rewriteBoundary(allocator: Allocator, left: ?*const Segment, right: ?*const Segment) !BoundaryAction {
_ = allocator;
if (left == null or right == null) return .{};
const left_seg = left.?;
const right_seg = right.?;
// [linestart, linestart] -> delete right (dedup)
if (left_seg.isLineStart() and right_seg.isLineStart()) {
return .{ .delete_right = true };
}
// [brk, brk] -> insert linestart between (represents empty line)
if (left_seg.isBreak() and right_seg.isBreak()) {
const linestart_segment = Segment{ .linestart = {} };
const insert_slice = &[_]Segment{linestart_segment};
return .{ .insert_between = insert_slice };
}
// [brk, text] -> insert linestart between
if (left_seg.isBreak() and right_seg.isText()) {
const linestart_segment = Segment{ .linestart = {} };
const insert_slice = &[_]Segment{linestart_segment};
return .{ .insert_between = insert_slice };
}
// [text, linestart] -> delete right (remove orphaned linestart when joining lines)
if (left_seg.isText() and right_seg.isLineStart()) {
return .{ .delete_right = true };
}
return .{};
}
/// Rewrite rope ends to enforce invariants
/// Rules:
/// - Rope must start with linestart (even when empty - ensures at least one line)
pub fn rewriteEnds(allocator: Allocator, first: ?*const Segment, last: ?*const Segment) !BoundaryAction {
_ = allocator;
_ = last;
// Ensure rope starts with linestart (insert even if empty)
if (first) |first_seg| {
if (!first_seg.isLineStart()) {
const linestart_segment = Segment{ .linestart = {} };
const insert_slice = &[_]Segment{linestart_segment};
return .{ .insert_between = insert_slice };
}
} else {
// Empty rope - insert linestart to ensure at least one line
const linestart_segment = Segment{ .linestart = {} };
const insert_slice = &[_]Segment{linestart_segment};
return .{ .insert_between = insert_slice };
}
return .{};
}
};
pub const UnifiedRope = rope_mod.Rope(Segment);