B2 fix — clamp rich-text slice boundaries to char boundaries (crash)

Typing in a large file crashed: "byte index N is not a char boundary;
it is inside '→'". `projected_rich_chunks` slices `current_text` at
span / decoration / adornment byte offsets, but those offsets come from
the daemon for a possibly-earlier generation than the rope this frame
holds (the one-frame edit race). After an edit a stale offset can land
inside a multi-byte codepoint, panicking `text[a..b]`.

Snap every boundary to the previous UTF-8 char boundary before slicing
(new stable `floor_char_boundary` helper; the older `style_runs_for_text`
path already did the equivalent `is_char_boundary` guard — this newer
adornment-aware path was missing it). Flooring only shifts a chunk edge
left to the start of the codepoint it fell inside; chunks still
reassemble the original text.

Tests: `projected_rich_chunks_tolerates_mid_codepoint_boundaries`
(span ending mid-'→' + a past-end diagnostic; chunks reassemble the
text) and `floor_char_boundary_snaps_into_multibyte_char`.

Gates: fmt; clippy --all-targets --workspace -D warnings; pmacs-gpu
unit 22 (+2).

NOTE: this fixes the crash, not the large-file slowness — that is the
whole-file reshape architecture (projected_rich_chunks + set_rich_text
are O(file), run per edit, and the daemon runs a whole-file tree-sitter
highlight query per edit). Making large-file editing usable needs
viewport-scoped rendering + scrolling, scoped on both the GPU and the
producer. That is its own session.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Levi Neuwirth 2026-05-30 10:21:09 -04:00
parent 5888ced15a
commit 0f5f4d1769
1 changed files with 59 additions and 5 deletions

View File

@ -1846,6 +1846,21 @@ fn should_forward_key(key: ProtocolKey, mods: Modifiers) -> bool {
) )
} }
/// Largest char-boundary `<= index` (stable equivalent of the unstable
/// `str::floor_char_boundary`). Used to snap externally-supplied byte
/// offsets to valid slice points so a stale, mid-codepoint offset can't
/// panic a `text[..]` slice.
fn floor_char_boundary(text: &str, index: usize) -> usize {
if index >= text.len() {
return text.len();
}
let mut i = index;
while i > 0 && !text.is_char_boundary(i) {
i -= 1;
}
i
}
/// Buffer-absolute byte offset of the start of each `\n`-delimited /// Buffer-absolute byte offset of the start of each `\n`-delimited
/// line (index 0 = byte 0). Indexed by cosmic-text's /// line (index 0 = byte 0). Indexed by cosmic-text's
/// `LayoutRun::line_i` to rebase line-relative glyph offsets. /// `LayoutRun::line_i` to rebase line-relative glyph offsets.
@ -1925,14 +1940,22 @@ fn projected_rich_chunks(
adornments: &[InlineAdornment], adornments: &[InlineAdornment],
) -> Vec<RichChunk> { ) -> Vec<RichChunk> {
let text_len = text.len() as u64; let text_len = text.len() as u64;
// Every boundary used to slice `text` must be snapped to a UTF-8
// char boundary. Span / decoration / adornment offsets come from
// the daemon for a possibly-earlier generation than the rope this
// frame holds (the one-frame edit race), so a raw offset can land
// inside a multi-byte char and panic the slice. Flooring to the
// previous char boundary is safe: it only shifts a chunk edge left
// to the start of the codepoint it fell inside.
let snap = |b: u64| floor_char_boundary(text, b.min(text_len) as usize) as u64;
let mut boundaries: Vec<u64> = vec![0, text_len]; let mut boundaries: Vec<u64> = vec![0, text_len];
for sp in spans { for sp in spans {
boundaries.push(sp.range.start.min(text_len)); boundaries.push(snap(sp.range.start));
boundaries.push(sp.range.end.min(text_len)); boundaries.push(snap(sp.range.end));
} }
for d in decorations { for d in decorations {
boundaries.push(d.range.start.min(text_len)); boundaries.push(snap(d.range.start));
boundaries.push(d.range.end.min(text_len)); boundaries.push(snap(d.range.end));
} }
let mut renderable_adornments: Vec<(usize, u64, &InlineAdornment)> = adornments let mut renderable_adornments: Vec<(usize, u64, &InlineAdornment)> = adornments
.iter() .iter()
@ -1940,7 +1963,7 @@ fn projected_rich_chunks(
.filter_map(|(idx, a)| renderable_adornment_anchor(a, text_len).map(|at| (idx, at, a))) .filter_map(|(idx, a)| renderable_adornment_anchor(a, text_len).map(|at| (idx, at, a)))
.collect(); .collect();
for (_, at, _) in &renderable_adornments { for (_, at, _) in &renderable_adornments {
boundaries.push(*at); boundaries.push(snap(*at));
} }
boundaries.sort_unstable(); boundaries.sort_unstable();
boundaries.dedup(); boundaries.dedup();
@ -2429,6 +2452,37 @@ mod tests {
} }
} }
#[test]
fn projected_rich_chunks_tolerates_mid_codepoint_boundaries() {
// Stale span offsets (from a prior generation) can land inside a
// multi-byte char after an edit. "ab→cd": '→' is the 3 bytes
// [2,5); a span ending at byte 3 is mid-codepoint and must not
// panic the slice — it floors to the char start.
let text = "ab→cd";
let chunks = projected_rich_chunks(
text,
&[span(0, 3, CellColor::Indexed(1))],
&[Decoration {
range: ByteRange { start: 4, end: 9 },
kind: DecorationKind::DiagnosticError,
}],
&[],
);
let rendered: String = chunks.iter().map(|chunk| chunk.text.as_str()).collect();
assert_eq!(rendered, text, "chunks must reassemble the original text");
}
#[test]
fn floor_char_boundary_snaps_into_multibyte_char() {
let text = "ab→cd"; // '→' = bytes [2,5)
assert_eq!(floor_char_boundary(text, 0), 0);
assert_eq!(floor_char_boundary(text, 2), 2);
assert_eq!(floor_char_boundary(text, 3), 2); // inside '→' → floor to 2
assert_eq!(floor_char_boundary(text, 4), 2);
assert_eq!(floor_char_boundary(text, 5), 5);
assert_eq!(floor_char_boundary(text, 99), text.len());
}
#[test] #[test]
fn projected_rich_chunks_inserts_at_offset_without_source_bytes() { fn projected_rich_chunks_inserts_at_offset_without_source_bytes() {
let chunks = projected_rich_chunks( let chunks = projected_rich_chunks(