B2 fix — clamp rich-text slice boundaries to char boundaries (crash)
Typing in a large file crashed: "byte index N is not a char boundary; it is inside '→'". `projected_rich_chunks` slices `current_text` at span / decoration / adornment byte offsets, but those offsets come from the daemon for a possibly-earlier generation than the rope this frame holds (the one-frame edit race). After an edit a stale offset can land inside a multi-byte codepoint, panicking `text[a..b]`. Snap every boundary to the previous UTF-8 char boundary before slicing (new stable `floor_char_boundary` helper; the older `style_runs_for_text` path already did the equivalent `is_char_boundary` guard — this newer adornment-aware path was missing it). Flooring only shifts a chunk edge left to the start of the codepoint it fell inside; chunks still reassemble the original text. Tests: `projected_rich_chunks_tolerates_mid_codepoint_boundaries` (span ending mid-'→' + a past-end diagnostic; chunks reassemble the text) and `floor_char_boundary_snaps_into_multibyte_char`. Gates: fmt; clippy --all-targets --workspace -D warnings; pmacs-gpu unit 22 (+2). NOTE: this fixes the crash, not the large-file slowness — that is the whole-file reshape architecture (projected_rich_chunks + set_rich_text are O(file), run per edit, and the daemon runs a whole-file tree-sitter highlight query per edit). Making large-file editing usable needs viewport-scoped rendering + scrolling, scoped on both the GPU and the producer. That is its own session. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
parent
5888ced15a
commit
0f5f4d1769
|
|
@ -1846,6 +1846,21 @@ fn should_forward_key(key: ProtocolKey, mods: Modifiers) -> bool {
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Largest char-boundary `<= index` (stable equivalent of the unstable
|
||||||
|
/// `str::floor_char_boundary`). Used to snap externally-supplied byte
|
||||||
|
/// offsets to valid slice points so a stale, mid-codepoint offset can't
|
||||||
|
/// panic a `text[..]` slice.
|
||||||
|
fn floor_char_boundary(text: &str, index: usize) -> usize {
|
||||||
|
if index >= text.len() {
|
||||||
|
return text.len();
|
||||||
|
}
|
||||||
|
let mut i = index;
|
||||||
|
while i > 0 && !text.is_char_boundary(i) {
|
||||||
|
i -= 1;
|
||||||
|
}
|
||||||
|
i
|
||||||
|
}
|
||||||
|
|
||||||
/// Buffer-absolute byte offset of the start of each `\n`-delimited
|
/// Buffer-absolute byte offset of the start of each `\n`-delimited
|
||||||
/// line (index 0 = byte 0). Indexed by cosmic-text's
|
/// line (index 0 = byte 0). Indexed by cosmic-text's
|
||||||
/// `LayoutRun::line_i` to rebase line-relative glyph offsets.
|
/// `LayoutRun::line_i` to rebase line-relative glyph offsets.
|
||||||
|
|
@ -1925,14 +1940,22 @@ fn projected_rich_chunks(
|
||||||
adornments: &[InlineAdornment],
|
adornments: &[InlineAdornment],
|
||||||
) -> Vec<RichChunk> {
|
) -> Vec<RichChunk> {
|
||||||
let text_len = text.len() as u64;
|
let text_len = text.len() as u64;
|
||||||
|
// Every boundary used to slice `text` must be snapped to a UTF-8
|
||||||
|
// char boundary. Span / decoration / adornment offsets come from
|
||||||
|
// the daemon for a possibly-earlier generation than the rope this
|
||||||
|
// frame holds (the one-frame edit race), so a raw offset can land
|
||||||
|
// inside a multi-byte char and panic the slice. Flooring to the
|
||||||
|
// previous char boundary is safe: it only shifts a chunk edge left
|
||||||
|
// to the start of the codepoint it fell inside.
|
||||||
|
let snap = |b: u64| floor_char_boundary(text, b.min(text_len) as usize) as u64;
|
||||||
let mut boundaries: Vec<u64> = vec![0, text_len];
|
let mut boundaries: Vec<u64> = vec![0, text_len];
|
||||||
for sp in spans {
|
for sp in spans {
|
||||||
boundaries.push(sp.range.start.min(text_len));
|
boundaries.push(snap(sp.range.start));
|
||||||
boundaries.push(sp.range.end.min(text_len));
|
boundaries.push(snap(sp.range.end));
|
||||||
}
|
}
|
||||||
for d in decorations {
|
for d in decorations {
|
||||||
boundaries.push(d.range.start.min(text_len));
|
boundaries.push(snap(d.range.start));
|
||||||
boundaries.push(d.range.end.min(text_len));
|
boundaries.push(snap(d.range.end));
|
||||||
}
|
}
|
||||||
let mut renderable_adornments: Vec<(usize, u64, &InlineAdornment)> = adornments
|
let mut renderable_adornments: Vec<(usize, u64, &InlineAdornment)> = adornments
|
||||||
.iter()
|
.iter()
|
||||||
|
|
@ -1940,7 +1963,7 @@ fn projected_rich_chunks(
|
||||||
.filter_map(|(idx, a)| renderable_adornment_anchor(a, text_len).map(|at| (idx, at, a)))
|
.filter_map(|(idx, a)| renderable_adornment_anchor(a, text_len).map(|at| (idx, at, a)))
|
||||||
.collect();
|
.collect();
|
||||||
for (_, at, _) in &renderable_adornments {
|
for (_, at, _) in &renderable_adornments {
|
||||||
boundaries.push(*at);
|
boundaries.push(snap(*at));
|
||||||
}
|
}
|
||||||
boundaries.sort_unstable();
|
boundaries.sort_unstable();
|
||||||
boundaries.dedup();
|
boundaries.dedup();
|
||||||
|
|
@ -2429,6 +2452,37 @@ mod tests {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn projected_rich_chunks_tolerates_mid_codepoint_boundaries() {
|
||||||
|
// Stale span offsets (from a prior generation) can land inside a
|
||||||
|
// multi-byte char after an edit. "ab→cd": '→' is the 3 bytes
|
||||||
|
// [2,5); a span ending at byte 3 is mid-codepoint and must not
|
||||||
|
// panic the slice — it floors to the char start.
|
||||||
|
let text = "ab→cd";
|
||||||
|
let chunks = projected_rich_chunks(
|
||||||
|
text,
|
||||||
|
&[span(0, 3, CellColor::Indexed(1))],
|
||||||
|
&[Decoration {
|
||||||
|
range: ByteRange { start: 4, end: 9 },
|
||||||
|
kind: DecorationKind::DiagnosticError,
|
||||||
|
}],
|
||||||
|
&[],
|
||||||
|
);
|
||||||
|
let rendered: String = chunks.iter().map(|chunk| chunk.text.as_str()).collect();
|
||||||
|
assert_eq!(rendered, text, "chunks must reassemble the original text");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn floor_char_boundary_snaps_into_multibyte_char() {
|
||||||
|
let text = "ab→cd"; // '→' = bytes [2,5)
|
||||||
|
assert_eq!(floor_char_boundary(text, 0), 0);
|
||||||
|
assert_eq!(floor_char_boundary(text, 2), 2);
|
||||||
|
assert_eq!(floor_char_boundary(text, 3), 2); // inside '→' → floor to 2
|
||||||
|
assert_eq!(floor_char_boundary(text, 4), 2);
|
||||||
|
assert_eq!(floor_char_boundary(text, 5), 5);
|
||||||
|
assert_eq!(floor_char_boundary(text, 99), text.len());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn projected_rich_chunks_inserts_at_offset_without_source_bytes() {
|
fn projected_rich_chunks_inserts_at_offset_without_source_bytes() {
|
||||||
let chunks = projected_rich_chunks(
|
let chunks = projected_rich_chunks(
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue