regex-search: smart-case multi-line find_all_regex (Q#RX1/RX2)

The regex sibling of find_all, on regex::bytes::Regex over the whole
buffer. Returns Option<Vec<ByteRange>>: Some for a valid pattern
(possibly empty), None when it fails to compile — so the caller can
tell an invalid pattern (show [invalid]) from a valid zero-match
search.

Smart-case mirrors the literal path: case-insensitive via a (?i)
prefix unless the pattern carries an uppercase letter. Multi-line is
free — the regex runs over the whole byte slice, so an explicit \n (or
(?s).) spans lines while `.` keeps its default. Zero-width matches
(a*, ^, $) are filtered. The regex crate (already transitive in the
lockfile) is promoted to a direct dependency; its linear-time engine
makes a pathological pattern slow at worst, never catastrophic.

Tests: pattern match, smart-case both ways, \n-spanning + dotall +
default-no-cross, invalid→None vs valid-zero→Some(empty), zero-width
filter.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Levi Neuwirth 2026-06-27 14:22:51 -04:00
parent e03f5620c9
commit 5c34b3c37a
3 changed files with 103 additions and 0 deletions

1
Cargo.lock generated
View File

@ -2434,6 +2434,7 @@ dependencies = [
"proptest",
"rand 0.8.6",
"rand_distr",
"regex",
"rmp-serde",
"semver",
"serde",

View File

@ -78,6 +78,10 @@ crdt = ["dep:loro", "pmacs-protocol/crdt"]
crossterm = "0.28"
thiserror = { workspace = true }
unicode-width = "0.2"
# Regex engine for in-buffer regex search (Q#RX1). `regex::bytes::Regex`
# matches over rope-snapshot bytes and yields byte offsets directly.
# Already in the lockfile transitively; promoted to a direct dependency.
regex = "1"
# Work-stealing deque, MPMC channels, and parking primitives for the
# M3 worker pool (spec §6.3). The umbrella crate re-exports
# `crossbeam_deque`, `crossbeam_channel`, and `crossbeam_utils`.

View File

@ -228,6 +228,51 @@ pub fn find_all(haystack: &[u8], query: &str) -> Vec<ByteRange> {
out
}
/// Smart-case regex search over `haystack` bytes for `pattern` (Q#RX1).
///
/// Returns `Some(matches)` — leftmost, non-overlapping, ascending byte
/// ranges — for a valid pattern (possibly empty), or `None` when
/// `pattern` fails to compile. The `Option` lets the caller tell an
/// *invalid* pattern (show `[invalid]`) apart from a valid one with no
/// matches (a failing search), which a flat `Vec` could not.
///
/// **Smart-case (Q#RX2).** Case-insensitive unless `pattern` contains
/// an uppercase letter, applied by compiling `(?i)` ahead of the
/// pattern. An uppercase letter inside an escape or class (`\D`,
/// `[A-Z]`) trips case-sensitivity — the same coarse rule the literal
/// path uses.
///
/// **Multi-line (Q#RX1)** falls out for free: the regex runs over the
/// whole byte slice, so an explicit `\n` (or `(?s).`) spans lines. `.`
/// keeps its default of not matching `\n`.
///
/// **Zero-width matches** (`a*`, `^`, `$`, anchors) are filtered — they
/// wash nothing and would otherwise flood the match list. The `regex`
/// crate's linear-time engine makes a pathological pattern slow at
/// worst, never catastrophic.
#[must_use]
pub fn find_all_regex(haystack: &[u8], pattern: &str) -> Option<Vec<ByteRange>> {
if pattern.is_empty() {
return Some(Vec::new());
}
let case_insensitive = !pattern.chars().any(char::is_uppercase);
let re = if case_insensitive {
regex::bytes::Regex::new(&format!("(?i){pattern}"))
} else {
regex::bytes::Regex::new(pattern)
}
.ok()?;
let matches = re
.find_iter(haystack)
.filter(|m| m.end() > m.start())
.map(|m| ByteRange {
start: m.start() as u64,
end: m.end() as u64,
})
.collect();
Some(matches)
}
// ---------------------------------------------------------------------------
// TUI view
// ---------------------------------------------------------------------------
@ -401,6 +446,59 @@ mod tests {
assert!(find_all(b"", "x").is_empty());
}
// ---- regex matcher (Q#RX1) -----------------------------------------
#[test]
fn find_all_regex_matches_pattern() {
// \d+ over "a1 bb 23 c" → "1" and "23".
assert_eq!(
find_all_regex(b"a1 bb 23 c", r"\d+"),
Some(vec![r(1, 2), r(6, 8)])
);
}
#[test]
fn find_all_regex_is_smart_case() {
// Lowercase pattern folds case (matches "Foo", "foo", "FOO").
assert_eq!(
find_all_regex(b"Foo foo FOO", "foo"),
Some(vec![r(0, 3), r(4, 7), r(8, 11)])
);
// An uppercase letter in the pattern makes it case-sensitive.
assert_eq!(find_all_regex(b"Foo foo FOO", "Foo"), Some(vec![r(0, 3)]));
}
#[test]
fn find_all_regex_spans_newlines() {
// An explicit \n in the pattern matches across the line break.
assert_eq!(
find_all_regex(b"foo\n bar", r"foo\n\s*bar"),
Some(vec![r(0, 9)])
);
// Plain `.` does NOT cross the newline (default, not dotall).
assert_eq!(find_all_regex(b"a\nb", "a.b"), Some(vec![]));
// ...but `(?s)` opts into dotall.
assert_eq!(find_all_regex(b"a\nb", "(?s)a.b"), Some(vec![r(0, 3)]));
}
#[test]
fn find_all_regex_invalid_pattern_is_none() {
// Unbalanced group — the incremental-typing case (`foo(`).
assert_eq!(find_all_regex(b"foo(", "foo("), None);
// A valid pattern with zero matches is Some(empty), distinct
// from invalid.
assert_eq!(find_all_regex(b"abc", "zzz"), Some(vec![]));
}
#[test]
fn find_all_regex_filters_zero_width_and_empty() {
// `a*` matches empty at non-'a' positions; only the non-empty
// runs survive the zero-width filter.
assert_eq!(find_all_regex(b"baab", "a*"), Some(vec![r(1, 3)]));
// An empty pattern yields no matches (not one-per-position).
assert_eq!(find_all_regex(b"abc", ""), Some(vec![]));
}
#[test]
fn store_set_clamps_active_and_clears_on_empty() {
let mut s = SearchStore::new();