hjkl_engine/search.rs
1//! Engine-owned search state + execution helpers.
2//!
3//! Patch 0.0.35 step 1 of the 33-method classification rollout
4//! (see `DESIGN_33_METHOD_CLASSIFICATION.md`). The pattern, per-row
5//! match cache, and `wrapscan` flag previously lived on
6//! [`hjkl_buffer::View`] (private `SearchState`). Moving the FSM
7//! state out of the buffer keeps multi-window hosts from sharing the
8//! "current search" across panes that happen to share content.
9//!
10//! The buffer keeps `Search::find_next` / `Search::find_prev` (the
11//! SPEC trait surface — pure observers, caller owns the regex). This
12//! module composes those primitives with the Editor-owned
13//! [`SearchState`] to drive `n` / `N` / `*` / `#` / `/` / `?`.
14//!
15//! 0.0.37: the buffer-inherent `search_forward` / `search_backward`
16//! / `search_matches` / `set_search_pattern` / `search_pattern` /
17//! `set_search_wrap` / `search_wraps` accessors are removed. Search
18//! state lives on `Editor::search_state`, the rendering path
19//! (`BufferView`) takes the active `&Regex` as a parameter, and the
20//! `Search` trait impl always wraps (engine controls non-wrap
21//! semantics).
22
23use regex::Regex;
24
25use crate::types::{Cursor, Query, Search};
26use hjkl_vim_types::Operator;
27
28/// Active `/` or `?` search prompt. Text mutations drive the textarea's
29/// live search pattern so matches highlight as the user types.
30#[derive(Debug, Clone)]
31pub struct SearchPrompt {
32 pub text: String,
33 pub cursor: usize,
34 pub forward: bool,
35 /// Operator-pending search (`d/pat`, `c/pat`, `y/pat`): the operator, its
36 /// count, and the cursor position where the operator started. `None` for a
37 /// plain `/` / `?` search. On commit the operator runs over the (exclusive,
38 /// charwise) range from `origin` to the match.
39 pub operator: Option<(Operator, usize, (usize, usize))>,
40}
41
42/// Case-sensitivity policy derived from `:set ignorecase` / `:set smartcase`.
43///
44/// Use [`CaseMode::from_options`] to build from two booleans, then pass to
45/// [`resolve_case_mode`] together with the raw pattern string.
46#[derive(Debug, Clone, Copy, PartialEq, Eq)]
47pub enum CaseMode {
48 /// Always case-sensitive regardless of the pattern.
49 Sensitive,
50 /// Always case-insensitive regardless of the pattern.
51 Insensitive,
52 /// Case-insensitive unless the pattern contains an uppercase rune
53 /// (vim's `smartcase` behaviour).
54 Smart,
55}
56
57impl CaseMode {
58 /// Build a `CaseMode` from the two option booleans.
59 ///
60 /// | `ignorecase` | `smartcase` | Result |
61 /// |---|---|---|
62 /// | `false` | `*` | `Sensitive` |
63 /// | `true` | `false` | `Insensitive` |
64 /// | `true` | `true` | `Smart` |
65 pub fn from_options(ignorecase: bool, smartcase: bool) -> Self {
66 if !ignorecase {
67 Self::Sensitive
68 } else if smartcase {
69 Self::Smart
70 } else {
71 Self::Insensitive
72 }
73 }
74}
75
76/// Vim's regex "magic" level — controls which characters are special
77/// (regex metacharacters) without a backslash prefix. See `:help magic`.
78///
79/// Ordering (most → least magic): `VeryMagic > Magic > NoMagic > VeryNoMagic`.
80/// A character's inherent level determines its behavior: it is special
81/// unescaped when the current level is *at or above* its inherent level, and
82/// backslash toggles that (forces the opposite treatment).
83#[derive(Debug, Clone, Copy, PartialEq, Eq)]
84enum MagicLevel {
85 /// `\v` — nearly every non-alnum/underscore ASCII character is special
86 /// unescaped (groups, quantifiers, alternation, anchors, boundaries).
87 VeryMagic,
88 /// Default / `\m` — vim's normal mode: `. * [ ] ~` are magic unescaped;
89 /// groups/quantifiers/alternation/boundaries need a backslash.
90 Magic,
91 /// `\M` — only `^ $` are magic unescaped; everything else (including
92 /// `. * [ ]`) is literal unless backslashed.
93 NoMagic,
94 /// `\V` — only `\` is special; every other character is literal unless
95 /// backslashed (mirrors `Magic`'s "very magic" meta chars).
96 VeryNoMagic,
97}
98
99/// Characters whose inherent magic level is "very magic" (`( ) + ? | { } = < >`).
100fn very_magic_special(ch: char) -> bool {
101 matches!(
102 ch,
103 '(' | ')' | '+' | '?' | '|' | '{' | '}' | '=' | '<' | '>'
104 )
105}
106
107/// Characters whose inherent magic level is "magic" (`. * [ ] ~`).
108fn magic_special(ch: char) -> bool {
109 matches!(ch, '.' | '*' | '[' | ']' | '~')
110}
111
112/// Characters whose inherent magic level is "nomagic" (`^ $`).
113fn nomagic_special(ch: char) -> bool {
114 matches!(ch, '^' | '$')
115}
116
117/// `true` when `ch` is a rust-`regex` metacharacter that must be
118/// backslash-escaped to appear as a literal.
119fn regex_meta(ch: char) -> bool {
120 matches!(
121 ch,
122 '\\' | '.' | '+' | '*' | '?' | '(' | ')' | '|' | '[' | ']' | '{' | '}' | '^' | '$'
123 )
124}
125
126/// Whether `ch` is special-without-a-backslash at the given magic `level`.
127fn is_special_unescaped(ch: char, level: MagicLevel) -> bool {
128 if very_magic_special(ch) {
129 level == MagicLevel::VeryMagic
130 } else if magic_special(ch) {
131 matches!(level, MagicLevel::VeryMagic | MagicLevel::Magic)
132 } else if nomagic_special(ch) {
133 level != MagicLevel::VeryNoMagic
134 } else {
135 false
136 }
137}
138
139/// Emit `ch`'s regex-special meaning into `out`. `chars` is consumed further
140/// only for `{` (counted-repeat body) and `[` (character class) is handled by
141/// the caller since it needs to flip a "bracket mode" flag.
142///
143/// `last_sub` is the previous `:s` replacement string, used to expand the
144/// magic `~` (`:h /~`, `:h s/~`).
145fn emit_special(
146 out: &mut String,
147 ch: char,
148 chars: &mut std::iter::Peekable<std::str::Chars>,
149 last_sub: &str,
150) {
151 match ch {
152 '(' => out.push('('),
153 ')' => out.push(')'),
154 '+' => out.push('+'),
155 '?' => out.push('?'),
156 '=' => out.push('?'), // vim `\=` / very-magic `=` — same as `\?`.
157 '|' => out.push('|'),
158 '<' | '>' => out.push_str(r"\b"),
159 '{' => {
160 out.push('{');
161 emit_counted_repeat(out, chars);
162 }
163 '}' => out.push('}'), // stray close — harmless as a literal.
164 '.' => out.push('.'),
165 '*' => out.push('*'),
166 ']' => out.push_str(r"\]"), // stray close — harmless as a literal.
167 // Magic `~` — expands to the previous `:s` replacement text
168 // (`:h /~`, `:h s/~`). Inserted verbatim into the translated
169 // (rust-regex) output: like vim, the text is dropped in "as pattern"
170 // without re-escaping. Ordinary word replacements (`BAR`) round-trip
171 // exactly; a replacement carrying regex metacharacters or vim
172 // replacement escapes (`\1`, `&`, `\u`…) is a documented
173 // sub-limitation and may not compile. Empty `last_sub` (no prior
174 // `:s`) → empty expansion (see `translate_pattern`).
175 '~' => out.push_str(last_sub),
176 '^' => out.push('^'),
177 '$' => out.push('$'),
178 _ => out.push(ch),
179 }
180}
181
182/// Copy a `\{n,m}` / `{n,m}` counted-repeat body through to `out`, closing on
183/// either a bare `}` (vim's permissive default-magic form, `\{n,m}`) or an
184/// escaped `\}`. Assumes the opening `{` has already been pushed to `out`.
185fn emit_counted_repeat(out: &mut String, chars: &mut std::iter::Peekable<std::str::Chars>) {
186 loop {
187 match chars.next() {
188 Some('\\') => {
189 if chars.peek() == Some(&'}') {
190 chars.next();
191 out.push('}');
192 return;
193 } else if let Some(c2) = chars.next() {
194 out.push(c2);
195 } else {
196 return;
197 }
198 }
199 Some('}') => {
200 out.push('}');
201 return;
202 }
203 Some(c2) => out.push(c2),
204 None => return,
205 }
206 }
207}
208
209/// Emit `ch` as a literal character, escaping it if it happens to be a rust
210/// `regex` metacharacter.
211fn emit_literal(out: &mut String, ch: char) {
212 if regex_meta(ch) {
213 out.push('\\');
214 }
215 out.push(ch);
216}
217
218/// Translate a raw vim pattern into rust-`regex` syntax and extract any
219/// `\c`/`\C` case override. This is the core of [`resolve_case_mode`].
220///
221/// Handles vim's default-magic transforms (`\( \) \+ \? \= \|` → group /
222/// quantifier / alternation syntax; the inverse — unescaped `( ) + ? | { }`
223/// become literals), the `\<` / `\>` word-boundary rewrite (already
224/// magic-level-independent), `\{n,m}` counted repeats (including vim's
225/// permissive unescaped-closing-brace form), and the `\v` / `\V` / `\m` /
226/// `\M` magic-level mode switches (mid-pattern, not just at the start).
227///
228/// `\1`-`\9` backreferences in the PATTERN (not the replacement) are not
229/// supported by the rust `regex` crate (no backtracking engine) — they pass
230/// through unchanged, which either fails to compile or fails to match,
231/// preserving the pre-fix "silent no-match" behavior rather than corrupting
232/// text.
233///
234/// A simple bracket-depth flag skips translation inside `[...]` character
235/// classes, mirroring how vim (and rust-regex) treat class contents mostly
236/// literally. A `~` inside `[...]` is therefore a literal class member (as in
237/// vim), never a last-substitute expansion.
238///
239/// ### Magic `~` (last-substitute expansion)
240///
241/// `last_sub` is the previous `:s` replacement string. Under default magic a
242/// bare `~` expands to it (`\~` stays a literal tilde); under `\M`/`\V` the
243/// roles swap (`\~` expands, bare `~` is literal) — both fall out of the
244/// existing symmetric magic-level logic since `~` is a "magic"-inherent char.
245/// The expansion is inserted verbatim into the rust-regex output (see
246/// [`emit_special`]). When `last_sub` is empty (no `:s` has run yet) the
247/// expansion is empty rather than an error — nvim raises `E33` here, but the
248/// empty-string choice is safe (never corrupts the buffer) and matches this
249/// repo's "silent no-op over hard error" search convention.
250fn translate_pattern(pat: &str, last_sub: &str) -> (String, Option<bool>) {
251 let mut out = String::with_capacity(pat.len());
252 let mut level = MagicLevel::Magic;
253 let mut override_mode: Option<bool> = None;
254 let mut chars = pat.chars().peekable();
255 let mut in_bracket = false;
256 // Parity of the run of backslashes immediately preceding the current
257 // in-bracket char: `\]` is an ESCAPED literal `]` member (vim
258 // `[a\]b]` = {a, ], b}), so a `]` closes the class only when the run is
259 // even; `\\]` closes. Reset on any non-backslash.
260 let mut bracket_backslash_parity = false;
261
262 while let Some(ch) = chars.next() {
263 if in_bracket {
264 // `\a` / `\A` inside a class: rust-regex would read `\a` as Bell
265 // and reject `\A`; vim reads them as the alphabetic class, so
266 // emit the range directly. `\A` → `^A-Za-z` is only correct as
267 // the class's FIRST element (vim treats it as a literal member
268 // otherwise) — best approximation.
269 if ch == '\\' && matches!(chars.peek(), Some('a') | Some('A')) {
270 let c = chars.next().unwrap();
271 out.push_str(if c == 'a' { "A-Za-z" } else { "^A-Za-z" });
272 // The consumed pair ends in a non-backslash.
273 bracket_backslash_parity = false;
274 continue;
275 }
276 if ch == '\\' {
277 bracket_backslash_parity = !bracket_backslash_parity;
278 out.push(ch);
279 continue;
280 }
281 if ch == ']' && !bracket_backslash_parity {
282 in_bracket = false;
283 }
284 bracket_backslash_parity = false;
285 out.push(ch);
286 continue;
287 }
288
289 if ch == '\\' {
290 match chars.next() {
291 Some('c') => override_mode = Some(true), // \c → insensitive
292 Some('C') => override_mode = Some(false), // \C → sensitive
293 // vim `\Z` — ignore case for the rest of the pattern,
294 // identical to `\c` (`:h /\Z`).
295 Some('Z') => override_mode = Some(true),
296 // vim `\a` = `[A-Za-z]` (rust-regex `\a` is Bell) and
297 // `\A` = `[^A-Za-z]` (rust-regex `\A` is a start-of-text
298 // anchor). `\e` = ESC (rust-regex has no `\e`).
299 Some('a') => out.push_str("[A-Za-z]"),
300 Some('A') => out.push_str("[^A-Za-z]"),
301 Some('e') => out.push('\u{001b}'),
302 Some('v') => level = MagicLevel::VeryMagic,
303 Some('V') => level = MagicLevel::VeryNoMagic,
304 Some('m') => level = MagicLevel::Magic,
305 Some('M') => level = MagicLevel::NoMagic,
306 Some(d @ '0'..='9') => {
307 // Backreference — unsupported by rust-regex. Pass through
308 // unchanged (keeps prior no-match/error behavior).
309 out.push('\\');
310 out.push(d);
311 }
312 Some(c2) if very_magic_special(c2) || magic_special(c2) || nomagic_special(c2) => {
313 if is_special_unescaped(c2, level) {
314 // Already special unescaped at this level — backslash
315 // forces the literal reading.
316 emit_literal(&mut out, c2);
317 } else if c2 == '[' {
318 out.push('[');
319 in_bracket = true;
320 } else {
321 emit_special(&mut out, c2, &mut chars, last_sub);
322 }
323 }
324 Some(other) => {
325 // \d \s \w \b \B \n \t \r \& \~ \\ etc. — already
326 // valid rust-regex syntax (or handled by the caller) and
327 // identical in vim's default magic. Pass through.
328 out.push('\\');
329 out.push(other);
330 }
331 None => out.push('\\'),
332 }
333 continue;
334 }
335
336 if is_special_unescaped(ch, level) {
337 if ch == '[' {
338 out.push('[');
339 in_bracket = true;
340 } else {
341 emit_special(&mut out, ch, &mut chars, last_sub);
342 }
343 } else {
344 emit_literal(&mut out, ch);
345 }
346 }
347
348 (out, override_mode)
349}
350
351/// Strip `\c` / `\C` overrides from `pat`, resolve the effective
352/// [`CaseMode`], and return the cleaned pattern together with the
353/// resolved mode.
354///
355/// ### Override rules (mirrors vim)
356///
357/// - `\c` anywhere in `pat` forces case-insensitive.
358/// - `\C` anywhere in `pat` forces case-sensitive.
359/// - When both appear the **last** one wins.
360/// - Both are stripped from the returned pattern.
361///
362/// ### Magic-mode translation
363///
364/// As of the default-magic regex fix, this function also translates vim's
365/// default-magic (and `\v`/`\V`/`\m`/`\M`-switched) regex syntax into
366/// rust-`regex` syntax — see [`translate_pattern`] for the full transform
367/// list. `vim_to_rust_regex` is a thin wrapper that discards the case mode.
368///
369/// ### Smart-case detection
370///
371/// When `base` is [`CaseMode::Smart`] and no `\c`/`\C` override was
372/// found, the pattern is scanned for uppercase Unicode letters. Any
373/// uppercase letter → `Sensitive`; otherwise → `Insensitive`.
374///
375/// ### Per-substitute flag interaction
376///
377/// The `:s/…/…/i` and `:s/…/…/I` flags are handled in
378/// `apply_substitute` **before** calling this function (they
379/// short-circuit entirely). This function is not involved.
380///
381/// ### Magic `~` expansion
382///
383/// `last_sub` is the previous `:s` replacement string (pass `""` when there is
384/// no substitute context, e.g. `*`/`#` word search). Callers get it from
385/// [`crate::editor::Editor::last_substitute_replacement`]. See
386/// [`translate_pattern`] for the expansion + escaping rules.
387pub fn resolve_case_mode(pat: &str, base: CaseMode, last_sub: &str) -> (String, CaseMode) {
388 let (out, override_mode) = translate_pattern(pat, last_sub);
389
390 let resolved = match override_mode {
391 Some(true) => CaseMode::Insensitive,
392 Some(false) => CaseMode::Sensitive,
393 None => match base {
394 CaseMode::Smart => {
395 // Any uppercase rune → sensitive. Scan the TRANSLATED
396 // pattern so control sequences consumed during translation
397 // (`\c` `\C` `\v` `\V` `\m` `\M`) don't spuriously count —
398 // matches the pre-existing behavior this function had before
399 // magic-mode translation was added.
400 if out.chars().any(|c| c.is_uppercase()) {
401 CaseMode::Sensitive
402 } else {
403 CaseMode::Insensitive
404 }
405 }
406 other => other,
407 },
408 };
409
410 (out, resolved)
411}
412
413/// Rewrite vim-style word-boundary escapes to Rust `regex`-compatible form
414/// **and** strip `\c`/`\C` case overrides.
415///
416/// The `regex` crate supports `\b` (symmetric word boundary) but not the
417/// vim/PCRE `\<` (word-boundary start) or `\>` (word-boundary end) variants.
418/// This function performs a single-pass rewrite:
419///
420/// - `\<` → `\b`
421/// - `\>` → `\b`
422/// - `\c` / `\C` stripped (case override — handled by [`resolve_case_mode`])
423/// - `\\<` / `\\>` (literal double-backslash followed by `<`/`>`) are left
424/// untouched — only the unescaped form transforms.
425/// - All other syntax (`\b`, `\B`, `\d`, anchors, …) passes through unchanged.
426///
427/// Call this on the raw user-typed pattern string **before** passing to
428/// `regex::Regex::new`. Keep the original string for display / history.
429///
430/// Prefer [`resolve_case_mode`] when you also need to apply case semantics;
431/// that function performs the same boundary rewrite internally.
432///
433/// This thin wrapper passes an empty last-substitute string, so a magic `~`
434/// expands to the empty string. Use [`resolve_case_mode`] directly with the
435/// editor's last-substitute replacement when `~` expansion matters.
436pub fn vim_to_rust_regex(pat: &str) -> String {
437 resolve_case_mode(pat, CaseMode::Sensitive, "").0
438}
439
440/// Per-row match cache keyed against the buffer's `dirty_gen`. Live
441/// alongside the active pattern so re-running `n` doesn't re-scan
442/// rows the buffer hasn't touched.
443#[derive(Debug, Clone, Default)]
444pub struct SearchState {
445 /// Active pattern, if any. `None` clears highlighting and makes
446 /// `n` / `N` no-op until the next `/` / `?` commit.
447 pub pattern: Option<Regex>,
448 /// `true` for `/`, `false` for `?` — drives `n` vs `N` direction.
449 /// Mirrors `vim.last_search_forward`; consolidated so future
450 /// patches can drop the duplicate.
451 pub forward: bool,
452 /// `matches[row]` is the `(byte_start, byte_end)` runs cached on
453 /// `row`, captured at `gen[row]`. Length grows lazily.
454 pub matches: Vec<Vec<(usize, usize)>>,
455 /// Per-row generation tag. When the buffer's `dirty_gen` for a
456 /// row diverges, the row gets re-scanned on next access.
457 pub generations: Vec<u64>,
458 /// Wrap past buffer ends. Mirrors `Settings::wrapscan`.
459 pub wrap_around: bool,
460}
461
462impl SearchState {
463 /// Empty state — no pattern, forward direction, wraps.
464 pub fn new() -> Self {
465 Self {
466 pattern: None,
467 forward: true,
468 matches: Vec::new(),
469 generations: Vec::new(),
470 wrap_around: true,
471 }
472 }
473
474 /// Replace the active pattern. Drops the cached match runs so
475 /// the next access re-scans against the new regex.
476 pub fn set_pattern(&mut self, re: Option<Regex>) {
477 self.pattern = re;
478 self.matches.clear();
479 self.generations.clear();
480 }
481
482 /// Refresh `matches[row]` if either the row's gen has rolled or
483 /// we never scanned it. Returns the cached slice.
484 ///
485 /// `get_line` is materialized lazily — only invoked on a cache
486 /// miss (never scanned, or the row's gen rolled). A steady-state
487 /// warm cache returns the cached runs without allocating the line.
488 pub fn matches_for(
489 &mut self,
490 row: usize,
491 dirty_gen: u64,
492 get_line: impl FnOnce() -> String,
493 ) -> &[(usize, usize)] {
494 let Some(ref re) = self.pattern else {
495 return &[];
496 };
497 if self.matches.len() <= row {
498 self.matches.resize_with(row + 1, Vec::new);
499 self.generations.resize(row + 1, u64::MAX);
500 }
501 if self.generations[row] != dirty_gen {
502 // Shared scanner (`hjkl_buffer::search_match_ranges`) — the same
503 // byte-range computation the hlsearch painter and the quickfix
504 // dock's match overlay use, so navigation and highlighting can
505 // never disagree about where a match is.
506 self.matches[row] = hjkl_buffer::search_match_ranges(re, &get_line());
507 self.generations[row] = dirty_gen;
508 }
509 &self.matches[row]
510 }
511}
512
513/// Move the cursor to the next match starting from (or just after,
514/// when `skip_current = true`) the cursor. Wraps end-of-buffer to
515/// row 0 when `state.wrap_around`. Returns `true` when a match was
516/// found.
517///
518/// Pure observe + cursor mutation — no auto-scroll. The Editor's
519/// post-step `ensure_cursor_in_scrolloff` reapplies viewport
520/// follow.
521pub fn search_forward<B: Cursor + Query + Search>(
522 buf: &mut B,
523 state: &mut SearchState,
524 skip_current: bool,
525) -> bool {
526 let Some(re) = state.pattern.clone() else {
527 return false;
528 };
529 let cursor = buf.cursor();
530 let total = buf.line_count();
531 if total == 0 {
532 return false;
533 }
534 // To "skip the current cell", advance `from` one char past the
535 // cursor before asking `find_next` for the at-or-after match.
536 // `pos_at_byte` rounds a mid-char byte DOWN to the enclosing
537 // char's start, so stepping a single byte from the first byte of
538 // a multi-byte char lands back on the cursor itself and `n` never
539 // advances. Step by the full char width instead; when the cursor
540 // sits past end-of-line (no char there), fall back to one byte —
541 // `pos_at_byte` clamps overflow to end-of-buffer so this is safe
542 // even when the cursor sits at the trailing edge.
543 let from = if skip_current {
544 let from_byte = buf.byte_offset(cursor);
545 let width = buf
546 .line(cursor.line)
547 .chars()
548 .nth(cursor.col as usize)
549 .map_or(1, char::len_utf8);
550 buf.pos_at_byte(from_byte.saturating_add(width))
551 } else {
552 cursor
553 };
554 if let Some(range) = buf.find_next(from, &re) {
555 // Honour engine wrap policy explicitly. The buffer impl uses
556 // its own (deprecated) wrap flag; for new search state the
557 // engine SearchState is the source of truth.
558 if !state.wrap_around && range.start.line < cursor.line {
559 return false;
560 }
561 Cursor::set_cursor(buf, range.start);
562 return true;
563 }
564 false
565}
566
567/// Symmetric counterpart of [`search_forward`].
568pub fn search_backward<B: Cursor + Query + Search>(
569 buf: &mut B,
570 state: &mut SearchState,
571 skip_current: bool,
572) -> bool {
573 let Some(re) = state.pattern.clone() else {
574 return false;
575 };
576 let cursor = buf.cursor();
577 let total = buf.line_count();
578 if total == 0 {
579 return false;
580 }
581 // View's `Search::find_prev` returns the at-or-before match
582 // for the anchor `from`. For `skip_current`, we want the
583 // rightmost match whose start is *strictly before* the cursor.
584 // Strategy: query find_prev(cursor); if the returned match
585 // covers/starts-at the cursor, step the anchor back one byte
586 // past that match's start and re-query so the next find_prev
587 // skips it. Otherwise the at-or-before match is already strictly
588 // before the cursor and we accept it.
589 let initial = buf.find_prev(cursor, &re);
590 let range = if skip_current {
591 match initial {
592 Some(m) if m.start == cursor => {
593 // Cursor sits exactly on a match start (typical post-
594 // commit state). Step past and re-query.
595 let cb = buf.byte_offset(m.start);
596 if cb == 0 {
597 // Current match starts at buffer byte 0 — there is
598 // nothing earlier to step back to. Wrap to the
599 // buffer's last match instead; when the current
600 // match is the only one, `find_prev` from
601 // end-of-buffer returns it again and the cursor
602 // stays (matches vim).
603 let end = buf.pos_at_byte(buf.len_bytes());
604 buf.find_prev(end, &re)
605 } else {
606 let anchor = buf.pos_at_byte(cb.saturating_sub(1));
607 buf.find_prev(anchor, &re)
608 }
609 }
610 other => other,
611 }
612 } else {
613 initial
614 };
615 if let Some(range) = range {
616 if !state.wrap_around && range.start.line > cursor.line {
617 return false;
618 }
619 Cursor::set_cursor(buf, range.start);
620 return true;
621 }
622 false
623}
624
625/// Match positions on `row` as `(byte_start, byte_end)`. Used by
626/// the engine's highlight pipeline. Reads through the cache so a
627/// steady-state buffer doesn't re-scan every frame.
628///
629/// Returns a borrow of the per-row cache (no per-call `Vec` clone).
630pub fn search_matches<'a, B: Query>(
631 buf: &B,
632 state: &'a mut SearchState,
633 dirty_gen: u64,
634 row: usize,
635) -> &'a [(usize, usize)] {
636 if state.pattern.is_none() {
637 return &[];
638 }
639 let line_count = buf.line_count() as usize;
640 if row >= line_count {
641 return &[];
642 }
643 // Materialize the line lazily — only when the cache misses. A warm
644 // steady-state cache skips the per-row allocation entirely.
645 state.matches_for(row, dirty_gen, || buf.line(row as u32))
646}
647
648/// Warm the per-row match cache for `row` without producing a result.
649///
650/// Same cache path as [`search_matches`] (identical miss/hit behaviour),
651/// but returns nothing — the renderer's pre-pass only wants the cache
652/// populated, and building a `Vec` per visible row just to drop it is
653/// pure allocation churn.
654pub fn warm_matches<B: Query>(buf: &B, state: &mut SearchState, dirty_gen: u64, row: usize) {
655 if state.pattern.is_none() {
656 return;
657 }
658 let line_count = buf.line_count() as usize;
659 if row >= line_count {
660 return;
661 }
662 state.matches_for(row, dirty_gen, || buf.line(row as u32));
663}
664
665#[cfg(test)]
666mod tests {
667 use super::*;
668 use crate::types::Pos;
669 use hjkl_buffer::View;
670
671 fn re(pat: &str) -> Regex {
672 Regex::new(pat).unwrap()
673 }
674
675 fn vim_re(pat: &str) -> Regex {
676 Regex::new(&vim_to_rust_regex(pat)).unwrap()
677 }
678
679 // ── vim_to_rust_regex unit tests ─────────────────────────────────────────
680
681 /// `\<` and `\>` both rewrite to `\b`.
682 #[test]
683 fn vim_boundary_rewrites_to_b() {
684 assert_eq!(vim_to_rust_regex(r"\<foo\>"), r"\bfoo\b");
685 assert_eq!(vim_to_rust_regex(r"\<"), r"\b");
686 assert_eq!(vim_to_rust_regex(r"\>"), r"\b");
687 }
688
689 /// A literal double-backslash before `<`/`>` must not be consumed.
690 /// `\\<` in the source string is two chars: `\` `\`; the rewriter sees
691 /// the first `\` followed by `\`, emits `\\`, then `<` is plain text.
692 #[test]
693 fn escaped_backslash_left_alone() {
694 // Input: \\< (three chars in source: '\', '\', '<')
695 // Expected output: \\< (the first \ escapes the second, < is literal)
696 let input = r"\\<";
697 let output = vim_to_rust_regex(input);
698 assert_eq!(output, r"\\<");
699 }
700
701 /// Other escape sequences (`\b`, `\B`, `\d`, `\w`, anchors) pass through.
702 #[test]
703 fn other_escapes_unchanged() {
704 assert_eq!(vim_to_rust_regex(r"\b"), r"\b");
705 assert_eq!(vim_to_rust_regex(r"\B"), r"\B");
706 // vim default magic: `+` is a literal unless backslashed. `\d\+`
707 // (digit class, one-or-more quantifier) translates to `\d\+` in
708 // rust-regex syntax (identical spelling — `\+` IS rust-regex's own
709 // escaped-literal-plus, but since the quantifier here is coming from
710 // vim's `\+` we want the rust-regex QUANTIFIER `+`, unescaped).
711 assert_eq!(vim_to_rust_regex(r"\d\+"), r"\d+");
712 assert_eq!(vim_to_rust_regex(r"^\w\+$"), r"^\w+$");
713 }
714
715 /// Mixed: `\<\w\+\>` rewrites to `\b\w+\b` — matches whole words.
716 #[test]
717 fn mixed_boundary_and_word_class() {
718 assert_eq!(vim_to_rust_regex(r"\<\w\+\>"), r"\b\w+\b");
719 }
720
721 // ── Integration: compiled vim patterns match correctly ───────────────────
722
723 /// `/foo\<bar\>` — `bar` as a standalone word is matched, `foobar` is not.
724 #[test]
725 fn vim_boundary_matches_standalone_word_not_suffix() {
726 let re = vim_re(r"foo\<bar\>");
727 // "foobar" — `bar` follows directly after `foo` with no word boundary:
728 // the `\b` between `foo` and `bar` fails here.
729 assert!(!re.is_match("foobar"));
730 // "foo bar" — word boundary between `foo ` and `bar`:
731 // pattern `foo\bbar\b` does not match because `foo` is not adjacent.
732 // Use a pattern that directly tests the intent: `bar` as a whole word.
733 let re2 = vim_re(r"\<bar\>");
734 assert!(re2.is_match("foo bar baz"));
735 assert!(!re2.is_match("foobar"));
736 }
737
738 /// `\<word` matches `word` at start-of-word but not mid-word.
739 #[test]
740 fn vim_boundary_start_only() {
741 let re = vim_re(r"\<word");
742 assert!(re.is_match("word here"));
743 assert!(re.is_match("some word here"));
744 assert!(!re.is_match("sword"));
745 assert!(!re.is_match("aword"));
746 }
747
748 /// `word\>` matches `word` at end-of-word but not when followed by more.
749 #[test]
750 fn vim_boundary_end_only() {
751 let re = vim_re(r"word\>");
752 assert!(re.is_match("some word"));
753 assert!(re.is_match("word"));
754 assert!(!re.is_match("words"));
755 assert!(!re.is_match("wordsmith"));
756 }
757
758 /// Existing `\b` continues to work (sanity check — no double-transform).
759 #[test]
760 fn existing_b_boundary_unchanged() {
761 let re = vim_re(r"\bfoo\b");
762 assert!(re.is_match("foo"));
763 assert!(re.is_match("a foo b"));
764 assert!(!re.is_match("foobar"));
765 assert!(!re.is_match("afoo"));
766 }
767
768 /// Mixed: `\<\w+\>` matches whole words only.
769 #[test]
770 fn vim_whole_word_pattern() {
771 let re = vim_re(r"\<\w\+\>");
772 let matches: Vec<_> = re.find_iter("foo bar baz").map(|m| m.as_str()).collect();
773 assert_eq!(matches, vec!["foo", "bar", "baz"]);
774 }
775
776 #[test]
777 fn empty_state_no_match() {
778 let mut b = View::from_str("anything");
779 let mut s = SearchState::new();
780 assert!(!search_forward(&mut b, &mut s, false));
781 assert!(!search_backward(&mut b, &mut s, false));
782 }
783
784 // ── B8/B9: default-magic + \v/\V/\m/\M translation ───────────────────────
785
786 #[test]
787 fn default_magic_groups_and_backref_replacement_side() {
788 // \( \) → real groups; the PATTERN side is exercised end-to-end via
789 // substitute.rs (replacement-side \1 already worked before this fix).
790 assert_eq!(
791 vim_to_rust_regex(r"\(hello\) \(world\)"),
792 r"(hello) (world)"
793 );
794 }
795
796 #[test]
797 fn default_magic_quantifiers_and_alternation() {
798 assert_eq!(vim_to_rust_regex(r"a\+"), r"a+");
799 assert_eq!(vim_to_rust_regex(r"a\?"), r"a?");
800 assert_eq!(vim_to_rust_regex(r"a\="), r"a?");
801 assert_eq!(vim_to_rust_regex(r"a\|b"), r"a|b");
802 }
803
804 #[test]
805 fn default_magic_counted_repeat_bare_close() {
806 // vim allows `\{n,m}` with an UNESCAPED closing brace.
807 assert_eq!(vim_to_rust_regex(r"a\{1,2}"), r"a{1,2}");
808 // Fully-escaped form also works.
809 assert_eq!(vim_to_rust_regex(r"a\{1,2\}"), r"a{1,2}");
810 }
811
812 #[test]
813 fn default_magic_unescaped_group_chars_are_literal() {
814 // The INVERSE: unescaped ( ) + ? | { } are literals in default magic.
815 assert_eq!(vim_to_rust_regex("(a)"), r"\(a\)");
816 assert_eq!(vim_to_rust_regex("a+b"), r"a\+b");
817 assert_eq!(vim_to_rust_regex("a|b"), r"a\|b");
818 assert_eq!(vim_to_rust_regex("a?b"), r"a\?b");
819 }
820
821 #[test]
822 fn default_magic_dot_star_bracket_caret_dollar_stay_magic() {
823 assert_eq!(vim_to_rust_regex("a.b"), "a.b");
824 assert_eq!(vim_to_rust_regex("a*"), "a*");
825 assert_eq!(vim_to_rust_regex("[0-9]"), "[0-9]");
826 assert_eq!(vim_to_rust_regex("^foo$"), "^foo$");
827 }
828
829 #[test]
830 fn magic_tilde_expands_to_last_sub_empty_via_wrapper() {
831 // `vim_to_rust_regex` passes an empty last-substitute string, so a bare
832 // magic `~` expands to "" (nvim would `E33` with no prior `:s`; we pick
833 // the safe empty expansion). `\~` stays a literal tilde.
834 assert_eq!(vim_to_rust_regex("a~b"), "ab");
835 assert_eq!(vim_to_rust_regex(r"a\~b"), "a~b");
836 }
837
838 // ── Magic `~` PATTERN-side expansion (V5) ────────────────────────────────
839
840 /// `~` expands to the supplied last-substitute string; `\~` stays literal.
841 /// nvim-verified: after `:s/foo/BAR/`, `/~` matches the text `BAR`.
842 #[test]
843 fn magic_tilde_expands_to_last_sub() {
844 let (out, _) = resolve_case_mode("~", CaseMode::Sensitive, "BAR");
845 assert_eq!(out, "BAR");
846 // Surrounded by other pattern text.
847 let (out, _) = resolve_case_mode("x~y", CaseMode::Sensitive, "BAR");
848 assert_eq!(out, "xBARy");
849 }
850
851 /// `\~` is a literal tilde and must NOT expand, even with a last-sub set.
852 /// nvim-verified: `\~` in a pattern matches a real `~` character.
853 #[test]
854 fn escaped_tilde_stays_literal_and_does_not_expand() {
855 let (out, _) = resolve_case_mode(r"\~", CaseMode::Sensitive, "BAR");
856 assert_eq!(out, "~");
857 // Compiled: matches a real tilde, not "BAR".
858 let re = Regex::new(&out).unwrap();
859 assert!(re.is_match("a~b"));
860 assert!(!re.is_match("BAR"));
861 }
862
863 /// `~` inside a `[...]` class is a literal class member, never an
864 /// expansion. nvim-verified: `[~]` matches the tilde character.
865 #[test]
866 fn tilde_in_bracket_class_is_literal() {
867 let (out, _) = resolve_case_mode("[~]", CaseMode::Sensitive, "BAR");
868 assert_eq!(out, "[~]");
869 }
870
871 /// No previous substitute (empty last-sub) → `~` expands to empty.
872 /// Documented divergence from nvim's `E33`; the empty choice never
873 /// corrupts the buffer.
874 #[test]
875 fn magic_tilde_no_previous_sub_expands_empty() {
876 let (out, _) = resolve_case_mode("a~b", CaseMode::Sensitive, "");
877 assert_eq!(out, "ab");
878 }
879
880 #[test]
881 fn very_magic_mode_switch_at_start() {
882 // \v: groups/quantifiers/alternation/boundaries are magic unescaped.
883 assert_eq!(vim_to_rust_regex(r"\v(\w+) (\w+)"), r"(\w+) (\w+)");
884 assert_eq!(vim_to_rust_regex(r"\v\d+"), r"\d+");
885 assert_eq!(vim_to_rust_regex(r"\v<foo>"), r"\bfoo\b");
886 assert_eq!(vim_to_rust_regex(r"\va=b"), r"a?b");
887 }
888
889 #[test]
890 fn very_magic_mode_escaped_chars_are_literal() {
891 // In \v mode, backslash forces the LITERAL reading of an
892 // otherwise-special char.
893 assert_eq!(vim_to_rust_regex(r"\v\(a\)"), r"\(a\)");
894 assert_eq!(vim_to_rust_regex(r"\va\+b"), r"a\+b");
895 }
896
897 #[test]
898 fn very_nomagic_mode_is_all_literal_except_backslash() {
899 // \V: everything literal except `\`-escaped.
900 assert_eq!(vim_to_rust_regex(r"\Va.b"), r"a\.b");
901 assert_eq!(vim_to_rust_regex(r"\V(a)"), r"\(a\)");
902 // Backslash still activates special meaning (mirrors \v).
903 assert_eq!(vim_to_rust_regex(r"\Va\.b"), r"a.b");
904 }
905
906 #[test]
907 fn nomagic_mode_only_caret_dollar_special() {
908 // \M: only ^ $ special unescaped; `.` `*` `[` become literal.
909 assert_eq!(vim_to_rust_regex(r"\M^a.b$"), r"^a\.b$");
910 assert_eq!(vim_to_rust_regex(r"\Ma\.b"), r"a.b");
911 }
912
913 #[test]
914 fn mode_switch_mid_pattern() {
915 // Switching mode partway through the pattern applies from that point on.
916 assert_eq!(vim_to_rust_regex(r"(a)\v(b)"), r"\(a\)(b)");
917 assert_eq!(vim_to_rust_regex(r"\va\mb+"), r"ab\+");
918 }
919
920 #[test]
921 fn backreference_in_pattern_passes_through_unchanged() {
922 // \1-\9 in the PATTERN aren't supported by rust-regex (no
923 // backtracking) — kept as a literal backslash-digit escape so the
924 // net effect (no match / compile error) matches pre-fix behavior
925 // rather than silently corrupting text.
926 assert_eq!(vim_to_rust_regex(r"\(a\)\1"), r"(a)\1");
927 }
928
929 #[test]
930 fn character_class_contents_not_translated() {
931 // Unescaped `(` `)` inside `[...]` are literal class members in both
932 // vim and rust-regex — bracket tracking must not turn them into a
933 // group by escaping/unescaping their contents.
934 assert_eq!(vim_to_rust_regex("[()]"), "[()]");
935 }
936
937 // ── search reveals folds ─────────────────────────────────────────────────
938
939 /// `search_forward` on a buffer with a closed fold hiding the match row:
940 /// after finding the match, calling `reveal_row` opens the fold.
941 /// (Mirrors what `Editor::search_advance_forward` does.)
942 #[test]
943 fn search_forward_reveals_fold() {
944 use hjkl_buffer::View;
945
946 // View: row 0 = "header", row 1 = "needle", row 2 = "footer"
947 // Fold [0..2] closed → row 1 is hidden.
948 let mut buf = View::from_str("header\nneedle\nfooter");
949 buf.add_fold(0, 2, true);
950 assert!(buf.is_row_hidden(1), "row 1 must be hidden before search");
951
952 let mut state = SearchState::new();
953 state.set_pattern(Some(re("needle")));
954
955 // Use search_forward directly on the buffer.
956 let found = search_forward(&mut buf, &mut state, false);
957 assert!(found, "search_forward must find 'needle'");
958
959 // After search_forward, cursor is on row 1. Reveal as Editor does.
960 let row = crate::types::Cursor::cursor(&buf).line as usize;
961 buf.reveal_row(row);
962 assert!(
963 !buf.is_row_hidden(1),
964 "row 1 must be revealed after search finds it there"
965 );
966 }
967
968 /// `search_backward` similarly: finding a match then calling reveal_row opens folds.
969 #[test]
970 fn search_backward_reveals_fold() {
971 use hjkl_buffer::View;
972
973 // row 0 = "footer", row 1 = "needle", row 2 = "header"
974 // fold [0..2] closed → row 1 hidden. Start cursor at row 2.
975 let mut buf = View::from_str("footer\nneedle\nheader");
976 buf.add_fold(0, 2, true);
977 crate::types::Cursor::set_cursor(&mut buf, crate::types::Pos::new(2, 0));
978 assert!(buf.is_row_hidden(1), "row 1 must be hidden before search");
979
980 let mut state = SearchState::new();
981 state.set_pattern(Some(re("needle")));
982
983 let found = search_backward(&mut buf, &mut state, false);
984 assert!(found, "search_backward must find 'needle'");
985
986 let row = crate::types::Cursor::cursor(&buf).line as usize;
987 buf.reveal_row(row);
988 assert!(
989 !buf.is_row_hidden(1),
990 "row 1 must be revealed after backward search finds it"
991 );
992 }
993
994 #[test]
995 fn forward_finds_first_match() {
996 let mut b = View::from_str("foo bar foo baz");
997 let mut s = SearchState::new();
998 s.set_pattern(Some(re("foo")));
999 assert!(search_forward(&mut b, &mut s, false));
1000 assert_eq!(Cursor::cursor(&b), Pos::new(0, 0));
1001 }
1002
1003 #[test]
1004 fn forward_skip_current_walks_past() {
1005 let mut b = View::from_str("foo bar foo baz");
1006 let mut s = SearchState::new();
1007 s.set_pattern(Some(re("foo")));
1008 search_forward(&mut b, &mut s, false);
1009 search_forward(&mut b, &mut s, true);
1010 assert_eq!(Cursor::cursor(&b), Pos::new(0, 8));
1011 }
1012
1013 #[test]
1014 fn forward_wraps_to_top() {
1015 let mut b = View::from_str("zzz\nfoo");
1016 // 0.0.37: wrap policy lives entirely on `SearchState::wrap_around`;
1017 // the buffer-side `set_search_wrap` accessor is gone. Trait
1018 // `find_next` always wraps; the engine search free function
1019 // honours `s.wrap_around` directly.
1020 Cursor::set_cursor(&mut b, Pos::new(1, 2));
1021 let mut s = SearchState::new();
1022 s.set_pattern(Some(re("zzz")));
1023 s.wrap_around = true;
1024 assert!(search_forward(&mut b, &mut s, true));
1025 assert_eq!(Cursor::cursor(&b), Pos::new(0, 0));
1026 }
1027
1028 /// `n` from a match whose first byte begins a multi-byte char must
1029 /// advance to the next match. `pos_at_byte` rounds a mid-char byte
1030 /// DOWN to the enclosing char's start, so a one-byte step from the
1031 /// char's first byte lands back on the cursor itself — regression:
1032 /// `n` was permanently stuck on "éé".
1033 #[test]
1034 fn forward_skip_current_past_multibyte_char() {
1035 let mut b = View::from_str("éé");
1036 let mut s = SearchState::new();
1037 s.set_pattern(Some(re("é")));
1038 // Cursor starts on the first `é` (col 0); `n` must land on the
1039 // second one (col 1), not re-find the current match.
1040 assert!(search_forward(&mut b, &mut s, true));
1041 assert_eq!(Cursor::cursor(&b), Pos::new(0, 1));
1042 }
1043
1044 /// `N` from a match that starts at buffer byte 0 wraps to the last
1045 /// match of the buffer instead of staying put — regression: the
1046 /// "no earlier byte" branch returned `None` and never wrapped.
1047 #[test]
1048 fn backward_skip_current_wraps_from_byte_zero() {
1049 let mut b = View::from_str("foo\nfoo");
1050 Cursor::set_cursor(&mut b, Pos::new(0, 0));
1051 let mut s = SearchState::new();
1052 s.set_pattern(Some(re("foo")));
1053 assert!(search_backward(&mut b, &mut s, true));
1054 assert_eq!(Cursor::cursor(&b), Pos::new(1, 0));
1055 }
1056
1057 #[test]
1058 fn search_matches_caches_against_dirty_gen() {
1059 let b = View::from_str("foo bar");
1060 let mut s = SearchState::new();
1061 s.set_pattern(Some(re("bar")));
1062 let dgen = b.dirty_gen();
1063 let initial = search_matches(&b, &mut s, dgen, 0);
1064 assert_eq!(initial, &[(4, 7)][..]);
1065 }
1066
1067 // ── CaseMode::from_options matrix ────────────────────────────────────────
1068
1069 #[test]
1070 fn case_mode_from_options_matrix() {
1071 // ic=false, smart=* → Sensitive
1072 assert_eq!(CaseMode::from_options(false, false), CaseMode::Sensitive);
1073 assert_eq!(CaseMode::from_options(false, true), CaseMode::Sensitive);
1074 // ic=true, smart=false → Insensitive
1075 assert_eq!(CaseMode::from_options(true, false), CaseMode::Insensitive);
1076 // ic=true, smart=true → Smart
1077 assert_eq!(CaseMode::from_options(true, true), CaseMode::Smart);
1078 }
1079
1080 // ── resolve_case_mode unit tests ─────────────────────────────────────────
1081
1082 #[test]
1083 fn resolve_case_mode_no_override_smart_lowercase() {
1084 let (stripped, mode) = resolve_case_mode("foo", CaseMode::Smart, "");
1085 assert_eq!(stripped, "foo");
1086 assert_eq!(mode, CaseMode::Insensitive);
1087 }
1088
1089 #[test]
1090 fn resolve_case_mode_no_override_smart_uppercase() {
1091 let (stripped, mode) = resolve_case_mode("Foo", CaseMode::Smart, "");
1092 assert_eq!(stripped, "Foo");
1093 assert_eq!(mode, CaseMode::Sensitive);
1094 }
1095
1096 #[test]
1097 fn resolve_case_mode_lower_c_override() {
1098 // \c overrides Sensitive → Insensitive; stripped pattern is "Foo"
1099 let (stripped, mode) = resolve_case_mode(r"\cFoo", CaseMode::Sensitive, "");
1100 assert_eq!(stripped, "Foo");
1101 assert_eq!(mode, CaseMode::Insensitive);
1102 }
1103
1104 #[test]
1105 fn resolve_case_mode_upper_c_override() {
1106 // \C overrides Smart → Sensitive; stripped pattern is "foo"
1107 let (stripped, mode) = resolve_case_mode(r"foo\C", CaseMode::Smart, "");
1108 assert_eq!(stripped, "foo");
1109 assert_eq!(mode, CaseMode::Sensitive);
1110 }
1111
1112 #[test]
1113 fn resolve_case_mode_last_wins() {
1114 // \c then \C → last-wins → Sensitive; stripped "foo"
1115 let (stripped, mode) = resolve_case_mode(r"\cfoo\C", CaseMode::Smart, "");
1116 assert_eq!(stripped, "foo");
1117 assert_eq!(mode, CaseMode::Sensitive);
1118 }
1119
1120 // ── Integration: search with smartcase / \c / \C ─────────────────────────
1121
1122 fn build_regex_from(pat: &str, ic: bool, smart: bool) -> Regex {
1123 let base = CaseMode::from_options(ic, smart);
1124 let (stripped, mode) = resolve_case_mode(pat, base, "");
1125 let src = if mode == CaseMode::Insensitive {
1126 format!("(?i){stripped}")
1127 } else {
1128 stripped
1129 };
1130 Regex::new(&src).unwrap()
1131 }
1132
1133 #[test]
1134 fn search_finds_capital_with_smartcase_lowercase_pattern() {
1135 // ic=true, smart=true, pattern "foo" → Insensitive → matches "FOO"
1136 let re = build_regex_from("foo", true, true);
1137 assert!(re.is_match("FOO"), "expected match on 'FOO'");
1138 assert!(re.is_match("foo"), "expected match on 'foo'");
1139 }
1140
1141 #[test]
1142 fn search_skips_capital_with_smartcase_mixed_pattern() {
1143 // ic=true, smart=true, pattern "Foo" → Sensitive → does NOT match "FOO"
1144 let re = build_regex_from("Foo", true, true);
1145 assert!(!re.is_match("FOO"), "must not match 'FOO' (case-sensitive)");
1146 assert!(re.is_match("Foo"), "must match exact 'Foo'");
1147 }
1148
1149 #[test]
1150 fn search_lower_c_override_finds_capital() {
1151 // \cFoo + Sensitive base → Insensitive override → matches "FOO"
1152 let re = build_regex_from(r"\cFoo", false, false);
1153 assert!(re.is_match("FOO"), "\\c override must match 'FOO'");
1154 assert!(re.is_match("foo"), "\\c override must match 'foo'");
1155 }
1156
1157 #[test]
1158 fn vim_to_rust_regex_strips_case_overrides() {
1159 // vim_to_rust_regex is now a thin wrapper; \c and \C are stripped
1160 assert_eq!(vim_to_rust_regex(r"\cfoo"), "foo");
1161 assert_eq!(vim_to_rust_regex(r"foo\C"), "foo");
1162 assert_eq!(vim_to_rust_regex(r"\<bar\>"), r"\bbar\b");
1163 }
1164
1165 /// `*` on word "foo" emits the pattern `\bfoo\b` (all lowercase). Under
1166 /// smartcase that resolves to Insensitive → should match "FOO". This test
1167 /// simulates the word_at_cursor_search pattern-build path.
1168 #[test]
1169 fn star_search_finds_lowercase_when_smartcase_lower_word() {
1170 // word_at_cursor_search escapes the word then wraps \b..\b.
1171 // "foo" is all-lowercase after word-extraction → Smart → Insensitive.
1172 let pat = r"\bfoo\b";
1173 let re = build_regex_from(pat, true, true);
1174 // Case-insensitive → matches "FOO foo Foo".
1175 let text = "FOO foo Foo";
1176 let hits: Vec<_> = re.find_iter(text).map(|m| m.as_str()).collect();
1177 assert!(
1178 hits.contains(&"FOO"),
1179 "smartcase lower-word * must match FOO: {hits:?}"
1180 );
1181 assert!(
1182 hits.contains(&"foo"),
1183 "smartcase lower-word * must match foo: {hits:?}"
1184 );
1185 }
1186
1187 // ── \a / \A / \Z / \e escape translation ─────────────────────────────────
1188
1189 /// vim `\a` = alphabetic. rust-regex would read `\a` as Bell (U+0007),
1190 /// so `:s/\a/x/g` used to silently no-op; it must now match letters.
1191 #[test]
1192 fn backslash_a_is_alphabetic() {
1193 let re = vim_re(r"\a");
1194 assert!(re.is_match("a"));
1195 assert!(re.is_match("B"));
1196 assert!(!re.is_match("1"));
1197 let hits: Vec<_> = Regex::new(&vim_to_rust_regex(r"\a"))
1198 .unwrap()
1199 .find_iter("ab1")
1200 .map(|m| m.as_str())
1201 .collect();
1202 assert_eq!(hits, vec!["a", "b"]);
1203 }
1204
1205 /// vim `\A` = non-alphabetic. rust-regex `\A` is a start-of-text anchor,
1206 /// so `:s/\A/x/` used to insert at position 0 ("ab1" → "xab1"); it must
1207 /// now match the `1` and give vim's "abx".
1208 #[test]
1209 fn backslash_upper_a_is_non_alphabetic() {
1210 let re = vim_re(r"\A");
1211 assert!(re.is_match("1"));
1212 assert!(!re.is_match("a"));
1213 let hits: Vec<_> = Regex::new(&vim_to_rust_regex(r"\A"))
1214 .unwrap()
1215 .find_iter("ab1")
1216 .map(|m| m.as_str())
1217 .collect();
1218 assert_eq!(hits, vec!["1"]);
1219 }
1220
1221 /// `[\a]` inside a class is the alphabetic range, not a Bell escape.
1222 #[test]
1223 fn backslash_a_inside_class_is_alpha_range() {
1224 assert_eq!(vim_to_rust_regex(r"[\a]"), "[A-Za-z]");
1225 let re = vim_re(r"[\a]");
1226 assert!(re.is_match("x"));
1227 assert!(!re.is_match("1"));
1228 }
1229
1230 /// An ESCAPED `]` inside a class is a literal member — vim `[a\]b]` =
1231 /// {a, ], b} — and must not close the class early. The pre-fix translator
1232 /// closed on the escaped `]`, emitting `[a\]b\]` which rust-regex rejects
1233 /// as an unclosed class, so `:s/[a\]b]/x/` errored where vim substitutes.
1234 #[test]
1235 fn escaped_close_bracket_inside_class_is_literal_member() {
1236 let re = vim_re(r"[a\]b]");
1237 assert!(re.is_match("a"), "class must contain a");
1238 assert!(re.is_match("]"), "escaped ] must be a literal member");
1239 assert!(re.is_match("b"), "class must contain b");
1240 assert!(!re.is_match("x"), "class must be exactly {{a, ], b}}");
1241 // The canonical `[a\]]` form, and `\\]` (even backslash run) closing.
1242 assert!(vim_re(r"[a\]]").is_match("]"));
1243 assert!(vim_re(r"[\\]").is_match("\\"));
1244 assert!(!vim_re(r"[\\]").is_match("]"));
1245 }
1246
1247 /// vim `\Z` — ignore case for the rest of the pattern, identical to `\c`
1248 /// (rust-regex rejects `\Z` outright).
1249 #[test]
1250 fn backslash_z_makes_pattern_case_insensitive() {
1251 let (stripped, mode) = resolve_case_mode(r"\Zfoo", CaseMode::Sensitive, "");
1252 assert_eq!(stripped, "foo");
1253 assert_eq!(mode, CaseMode::Insensitive);
1254 // End-to-end: a Sensitive base must be overridden by `\Z`.
1255 let re = build_regex_from(r"\Zfoo", false, false);
1256 assert!(re.is_match("FOO"), "\\Z must make pattern insensitive");
1257 assert!(re.is_match("foo"));
1258 }
1259
1260 /// vim `\e` = ESC (U+001B); rust-regex has no `\e` escape and rejects it.
1261 #[test]
1262 fn backslash_e_is_esc() {
1263 assert_eq!(vim_to_rust_regex(r"\e"), "\u{1b}");
1264 let re = vim_re(r"\e");
1265 assert!(re.is_match("\u{1b}"));
1266 assert!(!re.is_match("e"));
1267 }
1268
1269 // ── substitution-level regression for \a / \A (bug 2) ────────────────────
1270
1271 fn editor_for_substitute(
1272 content: &str,
1273 ) -> crate::Editor<hjkl_buffer::View, crate::types::DefaultHost> {
1274 let mut e = crate::Editor::new(
1275 hjkl_buffer::View::new(),
1276 crate::types::DefaultHost::new(),
1277 crate::types::Options::default(),
1278 );
1279 e.set_content(content);
1280 e
1281 }
1282
1283 /// `:s/\a/x/g` on "ab1" must replace both letters — the pre-fix rust-regex
1284 /// reading of `\a` (Bell) matched nothing and the command silently no-oped.
1285 #[test]
1286 fn substitute_backslash_a_replaces_letters() {
1287 let mut e = editor_for_substitute("ab1");
1288 let cmd = crate::substitute::parse_substitute(r"/\a/x/g").unwrap();
1289 let out = crate::substitute::apply_substitute(&mut e, &cmd, 0..=0).unwrap();
1290 assert_eq!(out.replacements, 2);
1291 assert_eq!(hjkl_buffer::rope_line_str(&e.buffer().rope(), 0), "xx1");
1292 }
1293
1294 /// `:s/\A/x/` (first match per line) on "ab1" gives "abx" — the pre-fix
1295 /// rust-regex `\A` (start-of-text anchor) replaced at position 0 instead,
1296 /// producing "xab1".
1297 #[test]
1298 fn substitute_backslash_upper_a_replaces_first_non_alpha() {
1299 let mut e = editor_for_substitute("ab1");
1300 let cmd = crate::substitute::parse_substitute(r"/\A/x/").unwrap();
1301 let out = crate::substitute::apply_substitute(&mut e, &cmd, 0..=0).unwrap();
1302 assert_eq!(out.replacements, 1);
1303 assert_eq!(hjkl_buffer::rope_line_str(&e.buffer().rope(), 0), "abx");
1304 }
1305}