g1t/services/search/src/snippet.rs

263 lines9,216 bytesCodeBlame

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Search across all of g1t, Explore, and a command palette1//! The part of a result that matched, with the matches marked: a window of
2//! prose, or numbered lines of code. Matching ignores case. Pure, so it is
3//! tested apart from the index.
4
5use g1t_contracts::search::{CodeLine, Segment};
6
7/// Characters of prose a snippet shows, about.
8pub const PROSE_CHARS: usize = 220;
9/// Lines of code a result shows, at most.
10pub const CODE_LINES: usize = 9;
11/// Characters of one line of code shown; the rest is cut.
12const LINE_CHARS: usize = 300;
13
14fn fold(c: char) -> char {
15 c.to_lowercase().next().unwrap_or(c)
16}
17
18/// Which characters of `text` are inside a match of any of `terms`.
19fn marks(text: &[char], terms: &[Vec<char>]) -> Vec<bool> {
20 let mut marked = vec![false; text.len()];
21 for term in terms.iter().filter(|term| !term.is_empty()) {
22 if term.len() > text.len() {
23 continue;
24 }
25 for start in 0..=text.len() - term.len() {
26 if term.iter().zip(&text[start..]).all(|(a, b)| *a == fold(*b)) {
27 marked[start..start + term.len()].iter_mut().for_each(|m| *m = true);
28 }
29 }
30 }
31 marked
32}
33
34fn folded(terms: &[String]) -> Vec<Vec<char>> {
35 terms.iter().map(|term| term.chars().map(fold).collect()).collect()
36}
37
38/// `text` in pieces, each marked if it matched.
39fn segments_of(text: &[char], marked: &[bool]) -> Vec<Segment> {
40 let mut out: Vec<Segment> = Vec::new();
41 for (c, highlight) in text.iter().zip(marked) {
42 match out.last_mut() {
43 Some(last) if last.highlight == *highlight => last.text.push(*c),
44 _ => out.push(Segment {
45 text: c.to_string(),
46 highlight: *highlight,
47 }),
48 }
49 }
50 out
51}
52
53/// All of `text`, with every match of `terms` marked.
54#[cfg(test)]
55pub fn highlight(text: &str, terms: &[String]) -> Vec<Segment> {
56 let chars: Vec<char> = text.chars().collect();
57 let marked = marks(&chars, &folded(terms));
58 segments_of(&chars, &marked)
59}
60
61/// Whether `text` has any of `terms`.
62pub fn contains_any(text: &str, terms: &[String]) -> bool {
63 let chars: Vec<char> = text.chars().collect();
64 marks(&chars, &folded(terms)).contains(&true)
65}
66
67/// About [`PROSE_CHARS`] of `text` around its first match, whitespace made
68/// single spaces, with `…` where it was cut and every match marked. The
69/// start of the text when nothing matches; empty for empty text.
70pub fn prose(text: &str, terms: &[String]) -> Vec<Segment> {
71 let flat: Vec<char> = text.split_whitespace().collect::<Vec<_>>().join(" ").chars().collect();
72 if flat.is_empty() {
73 return Vec::new();
74 }
75 let marked = marks(&flat, &folded(terms));
76 let first = marked.iter().position(|m| *m).unwrap_or(0);
77 // A little before the first match, so it reads in context.
78 let mut start = first.saturating_sub(PROSE_CHARS / 4);
79 let end = (start + PROSE_CHARS).min(flat.len());
80 if end - start < PROSE_CHARS {
81 start = end.saturating_sub(PROSE_CHARS);
82 }
83 // Start and end on a word boundary where there is one nearby.
Cleanup: clippy is quiet outside billing, the README says g1t, and http-cache-semantics is 4.3.084 if start > 0
85 && let Some(space) = flat[start..first.max(start)].iter().position(|c| *c == ' ') {
Search across all of g1t, Explore, and a command palette86 start += space + 1;
87 }
88 let mut end = end;
89 if end < flat.len()
90 && let Some(space) = flat[start..end].iter().rposition(|c| *c == ' ')
91 && start + space > first
92 {
93 end = start + space;
94 }
95 let mut out = segments_of(&flat[start..end], &marked[start..end]);
96 if start > 0 {
97 prepend(&mut out, "…");
98 }
99 if end < flat.len() {
100 append(&mut out, "…");
101 }
102 out
103}
104
105fn prepend(segments: &mut Vec<Segment>, text: &str) {
106 match segments.first_mut() {
107 Some(first) if !first.highlight => first.text.insert_str(0, text),
108 _ => segments.insert(0, Segment { text: text.to_owned(), highlight: false }),
109 }
110}
111
112fn append(segments: &mut Vec<Segment>, text: &str) {
113 match segments.last_mut() {
114 Some(last) if !last.highlight => last.text.push_str(text),
115 _ => segments.push(Segment { text: text.to_owned(), highlight: false }),
116 }
117}
118
119/// The lines of a file that matched, with the line before and after each,
120/// at most [`CODE_LINES`]. `pieces` are parts of the file as indexed, each
121/// with the number of its first line. When no line matched (a match on the
122/// file's path), its first lines.
123pub fn code(pieces: &[(u32, &str)], terms: &[String]) -> Vec<CodeLine> {
124 let folded_terms = folded(terms);
125 let mut lines: Vec<(u32, Vec<char>, Vec<bool>)> = Vec::new();
126 for (start, text) in pieces {
127 for (offset, line) in text.lines().enumerate() {
128 let chars: Vec<char> = line.trim_end_matches('\r').chars().take(LINE_CHARS).collect();
129 let marked = marks(&chars, &folded_terms);
130 lines.push((start + offset as u32, chars, marked));
131 }
132 }
133 lines.sort_by_key(|(number, _, _)| *number);
134 lines.dedup_by_key(|(number, _, _)| *number);
135 let hits: Vec<usize> = lines
136 .iter()
137 .enumerate()
138 .filter(|(_, (_, _, marked))| marked.contains(&true))
139 .map(|(index, _)| index)
140 .collect();
141 let mut wanted: Vec<usize> = Vec::new();
142 if hits.is_empty() {
143 wanted.extend(0..lines.len().min(3));
144 }
145 for hit in hits {
146 for index in hit.saturating_sub(1)..=(hit + 1).min(lines.len().saturating_sub(1)) {
147 if !wanted.contains(&index) {
148 wanted.push(index);
149 }
150 }
151 if wanted.len() >= CODE_LINES {
152 break;
153 }
154 }
155 wanted.sort_unstable();
156 wanted.truncate(CODE_LINES);
157 wanted
158 .into_iter()
159 .map(|index| {
160 let (number, chars, marked) = &lines[index];
161 CodeLine {
162 number: *number,
163 parts: segments_of(chars, marked),
164 }
165 })
166 .collect()
167}
168
169/// The first line number that matched, for a link straight to it.
170pub fn first_match(lines: &[CodeLine]) -> Option<u32> {
171 lines
172 .iter()
173 .find(|line| line.parts.iter().any(|part| part.highlight))
174 .map(|line| line.number)
175}
176
177#[cfg(test)]
178mod tests {
179 use super::*;
180
181 fn shown(segments: &[Segment]) -> String {
182 segments
183 .iter()
184 .map(|s| if s.highlight { format!("[{}]", s.text) } else { s.text.clone() })
185 .collect()
186 }
187
188 fn terms(list: &[&str]) -> Vec<String> {
189 list.iter().map(|t| t.to_string()).collect()
190 }
191
192 #[test]
193 fn matches_are_marked_whatever_their_case() {
194 let out = highlight("Parse the parser, PARSE it", &terms(&["parse"]));
195 assert_eq!(shown(&out), "[Parse] the [parse]r, [PARSE] it");
196 }
197
198 #[test]
199 fn overlapping_terms_make_one_mark() {
200 let out = highlight("foobar", &terms(&["foo", "oba"]));
201 assert_eq!(shown(&out), "[fooba]r");
202 }
203
204 #[test]
205 fn nothing_to_mark_is_plain_text() {
206 assert_eq!(shown(&highlight("plain", &terms(&["zzz"]))), "plain");
207 assert_eq!(shown(&highlight("plain", &[])), "plain");
208 }
209
210 #[test]
211 fn letters_outside_ascii_are_matched() {
212 let out = highlight("Ärger über Ölpreise", &terms(&["über", "öl"]));
213 assert_eq!(shown(&out), "Ärger [über] [Öl]preise");
214 }
215
216 #[test]
217 fn prose_is_a_window_around_the_first_match() {
218 let text = format!("{} the needle is here {}", "word ".repeat(100), "tail ".repeat(100));
219 let out = prose(&text, &terms(&["needle"]));
220 let plain = shown(&out);
221 assert!(plain.starts_with('…') && plain.ends_with('…'), "{plain}");
222 assert!(plain.contains("[needle]"));
223 assert!(plain.chars().count() <= PROSE_CHARS + 4);
224 }
225
226 #[test]
227 fn short_prose_is_shown_whole() {
228 let out = prose("Fix the\nlogin redirect", &terms(&["login"]));
229 assert_eq!(shown(&out), "Fix the [login] redirect");
230 assert!(prose("", &terms(&["x"])).is_empty());
231 }
232
233 #[test]
234 fn code_lines_are_numbered_with_context() {
235 let first = "use std::io;\n\nfn main() {\n let query = parse();\n}\n";
236 let out = code(&[(1, first)], &terms(&["parse"]));
237 let numbers: Vec<u32> = out.iter().map(|line| line.number).collect();
238 assert_eq!(numbers, vec![3, 4, 5]);
239 assert_eq!(shown(&out[1].parts), " let query = [parse]();");
240 assert_eq!(first_match(&out), Some(4));
241 }
242
243 #[test]
244 fn code_from_later_pieces_keeps_its_numbers() {
245 let piece = "a\nb match\nc\n";
246 let out = code(&[(121, piece)], &terms(&["match"]));
247 assert_eq!(out.iter().map(|l| l.number).collect::<Vec<_>>(), vec![121, 122, 123]);
248 }
249
250 #[test]
251 fn a_path_match_shows_the_first_lines() {
252 let out = code(&[(1, "one\ntwo\nthree\nfour\n")], &terms(&["zzz"]));
253 assert_eq!(out.iter().map(|l| l.number).collect::<Vec<_>>(), vec![1, 2, 3]);
254 assert_eq!(first_match(&out), None);
255 }
256
257 #[test]
258 fn many_matches_stop_at_the_limit() {
259 let text: String = (0..100).map(|n| format!("hit {n}\n")).collect();
260 let out = code(&[(1, &text)], &terms(&["hit"]));
261 assert_eq!(out.len(), CODE_LINES);
262 }
263}