g1t/services/search/src/snippet.rs
| 1 | //! The part of a result that matched, with the matches marked: a window of |
| 2 | //! prose, or numbered lines of code. Matching ignores case. Pure, so it is |
| 3 | //! tested apart from the index. |
| 4 | |
| 5 | use g1t_contracts::search::{CodeLine, Segment}; |
| 6 | |
| 7 | /// Characters of prose a snippet shows, about. |
| 8 | pub const PROSE_CHARS: usize = 220; |
| 9 | /// Lines of code a result shows, at most. |
| 10 | pub const CODE_LINES: usize = 9; |
| 11 | /// Characters of one line of code shown; the rest is cut. |
| 12 | const LINE_CHARS: usize = 300; |
| 13 | |
| 14 | fn fold(c: char) -> char { |
| 15 | c.to_lowercase().next().unwrap_or(c) |
| 16 | } |
| 17 | |
| 18 | /// Which characters of `text` are inside a match of any of `terms`. |
| 19 | fn marks(text: &[char], terms: &[Vec<char>]) -> Vec<bool> { |
| 20 | let mut marked = vec![false; text.len()]; |
| 21 | for term in terms.iter().filter(|term| !term.is_empty()) { |
| 22 | if term.len() > text.len() { |
| 23 | continue; |
| 24 | } |
| 25 | for start in 0..=text.len() - term.len() { |
| 26 | if term.iter().zip(&text[start..]).all(|(a, b)| *a == fold(*b)) { |
| 27 | marked[start..start + term.len()].iter_mut().for_each(|m| *m = true); |
| 28 | } |
| 29 | } |
| 30 | } |
| 31 | marked |
| 32 | } |
| 33 | |
| 34 | fn folded(terms: &[String]) -> Vec<Vec<char>> { |
| 35 | terms.iter().map(|term| term.chars().map(fold).collect()).collect() |
| 36 | } |
| 37 | |
| 38 | /// `text` in pieces, each marked if it matched. |
| 39 | fn segments_of(text: &[char], marked: &[bool]) -> Vec<Segment> { |
| 40 | let mut out: Vec<Segment> = Vec::new(); |
| 41 | for (c, highlight) in text.iter().zip(marked) { |
| 42 | match out.last_mut() { |
| 43 | Some(last) if last.highlight == *highlight => last.text.push(*c), |
| 44 | _ => out.push(Segment { |
| 45 | text: c.to_string(), |
| 46 | highlight: *highlight, |
| 47 | }), |
| 48 | } |
| 49 | } |
| 50 | out |
| 51 | } |
| 52 | |
| 53 | /// All of `text`, with every match of `terms` marked. |
| 54 | #[cfg(test)] |
| 55 | pub fn highlight(text: &str, terms: &[String]) -> Vec<Segment> { |
| 56 | let chars: Vec<char> = text.chars().collect(); |
| 57 | let marked = marks(&chars, &folded(terms)); |
| 58 | segments_of(&chars, &marked) |
| 59 | } |
| 60 | |
| 61 | /// Whether `text` has any of `terms`. |
| 62 | pub fn contains_any(text: &str, terms: &[String]) -> bool { |
| 63 | let chars: Vec<char> = text.chars().collect(); |
| 64 | marks(&chars, &folded(terms)).contains(&true) |
| 65 | } |
| 66 | |
| 67 | /// About [`PROSE_CHARS`] of `text` around its first match, whitespace made |
| 68 | /// single spaces, with `…` where it was cut and every match marked. The |
| 69 | /// start of the text when nothing matches; empty for empty text. |
| 70 | pub fn prose(text: &str, terms: &[String]) -> Vec<Segment> { |
| 71 | let flat: Vec<char> = text.split_whitespace().collect::<Vec<_>>().join(" ").chars().collect(); |
| 72 | if flat.is_empty() { |
| 73 | return Vec::new(); |
| 74 | } |
| 75 | let marked = marks(&flat, &folded(terms)); |
| 76 | let first = marked.iter().position(|m| *m).unwrap_or(0); |
| 77 | // A little before the first match, so it reads in context. |
| 78 | let mut start = first.saturating_sub(PROSE_CHARS / 4); |
| 79 | let end = (start + PROSE_CHARS).min(flat.len()); |
| 80 | if end - start < PROSE_CHARS { |
| 81 | start = end.saturating_sub(PROSE_CHARS); |
| 82 | } |
| 83 | // Start and end on a word boundary where there is one nearby. |
| 84 | if start > 0 { |
| 85 | if let Some(space) = flat[start..first.max(start)].iter().position(|c| *c == ' ') { |
| 86 | start += space + 1; |
| 87 | } |
| 88 | } |
| 89 | let mut end = end; |
| 90 | if end < flat.len() |
| 91 | && let Some(space) = flat[start..end].iter().rposition(|c| *c == ' ') |
| 92 | && start + space > first |
| 93 | { |
| 94 | end = start + space; |
| 95 | } |
| 96 | let mut out = segments_of(&flat[start..end], &marked[start..end]); |
| 97 | if start > 0 { |
| 98 | prepend(&mut out, "…"); |
| 99 | } |
| 100 | if end < flat.len() { |
| 101 | append(&mut out, "…"); |
| 102 | } |
| 103 | out |
| 104 | } |
| 105 | |
| 106 | fn prepend(segments: &mut Vec<Segment>, text: &str) { |
| 107 | match segments.first_mut() { |
| 108 | Some(first) if !first.highlight => first.text.insert_str(0, text), |
| 109 | _ => segments.insert(0, Segment { text: text.to_owned(), highlight: false }), |
| 110 | } |
| 111 | } |
| 112 | |
| 113 | fn append(segments: &mut Vec<Segment>, text: &str) { |
| 114 | match segments.last_mut() { |
| 115 | Some(last) if !last.highlight => last.text.push_str(text), |
| 116 | _ => segments.push(Segment { text: text.to_owned(), highlight: false }), |
| 117 | } |
| 118 | } |
| 119 | |
| 120 | /// The lines of a file that matched, with the line before and after each, |
| 121 | /// at most [`CODE_LINES`]. `pieces` are parts of the file as indexed, each |
| 122 | /// with the number of its first line. When no line matched (a match on the |
| 123 | /// file's path), its first lines. |
| 124 | pub fn code(pieces: &[(u32, &str)], terms: &[String]) -> Vec<CodeLine> { |
| 125 | let folded_terms = folded(terms); |
| 126 | let mut lines: Vec<(u32, Vec<char>, Vec<bool>)> = Vec::new(); |
| 127 | for (start, text) in pieces { |
| 128 | for (offset, line) in text.lines().enumerate() { |
| 129 | let chars: Vec<char> = line.trim_end_matches('\r').chars().take(LINE_CHARS).collect(); |
| 130 | let marked = marks(&chars, &folded_terms); |
| 131 | lines.push((start + offset as u32, chars, marked)); |
| 132 | } |
| 133 | } |
| 134 | lines.sort_by_key(|(number, _, _)| *number); |
| 135 | lines.dedup_by_key(|(number, _, _)| *number); |
| 136 | let hits: Vec<usize> = lines |
| 137 | .iter() |
| 138 | .enumerate() |
| 139 | .filter(|(_, (_, _, marked))| marked.contains(&true)) |
| 140 | .map(|(index, _)| index) |
| 141 | .collect(); |
| 142 | let mut wanted: Vec<usize> = Vec::new(); |
| 143 | if hits.is_empty() { |
| 144 | wanted.extend(0..lines.len().min(3)); |
| 145 | } |
| 146 | for hit in hits { |
| 147 | for index in hit.saturating_sub(1)..=(hit + 1).min(lines.len().saturating_sub(1)) { |
| 148 | if !wanted.contains(&index) { |
| 149 | wanted.push(index); |
| 150 | } |
| 151 | } |
| 152 | if wanted.len() >= CODE_LINES { |
| 153 | break; |
| 154 | } |
| 155 | } |
| 156 | wanted.sort_unstable(); |
| 157 | wanted.truncate(CODE_LINES); |
| 158 | wanted |
| 159 | .into_iter() |
| 160 | .map(|index| { |
| 161 | let (number, chars, marked) = &lines[index]; |
| 162 | CodeLine { |
| 163 | number: *number, |
| 164 | parts: segments_of(chars, marked), |
| 165 | } |
| 166 | }) |
| 167 | .collect() |
| 168 | } |
| 169 | |
| 170 | /// The first line number that matched, for a link straight to it. |
| 171 | pub fn first_match(lines: &[CodeLine]) -> Option<u32> { |
| 172 | lines |
| 173 | .iter() |
| 174 | .find(|line| line.parts.iter().any(|part| part.highlight)) |
| 175 | .map(|line| line.number) |
| 176 | } |
| 177 | |
| 178 | #[cfg(test)] |
| 179 | mod tests { |
| 180 | use super::*; |
| 181 | |
| 182 | fn shown(segments: &[Segment]) -> String { |
| 183 | segments |
| 184 | .iter() |
| 185 | .map(|s| if s.highlight { format!("[{}]", s.text) } else { s.text.clone() }) |
| 186 | .collect() |
| 187 | } |
| 188 | |
| 189 | fn terms(list: &[&str]) -> Vec<String> { |
| 190 | list.iter().map(|t| t.to_string()).collect() |
| 191 | } |
| 192 | |
| 193 | #[test] |
| 194 | fn matches_are_marked_whatever_their_case() { |
| 195 | let out = highlight("Parse the parser, PARSE it", &terms(&["parse"])); |
| 196 | assert_eq!(shown(&out), "[Parse] the [parse]r, [PARSE] it"); |
| 197 | } |
| 198 | |
| 199 | #[test] |
| 200 | fn overlapping_terms_make_one_mark() { |
| 201 | let out = highlight("foobar", &terms(&["foo", "oba"])); |
| 202 | assert_eq!(shown(&out), "[fooba]r"); |
| 203 | } |
| 204 | |
| 205 | #[test] |
| 206 | fn nothing_to_mark_is_plain_text() { |
| 207 | assert_eq!(shown(&highlight("plain", &terms(&["zzz"]))), "plain"); |
| 208 | assert_eq!(shown(&highlight("plain", &[])), "plain"); |
| 209 | } |
| 210 | |
| 211 | #[test] |
| 212 | fn letters_outside_ascii_are_matched() { |
| 213 | let out = highlight("Ärger über Ölpreise", &terms(&["über", "öl"])); |
| 214 | assert_eq!(shown(&out), "Ärger [über] [Öl]preise"); |
| 215 | } |
| 216 | |
| 217 | #[test] |
| 218 | fn prose_is_a_window_around_the_first_match() { |
| 219 | let text = format!("{} the needle is here {}", "word ".repeat(100), "tail ".repeat(100)); |
| 220 | let out = prose(&text, &terms(&["needle"])); |
| 221 | let plain = shown(&out); |
| 222 | assert!(plain.starts_with('…') && plain.ends_with('…'), "{plain}"); |
| 223 | assert!(plain.contains("[needle]")); |
| 224 | assert!(plain.chars().count() <= PROSE_CHARS + 4); |
| 225 | } |
| 226 | |
| 227 | #[test] |
| 228 | fn short_prose_is_shown_whole() { |
| 229 | let out = prose("Fix the\nlogin redirect", &terms(&["login"])); |
| 230 | assert_eq!(shown(&out), "Fix the [login] redirect"); |
| 231 | assert!(prose("", &terms(&["x"])).is_empty()); |
| 232 | } |
| 233 | |
| 234 | #[test] |
| 235 | fn code_lines_are_numbered_with_context() { |
| 236 | let first = "use std::io;\n\nfn main() {\n let query = parse();\n}\n"; |
| 237 | let out = code(&[(1, first)], &terms(&["parse"])); |
| 238 | let numbers: Vec<u32> = out.iter().map(|line| line.number).collect(); |
| 239 | assert_eq!(numbers, vec![3, 4, 5]); |
| 240 | assert_eq!(shown(&out[1].parts), " let query = [parse]();"); |
| 241 | assert_eq!(first_match(&out), Some(4)); |
| 242 | } |
| 243 | |
| 244 | #[test] |
| 245 | fn code_from_later_pieces_keeps_its_numbers() { |
| 246 | let piece = "a\nb match\nc\n"; |
| 247 | let out = code(&[(121, piece)], &terms(&["match"])); |
| 248 | assert_eq!(out.iter().map(|l| l.number).collect::<Vec<_>>(), vec![121, 122, 123]); |
| 249 | } |
| 250 | |
| 251 | #[test] |
| 252 | fn a_path_match_shows_the_first_lines() { |
| 253 | let out = code(&[(1, "one\ntwo\nthree\nfour\n")], &terms(&["zzz"])); |
| 254 | assert_eq!(out.iter().map(|l| l.number).collect::<Vec<_>>(), vec![1, 2, 3]); |
| 255 | assert_eq!(first_match(&out), None); |
| 256 | } |
| 257 | |
| 258 | #[test] |
| 259 | fn many_matches_stop_at_the_limit() { |
| 260 | let text: String = (0..100).map(|n| format!("hit {n}\n")).collect(); |
| 261 | let out = code(&[(1, &text)], &terms(&["hit"])); |
| 262 | assert_eq!(out.len(), CODE_LINES); |
| 263 | } |
| 264 | } |