g1t/services/search/src/rules.rs
| 1 | //! What gets indexed, and how: which files are skipped, what language a |
| 2 | //! file is in, and how a file is cut into pieces for the index. Pure, so |
| 3 | //! the rules are tested apart from the store. |
| 4 | |
| 5 | /// Files larger than this are not indexed. |
| 6 | pub const MAX_FILE_BYTES: u32 = 512 * 1024; |
| 7 | /// Lines in one piece of a file, at most. |
| 8 | pub const CHUNK_LINES: usize = 120; |
| 9 | /// Bytes in one piece of a file, at most; a longer line is a piece of its own. |
| 10 | pub const CHUNK_BYTES: usize = 16 * 1024; |
| 11 | |
| 12 | /// Directories never read: other people's code and what a build makes. |
| 13 | /// Matched by name at any depth. |
| 14 | pub const SKIP_DIRS: &[&str] = &[ |
| 15 | ".git", |
| 16 | "node_modules", |
| 17 | "bower_components", |
| 18 | "jspm_packages", |
| 19 | "vendor", |
| 20 | "vendored", |
| 21 | "third_party", |
| 22 | "third-party", |
| 23 | "Pods", |
| 24 | "Carthage", |
| 25 | ".yarn", |
| 26 | ".pnpm-store", |
| 27 | ".venv", |
| 28 | "venv", |
| 29 | "__pycache__", |
| 30 | ".next", |
| 31 | ".nuxt", |
| 32 | ".svelte-kit", |
| 33 | ".output", |
| 34 | ".turbo", |
| 35 | ".wrangler", |
| 36 | "dist", |
| 37 | "target", |
| 38 | "coverage", |
| 39 | ]; |
| 40 | |
| 41 | /// Lockfiles: long, generated, and never what anyone searches for. |
| 42 | const LOCKFILES: &[&str] = &[ |
| 43 | "package-lock.json", |
| 44 | "npm-shrinkwrap.json", |
| 45 | "yarn.lock", |
| 46 | "pnpm-lock.yaml", |
| 47 | "bun.lockb", |
| 48 | "bun.lock", |
| 49 | "deno.lock", |
| 50 | "cargo.lock", |
| 51 | "gemfile.lock", |
| 52 | "poetry.lock", |
| 53 | "pipfile.lock", |
| 54 | "uv.lock", |
| 55 | "pdm.lock", |
| 56 | "composer.lock", |
| 57 | "go.sum", |
| 58 | "go.work.sum", |
| 59 | "flake.lock", |
| 60 | "mix.lock", |
| 61 | "pubspec.lock", |
| 62 | "podfile.lock", |
| 63 | "package.resolved", |
| 64 | "packages.lock.json", |
| 65 | "gradle.lockfile", |
| 66 | "conan.lock", |
| 67 | ]; |
| 68 | |
| 69 | /// Extensions of files that are not text, or not worth reading as text. |
| 70 | const BINARY: &[&str] = &[ |
| 71 | "png", "jpg", "jpeg", "gif", "webp", "avif", "ico", "icns", "bmp", "tif", "tiff", "psd", "ai", "sketch", "fig", |
| 72 | "pdf", "doc", "docx", "xls", "xlsx", "ppt", "pptx", "key", "numbers", "pages", "zip", "gz", "tgz", "bz2", "xz", |
| 73 | "zst", "7z", "rar", "tar", "jar", "war", "ear", "class", "o", "obj", "a", "lib", "so", "dylib", "dll", "exe", |
| 74 | "bin", "dat", "wasm", "pyc", "pyo", "woff", "woff2", "ttf", "otf", "eot", "mp3", "mp4", "m4a", "mov", "avi", |
| 75 | "wav", "flac", "ogg", "webm", "mkv", "sqlite", "sqlite3", "db", "parquet", "arrow", "pb", "onnx", "pt", "ckpt", |
| 76 | "safetensors", "npy", "npz", "glb", "gltf", "fbx", "blend", "iso", "dmg", "img", "deb", "rpm", "apk", "ipa", |
| 77 | ]; |
| 78 | |
| 79 | /// Why a file is not indexed, judged by its path alone, or `None` to read it. |
| 80 | pub fn skip_path(path: &str) -> Option<&'static str> { |
| 81 | let mut parts = path.split('/').collect::<Vec<_>>(); |
| 82 | let name = parts.pop().unwrap_or_default(); |
| 83 | if parts.iter().any(|dir| SKIP_DIRS.iter().any(|skip| skip.eq_ignore_ascii_case(dir))) { |
| 84 | return Some("vendored"); |
| 85 | } |
| 86 | let lower = name.to_lowercase(); |
| 87 | if LOCKFILES.contains(&lower.as_str()) || lower.ends_with(".lock") || lower.ends_with(".lockb") { |
| 88 | return Some("lockfile"); |
| 89 | } |
| 90 | if let Some((_, extension)) = lower.rsplit_once('.') |
| 91 | && BINARY.contains(&extension) |
| 92 | { |
| 93 | return Some("binary"); |
| 94 | } |
| 95 | if lower.ends_with(".min.js") |
| 96 | || lower.ends_with(".min.css") |
| 97 | || lower.ends_with(".min.mjs") |
| 98 | || lower.ends_with(".map") |
| 99 | || lower.ends_with(".bundle.js") |
| 100 | { |
| 101 | return Some("minified"); |
| 102 | } |
| 103 | if lower == ".ds_store" || lower == "thumbs.db" { |
| 104 | return Some("binary"); |
| 105 | } |
| 106 | None |
| 107 | } |
| 108 | |
| 109 | /// Why a file's text is not indexed, or `None` to index it: too large, or |
| 110 | /// minified (some line far longer than code is written). |
| 111 | pub fn skip_text(text: &str) -> Option<&'static str> { |
| 112 | if text.len() > MAX_FILE_BYTES as usize { |
| 113 | return Some("too large"); |
| 114 | } |
| 115 | if text.len() > 16 * 1024 && text.lines().any(|line| line.len() > 4096) { |
| 116 | return Some("minified"); |
| 117 | } |
| 118 | None |
| 119 | } |
| 120 | |
| 121 | /// The language a file is written in, by its name. Names are as people |
| 122 | /// write them in `language:` (`rust`, `typescript`), lowercase. |
| 123 | pub fn language(path: &str) -> Option<&'static str> { |
| 124 | let name = path.rsplit('/').next().unwrap_or(path).to_lowercase(); |
| 125 | let by_name = match name.as_str() { |
| 126 | "dockerfile" | "containerfile" => Some("dockerfile"), |
| 127 | "makefile" | "gnumakefile" => Some("makefile"), |
| 128 | "cmakelists.txt" => Some("cmake"), |
| 129 | "gemfile" | "rakefile" => Some("ruby"), |
| 130 | "justfile" => Some("just"), |
| 131 | _ => None, |
| 132 | }; |
| 133 | if by_name.is_some() { |
| 134 | return by_name; |
| 135 | } |
| 136 | if name.starts_with("dockerfile.") { |
| 137 | return Some("dockerfile"); |
| 138 | } |
| 139 | let extension = name.rsplit_once('.').map(|(_, extension)| extension)?; |
| 140 | Some(match extension { |
| 141 | "rs" => "rust", |
| 142 | "ts" | "mts" | "cts" => "typescript", |
| 143 | "tsx" => "tsx", |
| 144 | "js" | "mjs" | "cjs" => "javascript", |
| 145 | "jsx" => "jsx", |
| 146 | "py" | "pyi" => "python", |
| 147 | "go" => "go", |
| 148 | "rb" => "ruby", |
| 149 | "java" => "java", |
| 150 | "kt" | "kts" => "kotlin", |
| 151 | "scala" => "scala", |
| 152 | "swift" => "swift", |
| 153 | "m" | "mm" => "objective-c", |
| 154 | "c" | "h" => "c", |
| 155 | "cc" | "cpp" | "cxx" | "hpp" | "hh" | "hxx" => "c++", |
| 156 | "cs" => "c#", |
| 157 | "fs" | "fsx" => "f#", |
| 158 | "php" => "php", |
| 159 | "pl" | "pm" => "perl", |
| 160 | "lua" => "lua", |
| 161 | "r" => "r", |
| 162 | "jl" => "julia", |
| 163 | "ex" | "exs" => "elixir", |
| 164 | "erl" | "hrl" => "erlang", |
| 165 | "hs" => "haskell", |
| 166 | "ml" | "mli" => "ocaml", |
| 167 | "clj" | "cljs" | "cljc" | "edn" => "clojure", |
| 168 | "dart" => "dart", |
| 169 | "zig" => "zig", |
| 170 | "nim" => "nim", |
| 171 | "v" => "v", |
| 172 | "sol" => "solidity", |
| 173 | "sh" | "bash" | "zsh" => "shell", |
| 174 | "fish" => "fish", |
| 175 | "ps1" | "psm1" => "powershell", |
| 176 | "sql" => "sql", |
| 177 | "graphql" | "gql" => "graphql", |
| 178 | "proto" => "protobuf", |
| 179 | "html" | "htm" => "html", |
| 180 | "css" => "css", |
| 181 | "scss" | "sass" => "scss", |
| 182 | "less" => "less", |
| 183 | "vue" => "vue", |
| 184 | "svelte" => "svelte", |
| 185 | "astro" => "astro", |
| 186 | "md" | "markdown" => "markdown", |
| 187 | "mdx" => "mdx", |
| 188 | "rst" => "restructuredtext", |
| 189 | "txt" => "text", |
| 190 | "json" | "jsonc" | "json5" => "json", |
| 191 | "yaml" | "yml" => "yaml", |
| 192 | "toml" => "toml", |
| 193 | "xml" | "svg" => "xml", |
| 194 | "ini" | "cfg" => "ini", |
| 195 | "tf" | "hcl" => "hcl", |
| 196 | "nix" => "nix", |
| 197 | "tex" => "tex", |
| 198 | "wgsl" => "wgsl", |
| 199 | "glsl" | "vert" | "frag" => "glsl", |
| 200 | _ => return None, |
| 201 | }) |
| 202 | } |
| 203 | |
| 204 | /// Languages that are prose or data rather than code, which do not count |
| 205 | /// when saying what language a repository is written in. |
| 206 | pub const DATA_LANGUAGES: &[&str] = |
| 207 | &["markdown", "mdx", "restructuredtext", "text", "json", "yaml", "toml", "xml", "ini", "html", "css"]; |
| 208 | |
| 209 | /// The language a `language:` qualifier names, as [`language`] writes it: |
| 210 | /// `ts` is `typescript`, `cpp` is `c++`, `js` is `javascript`. |
| 211 | pub fn normalize_language(name: &str) -> String { |
| 212 | let lower = name.trim().to_lowercase(); |
| 213 | match lower.as_str() { |
| 214 | "ts" => "typescript", |
| 215 | "js" => "javascript", |
| 216 | "py" => "python", |
| 217 | "rb" => "ruby", |
| 218 | "rs" => "rust", |
| 219 | "golang" => "go", |
| 220 | "cpp" | "cxx" => "c++", |
| 221 | "csharp" | "cs" => "c#", |
| 222 | "fsharp" => "f#", |
| 223 | "objc" | "objectivec" => "objective-c", |
| 224 | "bash" | "sh" | "zsh" => "shell", |
| 225 | "yml" => "yaml", |
| 226 | "md" => "markdown", |
| 227 | "kt" => "kotlin", |
| 228 | other => return other.to_owned(), |
| 229 | } |
| 230 | .to_owned() |
| 231 | } |
| 232 | |
| 233 | /// A file cut into pieces for the index: each with the number of its first |
| 234 | /// line (from 1) and its text, at most [`CHUNK_LINES`] lines and, unless one |
| 235 | /// line is longer, [`CHUNK_BYTES`] bytes. |
| 236 | pub fn chunks(text: &str) -> Vec<(u32, String)> { |
| 237 | let mut out = Vec::new(); |
| 238 | let mut current = String::new(); |
| 239 | let mut start = 1u32; |
| 240 | let mut lines = 0usize; |
| 241 | for (index, line) in text.split_inclusive('\n').enumerate() { |
| 242 | if lines > 0 && (lines >= CHUNK_LINES || current.len() + line.len() > CHUNK_BYTES) { |
| 243 | out.push((start, std::mem::take(&mut current))); |
| 244 | start = index as u32 + 1; |
| 245 | lines = 0; |
| 246 | } |
| 247 | current.push_str(line); |
| 248 | lines += 1; |
| 249 | } |
| 250 | if !current.is_empty() { |
| 251 | out.push((start, current)); |
| 252 | } |
| 253 | out |
| 254 | } |
| 255 | |
| 256 | #[cfg(test)] |
| 257 | mod tests { |
| 258 | use super::*; |
| 259 | |
| 260 | #[test] |
| 261 | fn vendored_directories_are_skipped_at_any_depth() { |
| 262 | assert_eq!(skip_path("node_modules/react/index.js"), Some("vendored")); |
| 263 | assert_eq!(skip_path("apps/web/node_modules/x/y.ts"), Some("vendored")); |
| 264 | assert_eq!(skip_path("crates/foo/vendor/lib.c"), Some("vendored")); |
| 265 | assert_eq!(skip_path("target/release/build.rs"), Some("vendored")); |
| 266 | // A file named like a skipped directory is still read. |
| 267 | assert_eq!(skip_path("src/vendor.rs"), None); |
| 268 | assert_eq!(skip_path("src/dist.ts"), None); |
| 269 | } |
| 270 | |
| 271 | #[test] |
| 272 | fn lockfiles_are_skipped() { |
| 273 | for path in ["package-lock.json", "web/yarn.lock", "Cargo.lock", "go.sum", "pnpm-lock.yaml", "x/flake.lock", "bun.lockb"] { |
| 274 | assert_eq!(skip_path(path), Some("lockfile"), "{path}"); |
| 275 | } |
| 276 | assert_eq!(skip_path("Cargo.toml"), None); |
| 277 | assert_eq!(skip_path("package.json"), None); |
| 278 | } |
| 279 | |
| 280 | #[test] |
| 281 | fn binaries_and_minified_files_are_skipped() { |
| 282 | assert_eq!(skip_path("public/logo.PNG"), Some("binary")); |
| 283 | assert_eq!(skip_path("fonts/inter.woff2"), Some("binary")); |
| 284 | assert_eq!(skip_path("build/app.wasm"), Some("binary")); |
| 285 | assert_eq!(skip_path("static/app.min.js"), Some("minified")); |
| 286 | assert_eq!(skip_path("static/app.js.map"), Some("minified")); |
| 287 | assert_eq!(skip_path("src/main.rs"), None); |
| 288 | assert_eq!(skip_path("Makefile"), None); |
| 289 | } |
| 290 | |
| 291 | #[test] |
| 292 | fn large_and_minified_text_is_skipped() { |
| 293 | assert_eq!(skip_text("fn main() {}\n"), None); |
| 294 | let big = "a\n".repeat(300 * 1024); |
| 295 | assert_eq!(skip_text(&big), Some("too large")); |
| 296 | let one_line = "x".repeat(20 * 1024); |
| 297 | assert_eq!(skip_text(&one_line), Some("minified")); |
| 298 | // Exactly the limit is kept. |
| 299 | let edge = "a".repeat(MAX_FILE_BYTES as usize - 1) + "\n"; |
| 300 | assert_eq!(skip_text(&edge), Some("minified")); |
| 301 | let edge_lines = "abcdefg\n".repeat(MAX_FILE_BYTES as usize / 8); |
| 302 | assert_eq!(skip_text(&edge_lines), None); |
| 303 | } |
| 304 | |
| 305 | #[test] |
| 306 | fn languages_come_from_names() { |
| 307 | assert_eq!(language("src/main.rs"), Some("rust")); |
| 308 | assert_eq!(language("app/routes/x.tsx"), Some("tsx")); |
| 309 | assert_eq!(language("Dockerfile"), Some("dockerfile")); |
| 310 | assert_eq!(language("docs/README.md"), Some("markdown")); |
| 311 | assert_eq!(language("LICENSE"), None); |
| 312 | assert_eq!(normalize_language("TS"), "typescript"); |
| 313 | assert_eq!(normalize_language("cpp"), "c++"); |
| 314 | assert_eq!(normalize_language("Rust"), "rust"); |
| 315 | } |
| 316 | |
| 317 | #[test] |
| 318 | fn chunks_keep_line_numbers() { |
| 319 | let text: String = (1..=250).map(|n| format!("line {n}\n")).collect(); |
| 320 | let pieces = chunks(&text); |
| 321 | assert_eq!(pieces.iter().map(|(start, _)| *start).collect::<Vec<_>>(), vec![1, 121, 241]); |
| 322 | assert!(pieces[1].1.starts_with("line 121\n")); |
| 323 | assert_eq!(pieces.iter().map(|(_, text)| text.as_str()).collect::<String>(), text); |
| 324 | } |
| 325 | |
| 326 | #[test] |
| 327 | fn chunks_stay_under_their_size() { |
| 328 | let line = format!("{}\n", "y".repeat(1000)); |
| 329 | let text = line.repeat(40); |
| 330 | let pieces = chunks(&text); |
| 331 | assert!(pieces.iter().all(|(_, piece)| piece.len() <= CHUNK_BYTES)); |
| 332 | assert_eq!(pieces[1].0, 17); |
| 333 | } |
| 334 | } |