flagon-io/g1t

public

Where people and agents ship software together. The open-source git platform for the whole job: issues, agents, checks and deploys to the edge.

g1t/services/search/src/rules.rs

334 lines11,623 bytesCodeBlame
1//! What gets indexed, and how: which files are skipped, what language a
2//! file is in, and how a file is cut into pieces for the index. Pure, so
3//! the rules are tested apart from the store.
4
5/// Files larger than this are not indexed.
6pub const MAX_FILE_BYTES: u32 = 512 * 1024;
7/// Lines in one piece of a file, at most.
8pub const CHUNK_LINES: usize = 120;
9/// Bytes in one piece of a file, at most; a longer line is a piece of its own.
10pub const CHUNK_BYTES: usize = 16 * 1024;
11
12/// Directories never read: other people's code and what a build makes.
13/// Matched by name at any depth.
14pub const SKIP_DIRS: &[&str] = &[
15 ".git",
16 "node_modules",
17 "bower_components",
18 "jspm_packages",
19 "vendor",
20 "vendored",
21 "third_party",
22 "third-party",
23 "Pods",
24 "Carthage",
25 ".yarn",
26 ".pnpm-store",
27 ".venv",
28 "venv",
29 "__pycache__",
30 ".next",
31 ".nuxt",
32 ".svelte-kit",
33 ".output",
34 ".turbo",
35 ".wrangler",
36 "dist",
37 "target",
38 "coverage",
39];
40
41/// Lockfiles: long, generated, and never what anyone searches for.
42const LOCKFILES: &[&str] = &[
43 "package-lock.json",
44 "npm-shrinkwrap.json",
45 "yarn.lock",
46 "pnpm-lock.yaml",
47 "bun.lockb",
48 "bun.lock",
49 "deno.lock",
50 "cargo.lock",
51 "gemfile.lock",
52 "poetry.lock",
53 "pipfile.lock",
54 "uv.lock",
55 "pdm.lock",
56 "composer.lock",
57 "go.sum",
58 "go.work.sum",
59 "flake.lock",
60 "mix.lock",
61 "pubspec.lock",
62 "podfile.lock",
63 "package.resolved",
64 "packages.lock.json",
65 "gradle.lockfile",
66 "conan.lock",
67];
68
69/// Extensions of files that are not text, or not worth reading as text.
70const BINARY: &[&str] = &[
71 "png", "jpg", "jpeg", "gif", "webp", "avif", "ico", "icns", "bmp", "tif", "tiff", "psd", "ai", "sketch", "fig",
72 "pdf", "doc", "docx", "xls", "xlsx", "ppt", "pptx", "key", "numbers", "pages", "zip", "gz", "tgz", "bz2", "xz",
73 "zst", "7z", "rar", "tar", "jar", "war", "ear", "class", "o", "obj", "a", "lib", "so", "dylib", "dll", "exe",
74 "bin", "dat", "wasm", "pyc", "pyo", "woff", "woff2", "ttf", "otf", "eot", "mp3", "mp4", "m4a", "mov", "avi",
75 "wav", "flac", "ogg", "webm", "mkv", "sqlite", "sqlite3", "db", "parquet", "arrow", "pb", "onnx", "pt", "ckpt",
76 "safetensors", "npy", "npz", "glb", "gltf", "fbx", "blend", "iso", "dmg", "img", "deb", "rpm", "apk", "ipa",
77];
78
79/// Why a file is not indexed, judged by its path alone, or `None` to read it.
80pub fn skip_path(path: &str) -> Option<&'static str> {
81 let mut parts = path.split('/').collect::<Vec<_>>();
82 let name = parts.pop().unwrap_or_default();
83 if parts.iter().any(|dir| SKIP_DIRS.iter().any(|skip| skip.eq_ignore_ascii_case(dir))) {
84 return Some("vendored");
85 }
86 let lower = name.to_lowercase();
87 if LOCKFILES.contains(&lower.as_str()) || lower.ends_with(".lock") || lower.ends_with(".lockb") {
88 return Some("lockfile");
89 }
90 if let Some((_, extension)) = lower.rsplit_once('.')
91 && BINARY.contains(&extension)
92 {
93 return Some("binary");
94 }
95 if lower.ends_with(".min.js")
96 || lower.ends_with(".min.css")
97 || lower.ends_with(".min.mjs")
98 || lower.ends_with(".map")
99 || lower.ends_with(".bundle.js")
100 {
101 return Some("minified");
102 }
103 if lower == ".ds_store" || lower == "thumbs.db" {
104 return Some("binary");
105 }
106 None
107}
108
109/// Why a file's text is not indexed, or `None` to index it: too large, or
110/// minified (some line far longer than code is written).
111pub fn skip_text(text: &str) -> Option<&'static str> {
112 if text.len() > MAX_FILE_BYTES as usize {
113 return Some("too large");
114 }
115 if text.len() > 16 * 1024 && text.lines().any(|line| line.len() > 4096) {
116 return Some("minified");
117 }
118 None
119}
120
121/// The language a file is written in, by its name. Names are as people
122/// write them in `language:` (`rust`, `typescript`), lowercase.
123pub fn language(path: &str) -> Option<&'static str> {
124 let name = path.rsplit('/').next().unwrap_or(path).to_lowercase();
125 let by_name = match name.as_str() {
126 "dockerfile" | "containerfile" => Some("dockerfile"),
127 "makefile" | "gnumakefile" => Some("makefile"),
128 "cmakelists.txt" => Some("cmake"),
129 "gemfile" | "rakefile" => Some("ruby"),
130 "justfile" => Some("just"),
131 _ => None,
132 };
133 if by_name.is_some() {
134 return by_name;
135 }
136 if name.starts_with("dockerfile.") {
137 return Some("dockerfile");
138 }
139 let extension = name.rsplit_once('.').map(|(_, extension)| extension)?;
140 Some(match extension {
141 "rs" => "rust",
142 "ts" | "mts" | "cts" => "typescript",
143 "tsx" => "tsx",
144 "js" | "mjs" | "cjs" => "javascript",
145 "jsx" => "jsx",
146 "py" | "pyi" => "python",
147 "go" => "go",
148 "rb" => "ruby",
149 "java" => "java",
150 "kt" | "kts" => "kotlin",
151 "scala" => "scala",
152 "swift" => "swift",
153 "m" | "mm" => "objective-c",
154 "c" | "h" => "c",
155 "cc" | "cpp" | "cxx" | "hpp" | "hh" | "hxx" => "c++",
156 "cs" => "c#",
157 "fs" | "fsx" => "f#",
158 "php" => "php",
159 "pl" | "pm" => "perl",
160 "lua" => "lua",
161 "r" => "r",
162 "jl" => "julia",
163 "ex" | "exs" => "elixir",
164 "erl" | "hrl" => "erlang",
165 "hs" => "haskell",
166 "ml" | "mli" => "ocaml",
167 "clj" | "cljs" | "cljc" | "edn" => "clojure",
168 "dart" => "dart",
169 "zig" => "zig",
170 "nim" => "nim",
171 "v" => "v",
172 "sol" => "solidity",
173 "sh" | "bash" | "zsh" => "shell",
174 "fish" => "fish",
175 "ps1" | "psm1" => "powershell",
176 "sql" => "sql",
177 "graphql" | "gql" => "graphql",
178 "proto" => "protobuf",
179 "html" | "htm" => "html",
180 "css" => "css",
181 "scss" | "sass" => "scss",
182 "less" => "less",
183 "vue" => "vue",
184 "svelte" => "svelte",
185 "astro" => "astro",
186 "md" | "markdown" => "markdown",
187 "mdx" => "mdx",
188 "rst" => "restructuredtext",
189 "txt" => "text",
190 "json" | "jsonc" | "json5" => "json",
191 "yaml" | "yml" => "yaml",
192 "toml" => "toml",
193 "xml" | "svg" => "xml",
194 "ini" | "cfg" => "ini",
195 "tf" | "hcl" => "hcl",
196 "nix" => "nix",
197 "tex" => "tex",
198 "wgsl" => "wgsl",
199 "glsl" | "vert" | "frag" => "glsl",
200 _ => return None,
201 })
202}
203
204/// Languages that are prose or data rather than code, which do not count
205/// when saying what language a repository is written in.
206pub const DATA_LANGUAGES: &[&str] =
207 &["markdown", "mdx", "restructuredtext", "text", "json", "yaml", "toml", "xml", "ini", "html", "css"];
208
209/// The language a `language:` qualifier names, as [`language`] writes it:
210/// `ts` is `typescript`, `cpp` is `c++`, `js` is `javascript`.
211pub fn normalize_language(name: &str) -> String {
212 let lower = name.trim().to_lowercase();
213 match lower.as_str() {
214 "ts" => "typescript",
215 "js" => "javascript",
216 "py" => "python",
217 "rb" => "ruby",
218 "rs" => "rust",
219 "golang" => "go",
220 "cpp" | "cxx" => "c++",
221 "csharp" | "cs" => "c#",
222 "fsharp" => "f#",
223 "objc" | "objectivec" => "objective-c",
224 "bash" | "sh" | "zsh" => "shell",
225 "yml" => "yaml",
226 "md" => "markdown",
227 "kt" => "kotlin",
228 other => return other.to_owned(),
229 }
230 .to_owned()
231}
232
233/// A file cut into pieces for the index: each with the number of its first
234/// line (from 1) and its text, at most [`CHUNK_LINES`] lines and, unless one
235/// line is longer, [`CHUNK_BYTES`] bytes.
236pub fn chunks(text: &str) -> Vec<(u32, String)> {
237 let mut out = Vec::new();
238 let mut current = String::new();
239 let mut start = 1u32;
240 let mut lines = 0usize;
241 for (index, line) in text.split_inclusive('\n').enumerate() {
242 if lines > 0 && (lines >= CHUNK_LINES || current.len() + line.len() > CHUNK_BYTES) {
243 out.push((start, std::mem::take(&mut current)));
244 start = index as u32 + 1;
245 lines = 0;
246 }
247 current.push_str(line);
248 lines += 1;
249 }
250 if !current.is_empty() {
251 out.push((start, current));
252 }
253 out
254}
255
256#[cfg(test)]
257mod tests {
258 use super::*;
259
260 #[test]
261 fn vendored_directories_are_skipped_at_any_depth() {
262 assert_eq!(skip_path("node_modules/react/index.js"), Some("vendored"));
263 assert_eq!(skip_path("apps/web/node_modules/x/y.ts"), Some("vendored"));
264 assert_eq!(skip_path("crates/foo/vendor/lib.c"), Some("vendored"));
265 assert_eq!(skip_path("target/release/build.rs"), Some("vendored"));
266 // A file named like a skipped directory is still read.
267 assert_eq!(skip_path("src/vendor.rs"), None);
268 assert_eq!(skip_path("src/dist.ts"), None);
269 }
270
271 #[test]
272 fn lockfiles_are_skipped() {
273 for path in ["package-lock.json", "web/yarn.lock", "Cargo.lock", "go.sum", "pnpm-lock.yaml", "x/flake.lock", "bun.lockb"] {
274 assert_eq!(skip_path(path), Some("lockfile"), "{path}");
275 }
276 assert_eq!(skip_path("Cargo.toml"), None);
277 assert_eq!(skip_path("package.json"), None);
278 }
279
280 #[test]
281 fn binaries_and_minified_files_are_skipped() {
282 assert_eq!(skip_path("public/logo.PNG"), Some("binary"));
283 assert_eq!(skip_path("fonts/inter.woff2"), Some("binary"));
284 assert_eq!(skip_path("build/app.wasm"), Some("binary"));
285 assert_eq!(skip_path("static/app.min.js"), Some("minified"));
286 assert_eq!(skip_path("static/app.js.map"), Some("minified"));
287 assert_eq!(skip_path("src/main.rs"), None);
288 assert_eq!(skip_path("Makefile"), None);
289 }
290
291 #[test]
292 fn large_and_minified_text_is_skipped() {
293 assert_eq!(skip_text("fn main() {}\n"), None);
294 let big = "a\n".repeat(300 * 1024);
295 assert_eq!(skip_text(&big), Some("too large"));
296 let one_line = "x".repeat(20 * 1024);
297 assert_eq!(skip_text(&one_line), Some("minified"));
298 // Exactly the limit is kept.
299 let edge = "a".repeat(MAX_FILE_BYTES as usize - 1) + "\n";
300 assert_eq!(skip_text(&edge), Some("minified"));
301 let edge_lines = "abcdefg\n".repeat(MAX_FILE_BYTES as usize / 8);
302 assert_eq!(skip_text(&edge_lines), None);
303 }
304
305 #[test]
306 fn languages_come_from_names() {
307 assert_eq!(language("src/main.rs"), Some("rust"));
308 assert_eq!(language("app/routes/x.tsx"), Some("tsx"));
309 assert_eq!(language("Dockerfile"), Some("dockerfile"));
310 assert_eq!(language("docs/README.md"), Some("markdown"));
311 assert_eq!(language("LICENSE"), None);
312 assert_eq!(normalize_language("TS"), "typescript");
313 assert_eq!(normalize_language("cpp"), "c++");
314 assert_eq!(normalize_language("Rust"), "rust");
315 }
316
317 #[test]
318 fn chunks_keep_line_numbers() {
319 let text: String = (1..=250).map(|n| format!("line {n}\n")).collect();
320 let pieces = chunks(&text);
321 assert_eq!(pieces.iter().map(|(start, _)| *start).collect::<Vec<_>>(), vec![1, 121, 241]);
322 assert!(pieces[1].1.starts_with("line 121\n"));
323 assert_eq!(pieces.iter().map(|(_, text)| text.as_str()).collect::<String>(), text);
324 }
325
326 #[test]
327 fn chunks_stay_under_their_size() {
328 let line = format!("{}\n", "y".repeat(1000));
329 let text = line.repeat(40);
330 let pieces = chunks(&text);
331 assert!(pieces.iter().all(|(_, piece)| piece.len() <= CHUNK_BYTES));
332 assert_eq!(pieces[1].0, 17);
333 }
334}