Skip to content

g1t/crates/scan/src/sarif.rs

670 lines29,425 bytesCodeBlame
1//! Code scanning results in SARIF 2.1.0, the format static analysis tools
2//! write: reading an upload (gzipped, base64-encoded JSON), turning each
3//! result into an alert with its rule, severity, message and location, and
4//! fingerprinting it so the same problem found by the next run is the same
5//! alert, and one no longer reported is fixed.
6
7use std::collections::{BTreeMap, BTreeSet, HashMap};
8
9use serde_json::Value;
10use sha2::{Digest, Sha256};
11
12use crate::osv::Severity;
13
14/// The largest upload, gzipped and base64-encoded.
15pub const MAX_UPLOAD_BYTES: usize = 10 * 1024 * 1024;
16/// The largest SARIF document once unzipped.
17pub const MAX_SARIF_BYTES: usize = 40 * 1024 * 1024;
18/// Runs one upload may hold.
19pub const MAX_RUNS: usize = 20;
20/// Results kept from one upload; the rest are counted and dropped.
21pub const MAX_RESULTS: usize = 5_000;
22/// The longest message kept.
23const MAX_MESSAGE_CHARS: usize = 2_000;
24
25/// How a tool rates a result.
26#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)]
27pub enum Level {
28 None,
29 Note,
30 Warning,
31 Error,
32}
33
34impl Level {
35 pub fn as_str(self) -> &'static str {
36 match self {
37 Level::Error => "error",
38 Level::Warning => "warning",
39 Level::Note => "note",
40 Level::None => "none",
41 }
42 }
43
44 pub fn parse(text: &str) -> Option<Level> {
45 Some(match text {
46 "error" => Level::Error,
47 "warning" => Level::Warning,
48 "note" => Level::Note,
49 "none" => Level::None,
50 _ => return None,
51 })
52 }
53}
54
55/// Where a result is.
56#[derive(Clone, Debug, Default, PartialEq, Eq)]
57pub struct Location {
58 /// From the repository's root.
59 pub path: String,
60 pub start_line: u32,
61 pub end_line: u32,
62 pub start_column: Option<u32>,
63 pub end_column: Option<u32>,
64}
65
66/// One result, ready to be an alert.
67#[derive(Clone, Debug, PartialEq)]
68pub struct Finding {
69 pub rule_id: String,
70 pub rule_name: Option<String>,
71 pub rule_description: Option<String>,
72 /// Help for the rule, as markdown or text.
73 pub help: Option<String>,
74 pub help_uri: Option<String>,
75 pub tags: Vec<String>,
76 pub level: Level,
77 /// From the rule's `security-severity` score, for security rules.
78 pub security_severity: Option<Severity>,
79 pub message: String,
80 /// The primary location; absent for results about the whole project.
81 pub location: Option<Location>,
82 /// Names this result across runs (see [`fingerprint`]).
83 pub fingerprint: String,
84}
85
86impl Finding {
87 /// The severity people sort by: the security severity when the rule has
88 /// one, else the level (error high, warning medium, note low).
89 pub fn severity(&self) -> Severity {
90 self.security_severity.unwrap_or(match self.level {
91 Level::Error => Severity::High,
92 Level::Warning => Severity::Medium,
93 Level::Note => Severity::Low,
94 Level::None => Severity::Unknown,
95 })
96 }
97}
98
99/// One run of one tool.
100#[derive(Clone, Debug, PartialEq)]
101pub struct Run {
102 pub tool: String,
103 pub tool_version: Option<String>,
104 /// Which analysis this is, when a repository runs several of one tool
105 /// (one per language, say): from the upload, or the run's
106 /// `automationDetails.id` up to its last `/`, or the tool's name.
107 pub category: String,
108 pub findings: Vec<Finding>,
109 /// Results past [`MAX_RESULTS`], not kept.
110 pub dropped: usize,
111}
112
113/// What reading an upload can find wrong, in words for the person.
114pub type Problem = String;
115
116/// Decodes standard base64, ignoring line breaks.
117pub fn base64_decode(text: &str) -> Result<Vec<u8>, Problem> {
118 let mut out = Vec::with_capacity(text.len() * 3 / 4);
119 let (mut buffer, mut bits) = (0u32, 0u32);
120 for byte in text.bytes() {
121 let value = match byte {
122 b'A'..=b'Z' => byte - b'A',
123 b'a'..=b'z' => byte - b'a' + 26,
124 b'0'..=b'9' => byte - b'0' + 52,
125 b'+' | b'-' => 62,
126 b'/' | b'_' => 63,
127 b'=' | b'\n' | b'\r' | b' ' | b'\t' => continue,
128 _ => return Err("The sarif field is not base64.".to_owned()),
129 };
130 buffer = (buffer << 6) | u32::from(value);
131 bits += 6;
132 if bits >= 8 {
133 bits -= 8;
134 out.push((buffer >> bits) as u8);
135 }
136 }
137 Ok(out)
138}
139
140/// Unzips a gzip stream (RFC 1952) of one member, up to `limit` bytes.
141pub fn gunzip(bytes: &[u8], limit: usize) -> Result<Vec<u8>, Problem> {
142 const FEXTRA: u8 = 4;
143 const FNAME: u8 = 8;
144 const FCOMMENT: u8 = 16;
145 const FHCRC: u8 = 2;
146 if bytes.len() < 18 || bytes[0] != 0x1f || bytes[1] != 0x8b || bytes[2] != 8 {
147 return Err("The sarif field is not gzipped.".to_owned());
148 }
149 let flags = bytes[3];
150 let mut at = 10;
151 let truncated = || "The gzipped SARIF is truncated.".to_owned();
152 if flags & FEXTRA != 0 {
153 let length = usize::from(u16::from_le_bytes([*bytes.get(at).ok_or_else(truncated)?, *bytes.get(at + 1).ok_or_else(truncated)?]));
154 at += 2 + length;
155 }
156 for flag in [FNAME, FCOMMENT] {
157 if flags & flag != 0 {
158 let end = bytes.get(at..).and_then(|rest| rest.iter().position(|b| *b == 0)).ok_or_else(truncated)?;
159 at += end + 1;
160 }
161 }
162 if flags & FHCRC != 0 {
163 at += 2;
164 }
165 let body = bytes.get(at..bytes.len().saturating_sub(8)).ok_or_else(truncated)?;
166 miniz_oxide::inflate::decompress_to_vec_with_limit(body, limit).map_err(|error| match error.status {
167 miniz_oxide::inflate::TINFLStatus::HasMoreOutput => {
168 format!("The SARIF is larger than {} MB unzipped.", limit / (1024 * 1024))
169 }
170 _ => "The gzipped SARIF could not be unzipped.".to_owned(),
171 })
172}
173
174/// The SARIF document an upload carries: `sarif` is gzipped and base64
175/// encoded, as uploads conventionally are. Plain base64 JSON is accepted
176/// too.
177pub fn decode_upload(sarif: &str) -> Result<String, Problem> {
178 if sarif.len() > MAX_UPLOAD_BYTES {
179 return Err(format!("The upload is larger than {} MB.", MAX_UPLOAD_BYTES / (1024 * 1024)));
180 }
181 let bytes = base64_decode(sarif.trim())?;
182 let json = if bytes.starts_with(&[0x1f, 0x8b]) {
183 gunzip(&bytes, MAX_SARIF_BYTES)?
184 } else if bytes.len() > MAX_SARIF_BYTES {
185 return Err(format!("The SARIF is larger than {} MB.", MAX_SARIF_BYTES / (1024 * 1024)));
186 } else {
187 bytes
188 };
189 String::from_utf8(json).map_err(|_| "The SARIF is not UTF-8 text.".to_owned())
190}
191
192fn text_of(value: &Value) -> Option<String> {
193 value.get("text").and_then(Value::as_str).map(str::to_owned)
194}
195
196/// A message with `{0}`-style placeholders filled from its arguments.
197fn fill(template: &str, arguments: &[Value]) -> String {
198 let mut out = template.to_owned();
199 for (index, argument) in arguments.iter().enumerate() {
200 let value = argument.as_str().map(str::to_owned).unwrap_or_else(|| argument.to_string());
201 out = out.replace(&format!("{{{index}}}"), &value);
202 }
203 out
204}
205
206/// Percent-decoding for a URI's path.
207fn percent_decode(text: &str) -> String {
208 let bytes = text.as_bytes();
209 let mut out = Vec::with_capacity(bytes.len());
210 let mut at = 0;
211 while at < bytes.len() {
212 if bytes[at] == b'%'
213 && let Some(hex) = text.get(at + 1..at + 3)
214 && let Ok(byte) = u8::from_str_radix(hex, 16)
215 {
216 out.push(byte);
217 at += 3;
218 continue;
219 }
220 out.push(bytes[at]);
221 at += 1;
222 }
223 String::from_utf8_lossy(&out).into_owned()
224}
225
226/// A result's file, from the repository's root: a `file://` URI under the
227/// checkout (`checkout`, or the run's `%SRCROOT%`) loses that prefix, and
228/// a relative one its leading `./`.
229pub fn repository_path(uri: &str, roots: &[String]) -> String {
230 let mut path = percent_decode(uri);
231 if let Some(rest) = path.strip_prefix("file://") {
232 path = rest.to_owned();
233 }
234 for root in roots {
235 let root = percent_decode(root.strip_prefix("file://").unwrap_or(root));
236 let root = root.trim_end_matches('/');
237 if !root.is_empty() && let Some(rest) = path.strip_prefix(root) {
238 path = rest.to_owned();
239 break;
240 }
241 }
242 path.trim_start_matches("./").trim_start_matches('/').replace('\\', "/")
243}
244
245/// Names a result across runs. A tool's own `primaryLocationLineHash` (or
246/// another partial fingerprint) is used when it gives one, since it
247/// survives lines moving; otherwise the rule, file and the code it points
248/// at (or its message), so editing elsewhere in the file does not make it a
249/// new alert. `occurrence` tells apart identical results in one file.
250pub fn fingerprint(tool: &str, rule: &str, path: &str, partial: Option<&str>, content: &str, occurrence: usize) -> String {
251 let basis = match partial {
252 Some(partial) => format!("{tool}\n{rule}\n{path}\npartial:{partial}"),
253 None => format!("{tool}\n{rule}\n{path}\ncontent:{}\n{occurrence}", content.split_whitespace().collect::<Vec<_>>().join(" ")),
254 };
255 let digest = Sha256::digest(basis.as_bytes());
256 digest[..16].iter().map(|byte| format!("{byte:02x}")).collect()
257}
258
259/// The rules of a tool component, by id and by index.
260struct Rules<'a> {
261 list: Vec<&'a Value>,
262 by_id: HashMap<&'a str, &'a Value>,
263}
264
265impl<'a> Rules<'a> {
266 fn of(component: &'a Value) -> Rules<'a> {
267 let list: Vec<&Value> = component["rules"].as_array().map(|rules| rules.iter().collect()).unwrap_or_default();
268 let by_id = list.iter().filter_map(|rule| Some((rule["id"].as_str()?, *rule))).collect();
269 Rules { list, by_id }
270 }
271}
272
273/// The `security-severity` a rule's properties give, as a severity.
274fn security_severity(rule: Option<&Value>) -> Option<Severity> {
275 let raw = &rule?["properties"]["security-severity"];
276 let score = raw.as_f64().or_else(|| raw.as_str()?.trim().parse().ok())?;
277 Some(Severity::from_score(score)).filter(|severity| *severity != Severity::Unknown)
278}
279
280/// Reads a SARIF 2.1.0 document into its runs. `category` and `checkout`
281/// come from the upload, when it gives them.
282pub fn parse(text: &str, category: Option<&str>, checkout: Option<&str>) -> Result<Vec<Run>, Problem> {
283 let document: Value = serde_json::from_str(text).map_err(|error| format!("The SARIF is not valid JSON: {error}"))?;
284 let version = document["version"].as_str().unwrap_or_default();
285 if version != "2.1.0" {
286 return Err(format!("Only SARIF 2.1.0 is read; this document says {}.", if version.is_empty() { "no version" } else { version }));
287 }
288 let runs = document["runs"].as_array().ok_or("The SARIF has no runs.")?;
289 if runs.len() > MAX_RUNS {
290 return Err(format!("The SARIF has {} runs; at most {MAX_RUNS} are read per upload.", runs.len()));
291 }
292 let mut out = Vec::new();
293 let mut kept = 0usize;
294 for run in runs {
295 let driver = &run["tool"]["driver"];
296 let tool = driver["name"].as_str().filter(|name| !name.trim().is_empty()).ok_or("A run names no tool.")?.trim().to_owned();
297 let tool_version = driver["semanticVersion"].as_str().or(driver["version"].as_str()).map(str::to_owned);
298 let category = category
299 .map(str::to_owned)
300 .filter(|category| !category.trim().is_empty())
301 .or_else(|| {
302 let id = run["automationDetails"]["id"].as_str()?;
303 // `category/run-id`: the category is what precedes the last slash.
304 id.rsplit_once('/').map(|(category, _)| category.to_owned()).filter(|category| !category.is_empty())
305 })
306 .unwrap_or_else(|| tool.clone());
307 let mut roots: Vec<String> = Vec::new();
308 if let Some(checkout) = checkout {
309 roots.push(checkout.to_owned());
310 }
311 if let Some(bases) = run["originalUriBaseIds"].as_object() {
312 roots.extend(bases.values().filter_map(|base| base["uri"].as_str().map(str::to_owned)));
313 }
314 let driver_rules = Rules::of(driver);
315 let extensions: Vec<Rules> = run["tool"]["extensions"].as_array().map(|list| list.iter().map(Rules::of).collect()).unwrap_or_default();
316 let mut findings = Vec::new();
317 let mut dropped = 0usize;
318 let mut occurrences: HashMap<(String, String, String), usize> = HashMap::new();
319 for result in run["results"].as_array().into_iter().flatten() {
320 if kept >= MAX_RESULTS {
321 dropped += 1;
322 continue;
323 }
324 // Suppressed results (in the source, or accepted) are not alerts.
325 if result["suppressions"].as_array().is_some_and(|list| {
326 list.iter().any(|suppression| suppression["status"].as_str().is_none_or(|status| status == "accepted"))
327 }) {
328 continue;
329 }
330 let rules = match result["rule"]["toolComponent"]["index"].as_u64() {
331 Some(index) => extensions.get(index as usize).unwrap_or(&driver_rules),
332 None => &driver_rules,
333 };
334 let rule_id = result["ruleId"].as_str().or(result["rule"]["id"].as_str());
335 let rule = result["ruleIndex"]
336 .as_u64()
337 .or(result["rule"]["index"].as_u64())
338 .and_then(|index| rules.list.get(index as usize).copied())
339 .or_else(|| rule_id.and_then(|id| rules.by_id.get(id).copied()));
340 let Some(rule_id) = rule_id.or_else(|| rule?["id"].as_str()) else { continue };
341 let level = result["level"]
342 .as_str()
343 .and_then(Level::parse)
344 .or_else(|| rule?["defaultConfiguration"]["level"].as_str().and_then(Level::parse))
345 .unwrap_or(Level::Warning);
346 let message = {
347 let arguments: Vec<Value> = result["message"]["arguments"].as_array().cloned().unwrap_or_default();
348 let template = text_of(&result["message"]).or_else(|| {
349 let id = result["message"]["id"].as_str()?;
350 text_of(&rule?["messageStrings"][id])
351 });
352 let text = template.map(|template| fill(&template, &arguments)).unwrap_or_else(|| rule_id.to_owned());
353 text.chars().take(MAX_MESSAGE_CHARS).collect::<String>()
354 };
355 let physical = &result["locations"][0]["physicalLocation"];
356 let location = physical["artifactLocation"]["uri"].as_str().map(|uri| {
357 let region = &physical["region"];
358 let start_line = region["startLine"].as_u64().unwrap_or(1).max(1) as u32;
359 Location {
360 path: repository_path(uri, &roots),
361 start_line,
362 end_line: region["endLine"].as_u64().map_or(start_line, |line| (line as u32).max(start_line)),
363 start_column: region["startColumn"].as_u64().map(|c| c as u32),
364 end_column: region["endColumn"].as_u64().map(|c| c as u32),
365 }
366 });
367 let partial = result["partialFingerprints"]
368 .as_object()
369 .and_then(|partial| {
370 partial
371 .get("primaryLocationLineHash")
372 .or_else(|| partial.iter().min_by_key(|(key, _)| key.as_str()).map(|(_, value)| value))
373 })
374 .or_else(|| result["fingerprints"].as_object().and_then(|all| all.iter().min_by_key(|(key, _)| key.as_str()).map(|(_, v)| v)))
375 .and_then(Value::as_str)
376 // Some tools fill these with a placeholder ("requires
377 // login"); only a hash-like value names a result.
378 .filter(|value| value.len() >= 8 && !value.contains(char::is_whitespace));
379 let path = location.as_ref().map_or("", |location| location.path.as_str());
380 let content = physical["region"]["snippet"]["text"].as_str().unwrap_or(&message).to_owned();
381 let occurrence = {
382 let count = occurrences.entry((rule_id.to_owned(), path.to_owned(), content.clone())).or_insert(0);
383 *count += 1;
384 *count
385 };
386 let description = rule.and_then(|rule| text_of(&rule["shortDescription"]).or_else(|| text_of(&rule["fullDescription"])));
387 findings.push(Finding {
388 fingerprint: fingerprint(&tool, rule_id, path, partial, &content, occurrence),
389 rule_id: rule_id.to_owned(),
390 rule_name: rule.and_then(|rule| rule["name"].as_str().map(str::to_owned)),
391 rule_description: description,
392 help: rule.and_then(|rule| rule["help"]["markdown"].as_str().or(rule["help"]["text"].as_str()).map(str::to_owned)),
393 help_uri: rule.and_then(|rule| rule["helpUri"].as_str().map(str::to_owned)),
394 tags: rule
395 .and_then(|rule| rule["properties"]["tags"].as_array())
396 .map(|tags| tags.iter().filter_map(|tag| tag.as_str().map(str::to_owned)).collect())
397 .unwrap_or_default(),
398 security_severity: security_severity(rule),
399 level,
400 message,
401 location,
402 });
403 kept += 1;
404 }
405 out.push(Run { tool, tool_version, category, findings, dropped });
406 }
407 Ok(out)
408}
409
410/// What an analysis on the default branch does to the alerts of its tool
411/// and category: fingerprints of results not seen before (new alerts),
412/// and of open alerts it no longer reports (fixed).
413pub fn reconcile(open: &BTreeSet<String>, found: &[Finding]) -> (Vec<String>, Vec<String>) {
414 let now: BTreeSet<&str> = found.iter().map(|finding| finding.fingerprint.as_str()).collect();
415 let new = now.iter().filter(|fingerprint| !open.contains(**fingerprint)).map(|f| (*f).to_owned()).collect();
416 let fixed = open.iter().filter(|fingerprint| !now.contains(fingerprint.as_str())).cloned().collect();
417 (new, fixed)
418}
419
420/// The threshold a pull request's code scanning check fails at.
421#[derive(Clone, Copy, Debug, PartialEq, Eq)]
422pub enum Gate {
423 /// Never fails.
424 None,
425 /// Fails on results the tool calls errors.
426 Errors,
427 /// Fails on security results of this severity or worse, and on errors.
428 AtLeast(Severity),
429}
430
431impl Gate {
432 pub fn as_str(self) -> &'static str {
433 match self {
434 Gate::None => "none",
435 Gate::Errors => "errors",
436 Gate::AtLeast(Severity::Critical) => "critical",
437 Gate::AtLeast(Severity::High) => "high",
438 Gate::AtLeast(Severity::Medium) => "medium",
439 Gate::AtLeast(_) => "any",
440 }
441 }
442
443 pub fn parse(text: &str) -> Option<Gate> {
444 Some(match text {
445 "none" => Gate::None,
446 "errors" => Gate::Errors,
447 "critical" => Gate::AtLeast(Severity::Critical),
448 "high" => Gate::AtLeast(Severity::High),
449 "medium" => Gate::AtLeast(Severity::Medium),
450 "any" => Gate::AtLeast(Severity::Low),
451 _ => return None,
452 })
453 }
454
455 /// Whether a new result fails the check.
456 pub fn fails(self, finding: &Finding) -> bool {
457 match self {
458 Gate::None => false,
459 Gate::Errors => finding.level == Level::Error,
460 Gate::AtLeast(threshold) => {
461 finding.level == Level::Error || finding.security_severity.is_some_and(|severity| severity >= threshold)
462 }
463 }
464 }
465}
466
467/// Results on a pull request: those whose fingerprint the default branch
468/// already has open are existing; the others are new, and only new ones
469/// fail the check. Returns (new, existing).
470pub fn split_new<'a>(found: &'a [Finding], on_default: &BTreeSet<String>) -> (Vec<&'a Finding>, Vec<&'a Finding>) {
471 found.iter().partition(|finding| !on_default.contains(&finding.fingerprint))
472}
473
474/// The lines a diff adds, per file, from a unified diff's hunk headers and
475/// lines. Results on these lines are the pull request's own.
476pub fn added_lines(diff: &str) -> BTreeMap<String, BTreeSet<u32>> {
477 let mut out: BTreeMap<String, BTreeSet<u32>> = BTreeMap::new();
478 let mut file: Option<String> = None;
479 let mut line = 0u32;
480 for text in diff.lines() {
481 if let Some(path) = text.strip_prefix("+++ ") {
482 file = path.strip_prefix("b/").or(Some(path)).filter(|path| *path != "/dev/null").map(str::to_owned);
483 continue;
484 }
485 if text.starts_with("--- ") {
486 continue;
487 }
488 if let Some(header) = text.strip_prefix("@@ ") {
489 // `@@ -a,b +c,d @@`
490 line = header
491 .split_whitespace()
492 .find_map(|part| part.strip_prefix('+'))
493 .and_then(|range| range.split(',').next()?.parse().ok())
494 .unwrap_or(1);
495 continue;
496 }
497 let Some(path) = &file else { continue };
498 if text.starts_with('+') {
499 out.entry(path.clone()).or_default().insert(line);
500 line += 1;
501 } else if text.starts_with(' ') {
502 line += 1;
503 }
504 }
505 out
506}
507
508#[cfg(test)]
509mod tests {
510 use super::*;
511
512 const SEMGREP: &str = include_str!("../fixtures/semgrep.sarif");
513 const ESLINT: &str = include_str!("../fixtures/eslint.sarif");
514 const CODEQL: &str = include_str!("../fixtures/codeql.sarif");
515
516 #[test]
517 fn semgrep_results_become_findings() {
518 let runs = parse(SEMGREP, None, None).unwrap();
519 assert_eq!(runs.len(), 1);
520 let run = &runs[0];
521 assert_eq!(run.tool, "Semgrep OSS");
522 assert_eq!(run.category, "Semgrep OSS");
523 assert!(!run.findings.is_empty());
524 let first = &run.findings[0];
525 let location = first.location.as_ref().unwrap();
526 assert!(!location.path.starts_with('/') && !location.path.starts_with("file:"));
527 assert!(location.start_line >= 1);
528 assert!(first.rule_id.contains('.'), "semgrep rule ids are dotted: {}", first.rule_id);
529 assert_eq!(first.fingerprint.len(), 32);
530 }
531
532 #[test]
533 fn eslint_results_use_rule_indexes_and_levels() {
534 let runs = parse(ESLINT, None, Some("file:///home/runner/work/app/app")).unwrap();
535 let run = &runs[0];
536 assert_eq!(run.tool, "ESLint");
537 assert!(run.findings.iter().any(|finding| finding.level == Level::Error));
538 for finding in &run.findings {
539 let path = &finding.location.as_ref().unwrap().path;
540 assert!(!path.contains("home/runner"), "the checkout is cut off: {path}");
541 }
542 }
543
544 #[test]
545 fn codeql_security_severity_and_partial_fingerprints() {
546 let runs = parse(CODEQL, None, None).unwrap();
547 let run = &runs[0];
548 assert_eq!(run.tool, "CodeQL");
549 assert_eq!(run.category, "/language:javascript");
550 let injection = run.findings.iter().find(|finding| finding.rule_id == "js/sql-injection").unwrap();
551 assert_eq!(injection.security_severity, Some(Severity::High));
552 assert_eq!(injection.severity(), Severity::High);
553 assert_eq!(injection.message, "This query string depends on a user-provided value.");
554 // The tool's own fingerprint survives the lines moving.
555 let moved = CODEQL.replace("\"startLine\": 12", "\"startLine\": 40");
556 let again = parse(&moved, None, None).unwrap();
557 assert_eq!(again[0].findings.iter().find(|f| f.rule_id == "js/sql-injection").unwrap().fingerprint, injection.fingerprint);
558 // A rule found through an extension, with a message from messageStrings.
559 let extension = run.findings.iter().find(|finding| finding.rule_id == "js/clear-text-logging").unwrap();
560 assert_eq!(extension.security_severity, Some(Severity::High));
561 assert_eq!(extension.message, "This logs sensitive data returned by password as clear text.");
562 }
563
564 #[test]
565 fn fingerprints_without_a_tool_hash_ignore_the_line_and_count_duplicates() {
566 let a = fingerprint("t", "r", "a.js", None, "eval(x)", 1);
567 assert_eq!(a, fingerprint("t", "r", "a.js", None, "eval(x) ", 1));
568 assert_ne!(a, fingerprint("t", "r", "a.js", None, "eval(x)", 2));
569 assert_ne!(a, fingerprint("t", "r", "b.js", None, "eval(x)", 1));
570 assert_ne!(a, fingerprint("u", "r", "a.js", None, "eval(x)", 1));
571 }
572
573 #[test]
574 fn a_later_analysis_fixes_what_it_no_longer_reports() {
575 let runs = parse(SEMGREP, None, None).unwrap();
576 let findings = &runs[0].findings;
577 let all: BTreeSet<String> = findings.iter().map(|finding| finding.fingerprint.clone()).collect();
578 // First analysis: everything is new.
579 let (new, fixed) = reconcile(&BTreeSet::new(), findings);
580 assert_eq!((new.len(), fixed.len()), (all.len(), 0));
581 // The same again: nothing new, nothing fixed (dedupe across runs).
582 let (new, fixed) = reconcile(&all, findings);
583 assert!(new.is_empty() && fixed.is_empty());
584 // One result gone: its alert is fixed.
585 let (new, fixed) = reconcile(&all, &findings[1..]);
586 assert!(new.is_empty());
587 assert_eq!(fixed, [findings[0].fingerprint.clone()]);
588 }
589
590 #[test]
591 fn uploads_are_gzipped_base64() {
592 let gz = {
593 // A gzip member: header, raw deflate, crc32 and size (not checked).
594 let mut out = vec![0x1f, 0x8b, 8, 0, 0, 0, 0, 0, 0, 0xff];
595 out.extend(miniz_oxide::deflate::compress_to_vec(SEMGREP.as_bytes(), 6));
596 out.extend([0u8; 8]);
597 out
598 };
599 let encoded = encode(&gz);
600 assert_eq!(decode_upload(&encoded).unwrap(), SEMGREP);
601 assert_eq!(decode_upload(&encode(SEMGREP.as_bytes())).unwrap(), SEMGREP);
602 assert!(decode_upload("not base64!").unwrap_err().contains("base64"));
603 let bomb = {
604 let mut out = vec![0x1f, 0x8b, 8, 0, 0, 0, 0, 0, 0, 0xff];
605 out.extend(miniz_oxide::deflate::compress_to_vec(&vec![b' '; MAX_SARIF_BYTES + 1], 6));
606 out.extend([0u8; 8]);
607 out
608 };
609 assert!(decode_upload(&encode(&bomb)).unwrap_err().contains("larger than"));
610 }
611
612 fn encode(bytes: &[u8]) -> String {
613 const ALPHABET: &[u8] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/";
614 let mut out = String::new();
615 for chunk in bytes.chunks(3) {
616 let n = (u32::from(chunk[0]) << 16) | (u32::from(*chunk.get(1).unwrap_or(&0)) << 8) | u32::from(*chunk.get(2).unwrap_or(&0));
617 for i in 0..4 {
618 if i <= chunk.len() {
619 out.push(ALPHABET[(n >> (18 - 6 * i) & 63) as usize] as char);
620 } else {
621 out.push('=');
622 }
623 }
624 }
625 out
626 }
627
628 #[test]
629 fn bad_documents_say_what_is_wrong() {
630 assert!(parse("{", None, None).unwrap_err().contains("not valid JSON"));
631 assert!(parse(r#"{"version":"2.0.0","runs":[]}"#, None, None).unwrap_err().contains("2.1.0"));
632 assert!(parse(r#"{"version":"2.1.0"}"#, None, None).unwrap_err().contains("no runs"));
633 assert!(parse(r#"{"version":"2.1.0","runs":[{"tool":{"driver":{}}}]}"#, None, None).unwrap_err().contains("names no tool"));
634 let suppressed = r#"{"version":"2.1.0","runs":[{"tool":{"driver":{"name":"x"}},"results":[
635 {"ruleId":"a","message":{"text":"m"},"suppressions":[{"kind":"inSource"}]},
636 {"ruleId":"b","message":{"text":"m"}}]}]}"#;
637 let runs = parse(suppressed, Some("lint"), None).unwrap();
638 assert_eq!(runs[0].category, "lint");
639 assert_eq!(runs[0].findings.len(), 1);
640 assert_eq!(runs[0].findings[0].level, Level::Warning);
641 assert!(runs[0].findings[0].location.is_none());
642 }
643
644 #[test]
645 fn the_gate_fails_on_what_the_repository_chose() {
646 let runs = parse(CODEQL, None, None).unwrap();
647 let injection = runs[0].findings.iter().find(|finding| finding.rule_id == "js/sql-injection").unwrap();
648 assert!(Gate::AtLeast(Severity::High).fails(injection));
649 assert!(!Gate::AtLeast(Severity::Critical).fails(injection) || injection.level == Level::Error);
650 assert!(!Gate::None.fails(injection));
651 for gate in ["none", "errors", "critical", "high", "medium", "any"] {
652 assert_eq!(Gate::parse(gate).unwrap().as_str(), gate);
653 }
654 }
655
656 #[test]
657 fn a_diff_says_which_lines_it_adds() {
658 let diff = "diff --git a/src/db.js b/src/db.js\n--- a/src/db.js\n+++ b/src/db.js\n@@ -10,3 +10,4 @@ fn\n const a = 1;\n-const q = 'x';\n+const q = 'SELECT ' + id;\n+run(q);\n const b = 2;\n--- /dev/null\n+++ b/new.js\n@@ -0,0 +1,2 @@\n+one\n+two\n";
659 let added = added_lines(diff);
660 assert_eq!(added["src/db.js"], BTreeSet::from([11, 12]));
661 assert_eq!(added["new.js"], BTreeSet::from([1, 2]));
662 }
663
664 #[test]
665 fn paths_are_from_the_repository_root() {
666 assert_eq!(repository_path("file:///home/runner/work/app/src/a%20b.js", &["file:///home/runner/work/app/".into()]), "src/a b.js");
667 assert_eq!(repository_path("./src/a.js", &[]), "src/a.js");
668 assert_eq!(repository_path("src\\win.js", &[]), "src/win.js");
669 }
670}