Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Teams and CODEOWNERS, labels and milestones, dependency updates, the security suite, and a clearer top bar | 1 | //! Code scanning results in SARIF 2.1.0, the format static analysis tools |
| 2 | //! write: reading an upload (gzipped, base64-encoded JSON), turning each | |
| 3 | //! result into an alert with its rule, severity, message and location, and | |
| 4 | //! fingerprinting it so the same problem found by the next run is the same | |
| 5 | //! alert, and one no longer reported is fixed. | |
| 6 | ||
| 7 | use std::collections::{BTreeMap, BTreeSet, HashMap}; | |
| 8 | ||
| 9 | use serde_json::Value; | |
| 10 | use sha2::{Digest, Sha256}; | |
| 11 | ||
| 12 | use crate::osv::Severity; | |
| 13 | ||
| 14 | /// The largest upload, gzipped and base64-encoded. | |
| 15 | pub const MAX_UPLOAD_BYTES: usize = 10 * 1024 * 1024; | |
| 16 | /// The largest SARIF document once unzipped. | |
| 17 | pub const MAX_SARIF_BYTES: usize = 40 * 1024 * 1024; | |
| 18 | /// Runs one upload may hold. | |
| 19 | pub const MAX_RUNS: usize = 20; | |
| 20 | /// Results kept from one upload; the rest are counted and dropped. | |
| 21 | pub const MAX_RESULTS: usize = 5_000; | |
| 22 | /// The longest message kept. | |
| 23 | const MAX_MESSAGE_CHARS: usize = 2_000; | |
| 24 | ||
| 25 | /// How a tool rates a result. | |
| 26 | #[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] | |
| 27 | pub enum Level { | |
| 28 | None, | |
| 29 | Note, | |
| 30 | Warning, | |
| 31 | Error, | |
| 32 | } | |
| 33 | ||
| 34 | impl Level { | |
| 35 | pub fn as_str(self) -> &'static str { | |
| 36 | match self { | |
| 37 | Level::Error => "error", | |
| 38 | Level::Warning => "warning", | |
| 39 | Level::Note => "note", | |
| 40 | Level::None => "none", | |
| 41 | } | |
| 42 | } | |
| 43 | ||
| 44 | pub fn parse(text: &str) -> Option<Level> { | |
| 45 | Some(match text { | |
| 46 | "error" => Level::Error, | |
| 47 | "warning" => Level::Warning, | |
| 48 | "note" => Level::Note, | |
| 49 | "none" => Level::None, | |
| 50 | _ => return None, | |
| 51 | }) | |
| 52 | } | |
| 53 | } | |
| 54 | ||
| 55 | /// Where a result is. | |
| 56 | #[derive(Clone, Debug, Default, PartialEq, Eq)] | |
| 57 | pub struct Location { | |
| 58 | /// From the repository's root. | |
| 59 | pub path: String, | |
| 60 | pub start_line: u32, | |
| 61 | pub end_line: u32, | |
| 62 | pub start_column: Option<u32>, | |
| 63 | pub end_column: Option<u32>, | |
| 64 | } | |
| 65 | ||
| 66 | /// One result, ready to be an alert. | |
| 67 | #[derive(Clone, Debug, PartialEq)] | |
| 68 | pub struct Finding { | |
| 69 | pub rule_id: String, | |
| 70 | pub rule_name: Option<String>, | |
| 71 | pub rule_description: Option<String>, | |
| 72 | /// Help for the rule, as markdown or text. | |
| 73 | pub help: Option<String>, | |
| 74 | pub help_uri: Option<String>, | |
| 75 | pub tags: Vec<String>, | |
| 76 | pub level: Level, | |
| 77 | /// From the rule's `security-severity` score, for security rules. | |
| 78 | pub security_severity: Option<Severity>, | |
| 79 | pub message: String, | |
| 80 | /// The primary location; absent for results about the whole project. | |
| 81 | pub location: Option<Location>, | |
| 82 | /// Names this result across runs (see [`fingerprint`]). | |
| 83 | pub fingerprint: String, | |
| 84 | } | |
| 85 | ||
| 86 | impl Finding { | |
| 87 | /// The severity people sort by: the security severity when the rule has | |
| 88 | /// one, else the level (error high, warning medium, note low). | |
| 89 | pub fn severity(&self) -> Severity { | |
| 90 | self.security_severity.unwrap_or(match self.level { | |
| 91 | Level::Error => Severity::High, | |
| 92 | Level::Warning => Severity::Medium, | |
| 93 | Level::Note => Severity::Low, | |
| 94 | Level::None => Severity::Unknown, | |
| 95 | }) | |
| 96 | } | |
| 97 | } | |
| 98 | ||
| 99 | /// One run of one tool. | |
| 100 | #[derive(Clone, Debug, PartialEq)] | |
| 101 | pub struct Run { | |
| 102 | pub tool: String, | |
| 103 | pub tool_version: Option<String>, | |
| 104 | /// Which analysis this is, when a repository runs several of one tool | |
| 105 | /// (one per language, say): from the upload, or the run's | |
| 106 | /// `automationDetails.id` up to its last `/`, or the tool's name. | |
| 107 | pub category: String, | |
| 108 | pub findings: Vec<Finding>, | |
| 109 | /// Results past [`MAX_RESULTS`], not kept. | |
| 110 | pub dropped: usize, | |
| 111 | } | |
| 112 | ||
| 113 | /// What reading an upload can find wrong, in words for the person. | |
| 114 | pub type Problem = String; | |
| 115 | ||
| 116 | /// Decodes standard base64, ignoring line breaks. | |
| 117 | pub fn base64_decode(text: &str) -> Result<Vec<u8>, Problem> { | |
| 118 | let mut out = Vec::with_capacity(text.len() * 3 / 4); | |
| 119 | let (mut buffer, mut bits) = (0u32, 0u32); | |
| 120 | for byte in text.bytes() { | |
| 121 | let value = match byte { | |
| 122 | b'A'..=b'Z' => byte - b'A', | |
| 123 | b'a'..=b'z' => byte - b'a' + 26, | |
| 124 | b'0'..=b'9' => byte - b'0' + 52, | |
| 125 | b'+' | b'-' => 62, | |
| 126 | b'/' | b'_' => 63, | |
| 127 | b'=' | b'\n' | b'\r' | b' ' | b'\t' => continue, | |
| 128 | _ => return Err("The sarif field is not base64.".to_owned()), | |
| 129 | }; | |
| 130 | buffer = (buffer << 6) | u32::from(value); | |
| 131 | bits += 6; | |
| 132 | if bits >= 8 { | |
| 133 | bits -= 8; | |
| 134 | out.push((buffer >> bits) as u8); | |
| 135 | } | |
| 136 | } | |
| 137 | Ok(out) | |
| 138 | } | |
| 139 | ||
| 140 | /// Unzips a gzip stream (RFC 1952) of one member, up to `limit` bytes. | |
| 141 | pub fn gunzip(bytes: &[u8], limit: usize) -> Result<Vec<u8>, Problem> { | |
| 142 | const FEXTRA: u8 = 4; | |
| 143 | const FNAME: u8 = 8; | |
| 144 | const FCOMMENT: u8 = 16; | |
| 145 | const FHCRC: u8 = 2; | |
| 146 | if bytes.len() < 18 || bytes[0] != 0x1f || bytes[1] != 0x8b || bytes[2] != 8 { | |
| 147 | return Err("The sarif field is not gzipped.".to_owned()); | |
| 148 | } | |
| 149 | let flags = bytes[3]; | |
| 150 | let mut at = 10; | |
| 151 | let truncated = || "The gzipped SARIF is truncated.".to_owned(); | |
| 152 | if flags & FEXTRA != 0 { | |
| 153 | let length = usize::from(u16::from_le_bytes([*bytes.get(at).ok_or_else(truncated)?, *bytes.get(at + 1).ok_or_else(truncated)?])); | |
| 154 | at += 2 + length; | |
| 155 | } | |
| 156 | for flag in [FNAME, FCOMMENT] { | |
| 157 | if flags & flag != 0 { | |
| 158 | let end = bytes.get(at..).and_then(|rest| rest.iter().position(|b| *b == 0)).ok_or_else(truncated)?; | |
| 159 | at += end + 1; | |
| 160 | } | |
| 161 | } | |
| 162 | if flags & FHCRC != 0 { | |
| 163 | at += 2; | |
| 164 | } | |
| 165 | let body = bytes.get(at..bytes.len().saturating_sub(8)).ok_or_else(truncated)?; | |
| 166 | miniz_oxide::inflate::decompress_to_vec_with_limit(body, limit).map_err(|error| match error.status { | |
| 167 | miniz_oxide::inflate::TINFLStatus::HasMoreOutput => { | |
| 168 | format!("The SARIF is larger than {} MB unzipped.", limit / (1024 * 1024)) | |
| 169 | } | |
| 170 | _ => "The gzipped SARIF could not be unzipped.".to_owned(), | |
| 171 | }) | |
| 172 | } | |
| 173 | ||
| 174 | /// The SARIF document an upload carries: `sarif` is gzipped and base64 | |
| 175 | /// encoded, as uploads conventionally are. Plain base64 JSON is accepted | |
| 176 | /// too. | |
| 177 | pub fn decode_upload(sarif: &str) -> Result<String, Problem> { | |
| 178 | if sarif.len() > MAX_UPLOAD_BYTES { | |
| 179 | return Err(format!("The upload is larger than {} MB.", MAX_UPLOAD_BYTES / (1024 * 1024))); | |
| 180 | } | |
| 181 | let bytes = base64_decode(sarif.trim())?; | |
| 182 | let json = if bytes.starts_with(&[0x1f, 0x8b]) { | |
| 183 | gunzip(&bytes, MAX_SARIF_BYTES)? | |
| 184 | } else if bytes.len() > MAX_SARIF_BYTES { | |
| 185 | return Err(format!("The SARIF is larger than {} MB.", MAX_SARIF_BYTES / (1024 * 1024))); | |
| 186 | } else { | |
| 187 | bytes | |
| 188 | }; | |
| 189 | String::from_utf8(json).map_err(|_| "The SARIF is not UTF-8 text.".to_owned()) | |
| 190 | } | |
| 191 | ||
| 192 | fn text_of(value: &Value) -> Option<String> { | |
| 193 | value.get("text").and_then(Value::as_str).map(str::to_owned) | |
| 194 | } | |
| 195 | ||
| 196 | /// A message with `{0}`-style placeholders filled from its arguments. | |
| 197 | fn fill(template: &str, arguments: &[Value]) -> String { | |
| 198 | let mut out = template.to_owned(); | |
| 199 | for (index, argument) in arguments.iter().enumerate() { | |
| 200 | let value = argument.as_str().map(str::to_owned).unwrap_or_else(|| argument.to_string()); | |
| 201 | out = out.replace(&format!("{{{index}}}"), &value); | |
| 202 | } | |
| 203 | out | |
| 204 | } | |
| 205 | ||
| 206 | /// Percent-decoding for a URI's path. | |
| 207 | fn percent_decode(text: &str) -> String { | |
| 208 | let bytes = text.as_bytes(); | |
| 209 | let mut out = Vec::with_capacity(bytes.len()); | |
| 210 | let mut at = 0; | |
| 211 | while at < bytes.len() { | |
| 212 | if bytes[at] == b'%' | |
| 213 | && let Some(hex) = text.get(at + 1..at + 3) | |
| 214 | && let Ok(byte) = u8::from_str_radix(hex, 16) | |
| 215 | { | |
| 216 | out.push(byte); | |
| 217 | at += 3; | |
| 218 | continue; | |
| 219 | } | |
| 220 | out.push(bytes[at]); | |
| 221 | at += 1; | |
| 222 | } | |
| 223 | String::from_utf8_lossy(&out).into_owned() | |
| 224 | } | |
| 225 | ||
| 226 | /// A result's file, from the repository's root: a `file://` URI under the | |
| 227 | /// checkout (`checkout`, or the run's `%SRCROOT%`) loses that prefix, and | |
| 228 | /// a relative one its leading `./`. | |
| 229 | pub fn repository_path(uri: &str, roots: &[String]) -> String { | |
| 230 | let mut path = percent_decode(uri); | |
| 231 | if let Some(rest) = path.strip_prefix("file://") { | |
| 232 | path = rest.to_owned(); | |
| 233 | } | |
| 234 | for root in roots { | |
| 235 | let root = percent_decode(root.strip_prefix("file://").unwrap_or(root)); | |
| 236 | let root = root.trim_end_matches('/'); | |
| 237 | if !root.is_empty() && let Some(rest) = path.strip_prefix(root) { | |
| 238 | path = rest.to_owned(); | |
| 239 | break; | |
| 240 | } | |
| 241 | } | |
| 242 | path.trim_start_matches("./").trim_start_matches('/').replace('\\', "/") | |
| 243 | } | |
| 244 | ||
| 245 | /// Names a result across runs. A tool's own `primaryLocationLineHash` (or | |
| 246 | /// another partial fingerprint) is used when it gives one, since it | |
| 247 | /// survives lines moving; otherwise the rule, file and the code it points | |
| 248 | /// at (or its message), so editing elsewhere in the file does not make it a | |
| 249 | /// new alert. `occurrence` tells apart identical results in one file. | |
| 250 | pub fn fingerprint(tool: &str, rule: &str, path: &str, partial: Option<&str>, content: &str, occurrence: usize) -> String { | |
| 251 | let basis = match partial { | |
| 252 | Some(partial) => format!("{tool}\n{rule}\n{path}\npartial:{partial}"), | |
| 253 | None => format!("{tool}\n{rule}\n{path}\ncontent:{}\n{occurrence}", content.split_whitespace().collect::<Vec<_>>().join(" ")), | |
| 254 | }; | |
| 255 | let digest = Sha256::digest(basis.as_bytes()); | |
| 256 | digest[..16].iter().map(|byte| format!("{byte:02x}")).collect() | |
| 257 | } | |
| 258 | ||
| 259 | /// The rules of a tool component, by id and by index. | |
| 260 | struct Rules<'a> { | |
| 261 | list: Vec<&'a Value>, | |
| 262 | by_id: HashMap<&'a str, &'a Value>, | |
| 263 | } | |
| 264 | ||
| 265 | impl<'a> Rules<'a> { | |
| 266 | fn of(component: &'a Value) -> Rules<'a> { | |
| 267 | let list: Vec<&Value> = component["rules"].as_array().map(|rules| rules.iter().collect()).unwrap_or_default(); | |
| 268 | let by_id = list.iter().filter_map(|rule| Some((rule["id"].as_str()?, *rule))).collect(); | |
| 269 | Rules { list, by_id } | |
| 270 | } | |
| 271 | } | |
| 272 | ||
| 273 | /// The `security-severity` a rule's properties give, as a severity. | |
| 274 | fn security_severity(rule: Option<&Value>) -> Option<Severity> { | |
| 275 | let raw = &rule?["properties"]["security-severity"]; | |
| 276 | let score = raw.as_f64().or_else(|| raw.as_str()?.trim().parse().ok())?; | |
| 277 | Some(Severity::from_score(score)).filter(|severity| *severity != Severity::Unknown) | |
| 278 | } | |
| 279 | ||
| 280 | /// Reads a SARIF 2.1.0 document into its runs. `category` and `checkout` | |
| 281 | /// come from the upload, when it gives them. | |
| 282 | pub fn parse(text: &str, category: Option<&str>, checkout: Option<&str>) -> Result<Vec<Run>, Problem> { | |
| 283 | let document: Value = serde_json::from_str(text).map_err(|error| format!("The SARIF is not valid JSON: {error}"))?; | |
| 284 | let version = document["version"].as_str().unwrap_or_default(); | |
| 285 | if version != "2.1.0" { | |
| 286 | return Err(format!("Only SARIF 2.1.0 is read; this document says {}.", if version.is_empty() { "no version" } else { version })); | |
| 287 | } | |
| 288 | let runs = document["runs"].as_array().ok_or("The SARIF has no runs.")?; | |
| 289 | if runs.len() > MAX_RUNS { | |
| 290 | return Err(format!("The SARIF has {} runs; at most {MAX_RUNS} are read per upload.", runs.len())); | |
| 291 | } | |
| 292 | let mut out = Vec::new(); | |
| 293 | let mut kept = 0usize; | |
| 294 | for run in runs { | |
| 295 | let driver = &run["tool"]["driver"]; | |
| 296 | let tool = driver["name"].as_str().filter(|name| !name.trim().is_empty()).ok_or("A run names no tool.")?.trim().to_owned(); | |
| 297 | let tool_version = driver["semanticVersion"].as_str().or(driver["version"].as_str()).map(str::to_owned); | |
| 298 | let category = category | |
| 299 | .map(str::to_owned) | |
| 300 | .filter(|category| !category.trim().is_empty()) | |
| 301 | .or_else(|| { | |
| 302 | let id = run["automationDetails"]["id"].as_str()?; | |
| 303 | // `category/run-id`: the category is what precedes the last slash. | |
| 304 | id.rsplit_once('/').map(|(category, _)| category.to_owned()).filter(|category| !category.is_empty()) | |
| 305 | }) | |
| 306 | .unwrap_or_else(|| tool.clone()); | |
| 307 | let mut roots: Vec<String> = Vec::new(); | |
| 308 | if let Some(checkout) = checkout { | |
| 309 | roots.push(checkout.to_owned()); | |
| 310 | } | |
| 311 | if let Some(bases) = run["originalUriBaseIds"].as_object() { | |
| 312 | roots.extend(bases.values().filter_map(|base| base["uri"].as_str().map(str::to_owned))); | |
| 313 | } | |
| 314 | let driver_rules = Rules::of(driver); | |
| 315 | let extensions: Vec<Rules> = run["tool"]["extensions"].as_array().map(|list| list.iter().map(Rules::of).collect()).unwrap_or_default(); | |
| 316 | let mut findings = Vec::new(); | |
| 317 | let mut dropped = 0usize; | |
| 318 | let mut occurrences: HashMap<(String, String, String), usize> = HashMap::new(); | |
| 319 | for result in run["results"].as_array().into_iter().flatten() { | |
| 320 | if kept >= MAX_RESULTS { | |
| 321 | dropped += 1; | |
| 322 | continue; | |
| 323 | } | |
| 324 | // Suppressed results (in the source, or accepted) are not alerts. | |
| 325 | if result["suppressions"].as_array().is_some_and(|list| { | |
| 326 | list.iter().any(|suppression| suppression["status"].as_str().is_none_or(|status| status == "accepted")) | |
| 327 | }) { | |
| 328 | continue; | |
| 329 | } | |
| 330 | let rules = match result["rule"]["toolComponent"]["index"].as_u64() { | |
| 331 | Some(index) => extensions.get(index as usize).unwrap_or(&driver_rules), | |
| 332 | None => &driver_rules, | |
| 333 | }; | |
| 334 | let rule_id = result["ruleId"].as_str().or(result["rule"]["id"].as_str()); | |
| 335 | let rule = result["ruleIndex"] | |
| 336 | .as_u64() | |
| 337 | .or(result["rule"]["index"].as_u64()) | |
| 338 | .and_then(|index| rules.list.get(index as usize).copied()) | |
| 339 | .or_else(|| rule_id.and_then(|id| rules.by_id.get(id).copied())); | |
| 340 | let Some(rule_id) = rule_id.or_else(|| rule?["id"].as_str()) else { continue }; | |
| 341 | let level = result["level"] | |
| 342 | .as_str() | |
| 343 | .and_then(Level::parse) | |
| 344 | .or_else(|| rule?["defaultConfiguration"]["level"].as_str().and_then(Level::parse)) | |
| 345 | .unwrap_or(Level::Warning); | |
| 346 | let message = { | |
| 347 | let arguments: Vec<Value> = result["message"]["arguments"].as_array().cloned().unwrap_or_default(); | |
| 348 | let template = text_of(&result["message"]).or_else(|| { | |
| 349 | let id = result["message"]["id"].as_str()?; | |
| 350 | text_of(&rule?["messageStrings"][id]) | |
| 351 | }); | |
| 352 | let text = template.map(|template| fill(&template, &arguments)).unwrap_or_else(|| rule_id.to_owned()); | |
| 353 | text.chars().take(MAX_MESSAGE_CHARS).collect::<String>() | |
| 354 | }; | |
| 355 | let physical = &result["locations"][0]["physicalLocation"]; | |
| 356 | let location = physical["artifactLocation"]["uri"].as_str().map(|uri| { | |
| 357 | let region = &physical["region"]; | |
| 358 | let start_line = region["startLine"].as_u64().unwrap_or(1).max(1) as u32; | |
| 359 | Location { | |
| 360 | path: repository_path(uri, &roots), | |
| 361 | start_line, | |
| 362 | end_line: region["endLine"].as_u64().map_or(start_line, |line| (line as u32).max(start_line)), | |
| 363 | start_column: region["startColumn"].as_u64().map(|c| c as u32), | |
| 364 | end_column: region["endColumn"].as_u64().map(|c| c as u32), | |
| 365 | } | |
| 366 | }); | |
| 367 | let partial = result["partialFingerprints"] | |
| 368 | .as_object() | |
| 369 | .and_then(|partial| { | |
| 370 | partial | |
| 371 | .get("primaryLocationLineHash") | |
| 372 | .or_else(|| partial.iter().min_by_key(|(key, _)| key.as_str()).map(|(_, value)| value)) | |
| 373 | }) | |
| 374 | .or_else(|| result["fingerprints"].as_object().and_then(|all| all.iter().min_by_key(|(key, _)| key.as_str()).map(|(_, v)| v))) | |
| 375 | .and_then(Value::as_str) | |
| 376 | // Some tools fill these with a placeholder ("requires | |
| 377 | // login"); only a hash-like value names a result. | |
| 378 | .filter(|value| value.len() >= 8 && !value.contains(char::is_whitespace)); | |
| 379 | let path = location.as_ref().map_or("", |location| location.path.as_str()); | |
| 380 | let content = physical["region"]["snippet"]["text"].as_str().unwrap_or(&message).to_owned(); | |
| 381 | let occurrence = { | |
| 382 | let count = occurrences.entry((rule_id.to_owned(), path.to_owned(), content.clone())).or_insert(0); | |
| 383 | *count += 1; | |
| 384 | *count | |
| 385 | }; | |
| 386 | let description = rule.and_then(|rule| text_of(&rule["shortDescription"]).or_else(|| text_of(&rule["fullDescription"]))); | |
| 387 | findings.push(Finding { | |
| 388 | fingerprint: fingerprint(&tool, rule_id, path, partial, &content, occurrence), | |
| 389 | rule_id: rule_id.to_owned(), | |
| 390 | rule_name: rule.and_then(|rule| rule["name"].as_str().map(str::to_owned)), | |
| 391 | rule_description: description, | |
| 392 | help: rule.and_then(|rule| rule["help"]["markdown"].as_str().or(rule["help"]["text"].as_str()).map(str::to_owned)), | |
| 393 | help_uri: rule.and_then(|rule| rule["helpUri"].as_str().map(str::to_owned)), | |
| 394 | tags: rule | |
| 395 | .and_then(|rule| rule["properties"]["tags"].as_array()) | |
| 396 | .map(|tags| tags.iter().filter_map(|tag| tag.as_str().map(str::to_owned)).collect()) | |
| 397 | .unwrap_or_default(), | |
| 398 | security_severity: security_severity(rule), | |
| 399 | level, | |
| 400 | message, | |
| 401 | location, | |
| 402 | }); | |
| 403 | kept += 1; | |
| 404 | } | |
| 405 | out.push(Run { tool, tool_version, category, findings, dropped }); | |
| 406 | } | |
| 407 | Ok(out) | |
| 408 | } | |
| 409 | ||
| 410 | /// What an analysis on the default branch does to the alerts of its tool | |
| 411 | /// and category: fingerprints of results not seen before (new alerts), | |
| 412 | /// and of open alerts it no longer reports (fixed). | |
| 413 | pub fn reconcile(open: &BTreeSet<String>, found: &[Finding]) -> (Vec<String>, Vec<String>) { | |
| 414 | let now: BTreeSet<&str> = found.iter().map(|finding| finding.fingerprint.as_str()).collect(); | |
| 415 | let new = now.iter().filter(|fingerprint| !open.contains(**fingerprint)).map(|f| (*f).to_owned()).collect(); | |
| 416 | let fixed = open.iter().filter(|fingerprint| !now.contains(fingerprint.as_str())).cloned().collect(); | |
| 417 | (new, fixed) | |
| 418 | } | |
| 419 | ||
| 420 | /// The threshold a pull request's code scanning check fails at. | |
| 421 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 422 | pub enum Gate { | |
| 423 | /// Never fails. | |
| 424 | None, | |
| 425 | /// Fails on results the tool calls errors. | |
| 426 | Errors, | |
| 427 | /// Fails on security results of this severity or worse, and on errors. | |
| 428 | AtLeast(Severity), | |
| 429 | } | |
| 430 | ||
| 431 | impl Gate { | |
| 432 | pub fn as_str(self) -> &'static str { | |
| 433 | match self { | |
| 434 | Gate::None => "none", | |
| 435 | Gate::Errors => "errors", | |
| 436 | Gate::AtLeast(Severity::Critical) => "critical", | |
| 437 | Gate::AtLeast(Severity::High) => "high", | |
| 438 | Gate::AtLeast(Severity::Medium) => "medium", | |
| 439 | Gate::AtLeast(_) => "any", | |
| 440 | } | |
| 441 | } | |
| 442 | ||
| 443 | pub fn parse(text: &str) -> Option<Gate> { | |
| 444 | Some(match text { | |
| 445 | "none" => Gate::None, | |
| 446 | "errors" => Gate::Errors, | |
| 447 | "critical" => Gate::AtLeast(Severity::Critical), | |
| 448 | "high" => Gate::AtLeast(Severity::High), | |
| 449 | "medium" => Gate::AtLeast(Severity::Medium), | |
| 450 | "any" => Gate::AtLeast(Severity::Low), | |
| 451 | _ => return None, | |
| 452 | }) | |
| 453 | } | |
| 454 | ||
| 455 | /// Whether a new result fails the check. | |
| 456 | pub fn fails(self, finding: &Finding) -> bool { | |
| 457 | match self { | |
| 458 | Gate::None => false, | |
| 459 | Gate::Errors => finding.level == Level::Error, | |
| 460 | Gate::AtLeast(threshold) => { | |
| 461 | finding.level == Level::Error || finding.security_severity.is_some_and(|severity| severity >= threshold) | |
| 462 | } | |
| 463 | } | |
| 464 | } | |
| 465 | } | |
| 466 | ||
| 467 | /// Results on a pull request: those whose fingerprint the default branch | |
| 468 | /// already has open are existing; the others are new, and only new ones | |
| 469 | /// fail the check. Returns (new, existing). | |
| 470 | pub fn split_new<'a>(found: &'a [Finding], on_default: &BTreeSet<String>) -> (Vec<&'a Finding>, Vec<&'a Finding>) { | |
| 471 | found.iter().partition(|finding| !on_default.contains(&finding.fingerprint)) | |
| 472 | } | |
| 473 | ||
| 474 | /// The lines a diff adds, per file, from a unified diff's hunk headers and | |
| 475 | /// lines. Results on these lines are the pull request's own. | |
| 476 | pub fn added_lines(diff: &str) -> BTreeMap<String, BTreeSet<u32>> { | |
| 477 | let mut out: BTreeMap<String, BTreeSet<u32>> = BTreeMap::new(); | |
| 478 | let mut file: Option<String> = None; | |
| 479 | let mut line = 0u32; | |
| 480 | for text in diff.lines() { | |
| 481 | if let Some(path) = text.strip_prefix("+++ ") { | |
| 482 | file = path.strip_prefix("b/").or(Some(path)).filter(|path| *path != "/dev/null").map(str::to_owned); | |
| 483 | continue; | |
| 484 | } | |
| 485 | if text.starts_with("--- ") { | |
| 486 | continue; | |
| 487 | } | |
| 488 | if let Some(header) = text.strip_prefix("@@ ") { | |
| 489 | // `@@ -a,b +c,d @@` | |
| 490 | line = header | |
| 491 | .split_whitespace() | |
| 492 | .find_map(|part| part.strip_prefix('+')) | |
| 493 | .and_then(|range| range.split(',').next()?.parse().ok()) | |
| 494 | .unwrap_or(1); | |
| 495 | continue; | |
| 496 | } | |
| 497 | let Some(path) = &file else { continue }; | |
| 498 | if text.starts_with('+') { | |
| 499 | out.entry(path.clone()).or_default().insert(line); | |
| 500 | line += 1; | |
| 501 | } else if text.starts_with(' ') { | |
| 502 | line += 1; | |
| 503 | } | |
| 504 | } | |
| 505 | out | |
| 506 | } | |
| 507 | ||
| 508 | #[cfg(test)] | |
| 509 | mod tests { | |
| 510 | use super::*; | |
| 511 | ||
| 512 | const SEMGREP: &str = include_str!("../fixtures/semgrep.sarif"); | |
| 513 | const ESLINT: &str = include_str!("../fixtures/eslint.sarif"); | |
| 514 | const CODEQL: &str = include_str!("../fixtures/codeql.sarif"); | |
| 515 | ||
| 516 | #[test] | |
| 517 | fn semgrep_results_become_findings() { | |
| 518 | let runs = parse(SEMGREP, None, None).unwrap(); | |
| 519 | assert_eq!(runs.len(), 1); | |
| 520 | let run = &runs[0]; | |
| 521 | assert_eq!(run.tool, "Semgrep OSS"); | |
| 522 | assert_eq!(run.category, "Semgrep OSS"); | |
| 523 | assert!(!run.findings.is_empty()); | |
| 524 | let first = &run.findings[0]; | |
| 525 | let location = first.location.as_ref().unwrap(); | |
| 526 | assert!(!location.path.starts_with('/') && !location.path.starts_with("file:")); | |
| 527 | assert!(location.start_line >= 1); | |
| 528 | assert!(first.rule_id.contains('.'), "semgrep rule ids are dotted: {}", first.rule_id); | |
| 529 | assert_eq!(first.fingerprint.len(), 32); | |
| 530 | } | |
| 531 | ||
| 532 | #[test] | |
| 533 | fn eslint_results_use_rule_indexes_and_levels() { | |
| 534 | let runs = parse(ESLINT, None, Some("file:///home/runner/work/app/app")).unwrap(); | |
| 535 | let run = &runs[0]; | |
| 536 | assert_eq!(run.tool, "ESLint"); | |
| 537 | assert!(run.findings.iter().any(|finding| finding.level == Level::Error)); | |
| 538 | for finding in &run.findings { | |
| 539 | let path = &finding.location.as_ref().unwrap().path; | |
| 540 | assert!(!path.contains("home/runner"), "the checkout is cut off: {path}"); | |
| 541 | } | |
| 542 | } | |
| 543 | ||
| 544 | #[test] | |
| 545 | fn codeql_security_severity_and_partial_fingerprints() { | |
| 546 | let runs = parse(CODEQL, None, None).unwrap(); | |
| 547 | let run = &runs[0]; | |
| 548 | assert_eq!(run.tool, "CodeQL"); | |
| 549 | assert_eq!(run.category, "/language:javascript"); | |
| 550 | let injection = run.findings.iter().find(|finding| finding.rule_id == "js/sql-injection").unwrap(); | |
| 551 | assert_eq!(injection.security_severity, Some(Severity::High)); | |
| 552 | assert_eq!(injection.severity(), Severity::High); | |
| 553 | assert_eq!(injection.message, "This query string depends on a user-provided value."); | |
| 554 | // The tool's own fingerprint survives the lines moving. | |
| 555 | let moved = CODEQL.replace("\"startLine\": 12", "\"startLine\": 40"); | |
| 556 | let again = parse(&moved, None, None).unwrap(); | |
| 557 | assert_eq!(again[0].findings.iter().find(|f| f.rule_id == "js/sql-injection").unwrap().fingerprint, injection.fingerprint); | |
| 558 | // A rule found through an extension, with a message from messageStrings. | |
| 559 | let extension = run.findings.iter().find(|finding| finding.rule_id == "js/clear-text-logging").unwrap(); | |
| 560 | assert_eq!(extension.security_severity, Some(Severity::High)); | |
| 561 | assert_eq!(extension.message, "This logs sensitive data returned by password as clear text."); | |
| 562 | } | |
| 563 | ||
| 564 | #[test] | |
| 565 | fn fingerprints_without_a_tool_hash_ignore_the_line_and_count_duplicates() { | |
| 566 | let a = fingerprint("t", "r", "a.js", None, "eval(x)", 1); | |
| 567 | assert_eq!(a, fingerprint("t", "r", "a.js", None, "eval(x) ", 1)); | |
| 568 | assert_ne!(a, fingerprint("t", "r", "a.js", None, "eval(x)", 2)); | |
| 569 | assert_ne!(a, fingerprint("t", "r", "b.js", None, "eval(x)", 1)); | |
| 570 | assert_ne!(a, fingerprint("u", "r", "a.js", None, "eval(x)", 1)); | |
| 571 | } | |
| 572 | ||
| 573 | #[test] | |
| 574 | fn a_later_analysis_fixes_what_it_no_longer_reports() { | |
| 575 | let runs = parse(SEMGREP, None, None).unwrap(); | |
| 576 | let findings = &runs[0].findings; | |
| 577 | let all: BTreeSet<String> = findings.iter().map(|finding| finding.fingerprint.clone()).collect(); | |
| 578 | // First analysis: everything is new. | |
| 579 | let (new, fixed) = reconcile(&BTreeSet::new(), findings); | |
| 580 | assert_eq!((new.len(), fixed.len()), (all.len(), 0)); | |
| 581 | // The same again: nothing new, nothing fixed (dedupe across runs). | |
| 582 | let (new, fixed) = reconcile(&all, findings); | |
| 583 | assert!(new.is_empty() && fixed.is_empty()); | |
| 584 | // One result gone: its alert is fixed. | |
| 585 | let (new, fixed) = reconcile(&all, &findings[1..]); | |
| 586 | assert!(new.is_empty()); | |
| 587 | assert_eq!(fixed, [findings[0].fingerprint.clone()]); | |
| 588 | } | |
| 589 | ||
| 590 | #[test] | |
| 591 | fn uploads_are_gzipped_base64() { | |
| 592 | let gz = { | |
| 593 | // A gzip member: header, raw deflate, crc32 and size (not checked). | |
| 594 | let mut out = vec![0x1f, 0x8b, 8, 0, 0, 0, 0, 0, 0, 0xff]; | |
| 595 | out.extend(miniz_oxide::deflate::compress_to_vec(SEMGREP.as_bytes(), 6)); | |
| 596 | out.extend([0u8; 8]); | |
| 597 | out | |
| 598 | }; | |
| 599 | let encoded = encode(&gz); | |
| 600 | assert_eq!(decode_upload(&encoded).unwrap(), SEMGREP); | |
| 601 | assert_eq!(decode_upload(&encode(SEMGREP.as_bytes())).unwrap(), SEMGREP); | |
| 602 | assert!(decode_upload("not base64!").unwrap_err().contains("base64")); | |
| 603 | let bomb = { | |
| 604 | let mut out = vec![0x1f, 0x8b, 8, 0, 0, 0, 0, 0, 0, 0xff]; | |
| 605 | out.extend(miniz_oxide::deflate::compress_to_vec(&vec![b' '; MAX_SARIF_BYTES + 1], 6)); | |
| 606 | out.extend([0u8; 8]); | |
| 607 | out | |
| 608 | }; | |
| 609 | assert!(decode_upload(&encode(&bomb)).unwrap_err().contains("larger than")); | |
| 610 | } | |
| 611 | ||
| 612 | fn encode(bytes: &[u8]) -> String { | |
| 613 | const ALPHABET: &[u8] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; | |
| 614 | let mut out = String::new(); | |
| 615 | for chunk in bytes.chunks(3) { | |
| 616 | let n = (u32::from(chunk[0]) << 16) | (u32::from(*chunk.get(1).unwrap_or(&0)) << 8) | u32::from(*chunk.get(2).unwrap_or(&0)); | |
| 617 | for i in 0..4 { | |
| 618 | if i <= chunk.len() { | |
| 619 | out.push(ALPHABET[(n >> (18 - 6 * i) & 63) as usize] as char); | |
| 620 | } else { | |
| 621 | out.push('='); | |
| 622 | } | |
| 623 | } | |
| 624 | } | |
| 625 | out | |
| 626 | } | |
| 627 | ||
| 628 | #[test] | |
| 629 | fn bad_documents_say_what_is_wrong() { | |
| 630 | assert!(parse("{", None, None).unwrap_err().contains("not valid JSON")); | |
| 631 | assert!(parse(r#"{"version":"2.0.0","runs":[]}"#, None, None).unwrap_err().contains("2.1.0")); | |
| 632 | assert!(parse(r#"{"version":"2.1.0"}"#, None, None).unwrap_err().contains("no runs")); | |
| 633 | assert!(parse(r#"{"version":"2.1.0","runs":[{"tool":{"driver":{}}}]}"#, None, None).unwrap_err().contains("names no tool")); | |
| 634 | let suppressed = r#"{"version":"2.1.0","runs":[{"tool":{"driver":{"name":"x"}},"results":[ | |
| 635 | {"ruleId":"a","message":{"text":"m"},"suppressions":[{"kind":"inSource"}]}, | |
| 636 | {"ruleId":"b","message":{"text":"m"}}]}]}"#; | |
| 637 | let runs = parse(suppressed, Some("lint"), None).unwrap(); | |
| 638 | assert_eq!(runs[0].category, "lint"); | |
| 639 | assert_eq!(runs[0].findings.len(), 1); | |
| 640 | assert_eq!(runs[0].findings[0].level, Level::Warning); | |
| 641 | assert!(runs[0].findings[0].location.is_none()); | |
| 642 | } | |
| 643 | ||
| 644 | #[test] | |
| 645 | fn the_gate_fails_on_what_the_repository_chose() { | |
| 646 | let runs = parse(CODEQL, None, None).unwrap(); | |
| 647 | let injection = runs[0].findings.iter().find(|finding| finding.rule_id == "js/sql-injection").unwrap(); | |
| 648 | assert!(Gate::AtLeast(Severity::High).fails(injection)); | |
| 649 | assert!(!Gate::AtLeast(Severity::Critical).fails(injection) || injection.level == Level::Error); | |
| 650 | assert!(!Gate::None.fails(injection)); | |
| 651 | for gate in ["none", "errors", "critical", "high", "medium", "any"] { | |
| 652 | assert_eq!(Gate::parse(gate).unwrap().as_str(), gate); | |
| 653 | } | |
| 654 | } | |
| 655 | ||
| 656 | #[test] | |
| 657 | fn a_diff_says_which_lines_it_adds() { | |
| 658 | let diff = "diff --git a/src/db.js b/src/db.js\n--- a/src/db.js\n+++ b/src/db.js\n@@ -10,3 +10,4 @@ fn\n const a = 1;\n-const q = 'x';\n+const q = 'SELECT ' + id;\n+run(q);\n const b = 2;\n--- /dev/null\n+++ b/new.js\n@@ -0,0 +1,2 @@\n+one\n+two\n"; | |
| 659 | let added = added_lines(diff); | |
| 660 | assert_eq!(added["src/db.js"], BTreeSet::from([11, 12])); | |
| 661 | assert_eq!(added["new.js"], BTreeSet::from([1, 2])); | |
| 662 | } | |
| 663 | ||
| 664 | #[test] | |
| 665 | fn paths_are_from_the_repository_root() { | |
| 666 | assert_eq!(repository_path("file:///home/runner/work/app/src/a%20b.js", &["file:///home/runner/work/app/".into()]), "src/a b.js"); | |
| 667 | assert_eq!(repository_path("./src/a.js", &[]), "src/a.js"); | |
| 668 | assert_eq!(repository_path("src\\win.js", &[]), "src/win.js"); | |
| 669 | } | |
| 670 | } |