Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Actions: keep workflow runs safe | 1 | //! Masking secrets in a job's log: every form a secret's value can take in |
| 2 | //! output, so that `***` replaces it wherever it shows. | |
| 3 | //! | |
| 4 | //! A value is masked as it is, each of its lines on its own (a PEM key is | |
| 5 | //! printed a line at a time), base64-encoded (as `echo $KEY | base64` or an | |
| 6 | //! HTTP basic header shows it, at each of the three byte offsets it can sit | |
| 7 | //! at inside longer base64), and JSON-escaped (as a value inside printed | |
| 8 | //! JSON). Values of one character are not masked: every log line would lose | |
| 9 | //! that character. | |
| 10 | ||
| 11 | /// The shortest value or part of one that is masked. | |
| 12 | pub const MIN_MASK_CHARS: usize = 2; | |
| 13 | ||
| 14 | /// Every form of `value` to mask, longest first. | |
| 15 | pub fn variants(value: &str) -> Vec<String> { | |
| 16 | let mut out: Vec<String> = Vec::new(); | |
| 17 | let mut add = |text: &str| { | |
| 18 | if text.trim().chars().count() >= MIN_MASK_CHARS && !out.iter().any(|known| known == text) { | |
| 19 | out.push(text.to_owned()); | |
| 20 | } | |
| 21 | }; | |
| 22 | if value.trim().chars().count() < MIN_MASK_CHARS { | |
| 23 | return Vec::new(); | |
| 24 | } | |
| 25 | add(value); | |
| 26 | let trimmed = value.trim_matches(|c| c == '\n' || c == '\r'); | |
| 27 | add(trimmed); | |
| 28 | // Each line, as it is printed on its own. | |
| 29 | if trimmed.contains('\n') { | |
| 30 | for line in trimmed.lines() { | |
| 31 | let line = line.trim_end_matches('\r'); | |
| 32 | add(line); | |
| 33 | add(line.trim()); | |
| 34 | } | |
| 35 | } | |
| 36 | // JSON-escaped, without its quotes. | |
| 37 | let escaped = serde_json::to_string(value).unwrap_or_default(); | |
| 38 | add(&escaped[1..escaped.len().saturating_sub(1).max(1)]); | |
| 39 | // Base64, at each offset inside a longer encoding, and on its own. | |
| 40 | for encoded in base64_variants(value.as_bytes()) { | |
| 41 | add(&encoded); | |
| 42 | } | |
| 43 | if trimmed != value { | |
| 44 | for encoded in base64_variants(trimmed.as_bytes()) { | |
| 45 | add(&encoded); | |
| 46 | } | |
| 47 | } | |
| 48 | out.sort_by_key(|mask| std::cmp::Reverse(mask.len())); | |
| 49 | out | |
| 50 | } | |
| 51 | ||
| 52 | /// Every form of each of `values`, longest first. | |
| 53 | pub fn all_variants<'a>(values: impl IntoIterator<Item = &'a str>) -> Vec<String> { | |
| 54 | let mut out: Vec<String> = Vec::new(); | |
| 55 | for value in values { | |
| 56 | for variant in variants(value) { | |
| 57 | if !out.contains(&variant) { | |
| 58 | out.push(variant); | |
| 59 | } | |
| 60 | } | |
| 61 | } | |
| 62 | out.sort_by_key(|mask| std::cmp::Reverse(mask.len())); | |
| 63 | out | |
| 64 | } | |
| 65 | ||
| 66 | /// `text` with every mask replaced by `***`. `masks` longest first. | |
| 67 | pub fn apply(text: &str, masks: &[String]) -> String { | |
| 68 | let mut out = text.to_owned(); | |
| 69 | for mask in masks.iter().filter(|mask| !mask.is_empty()) { | |
| 70 | if out.contains(mask.as_str()) { | |
| 71 | out = out.replace(mask.as_str(), "***"); | |
| 72 | } | |
| 73 | } | |
| 74 | out | |
| 75 | } | |
| 76 | ||
| 77 | /// Whether `text` holds any of `masks`. | |
| 78 | pub fn reveals(text: &str, masks: &[String]) -> bool { | |
| 79 | masks.iter().any(|mask| !mask.is_empty() && text.contains(mask.as_str())) | |
| 80 | } | |
| 81 | ||
| 82 | const ALPHABET: &[u8; 64] = b"ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"; | |
| 83 | ||
| 84 | fn base64(bytes: &[u8]) -> String { | |
| 85 | let mut out = String::with_capacity(bytes.len().div_ceil(3) * 4); | |
| 86 | for chunk in bytes.chunks(3) { | |
| 87 | let b = [chunk[0], *chunk.get(1).unwrap_or(&0), *chunk.get(2).unwrap_or(&0)]; | |
| 88 | let n = (u32::from(b[0]) << 16) | (u32::from(b[1]) << 8) | u32::from(b[2]); | |
| 89 | for (i, shift) in [18, 12, 6, 0].into_iter().enumerate() { | |
| 90 | if i <= chunk.len() { | |
| 91 | out.push(ALPHABET[((n >> shift) & 63) as usize] as char); | |
| 92 | } else { | |
| 93 | out.push('='); | |
| 94 | } | |
| 95 | } | |
| 96 | } | |
| 97 | out | |
| 98 | } | |
| 99 | ||
| 100 | /// The base64 of `bytes` as it reads inside a longer encoding, starting at | |
| 101 | /// each of the three byte offsets: only the characters that depend on | |
| 102 | /// `bytes` alone, so the mask matches whatever surrounds it. The whole | |
| 103 | /// encoding, padding and all, comes first. | |
| 104 | fn base64_variants(bytes: &[u8]) -> Vec<String> { | |
| 105 | let mut out = vec![base64(bytes)]; | |
| 106 | // Too short to mask without hiding ordinary text. | |
| 107 | if bytes.len() < 3 { | |
| 108 | return out; | |
| 109 | } | |
| 110 | for offset in 0..3usize { | |
| 111 | let mut padded = vec![0u8; offset]; | |
| 112 | padded.extend_from_slice(bytes); | |
| 113 | let encoded = base64(&padded); | |
| 114 | // Characters before the first whole group of `bytes` mix in what | |
| 115 | // comes before them; after the last whole group, what comes after. | |
| 116 | let start = if offset == 0 { 0 } else { 4 }; | |
| 117 | let whole = (offset + bytes.len()) / 3 * 4; | |
| 118 | if whole > start + 1 { | |
| 119 | out.push(encoded[start..whole].to_owned()); | |
| 120 | } | |
| 121 | } | |
| 122 | out | |
| 123 | } | |
| 124 | ||
| 125 | #[cfg(test)] | |
| 126 | mod tests { | |
| 127 | use super::*; | |
| 128 | ||
| 129 | #[test] | |
| 130 | fn base64_reads_as_the_standard_alphabet() { | |
| 131 | assert_eq!(base64(b"hunter2"), "aHVudGVyMg=="); | |
| 132 | assert_eq!(base64(b"ab"), "YWI="); | |
| 133 | assert_eq!(base64(b"abc"), "YWJj"); | |
| 134 | } | |
| 135 | ||
| 136 | #[test] | |
| 137 | fn a_multi_line_secret_is_masked_a_line_at_a_time() { | |
| 138 | let pem = "-----BEGIN PRIVATE KEY-----\nMIIEvQIBADANBgkqhkiG9w0BAQEFAAS\nCBKcwggSjAgEAAoIBAQC7\n-----END PRIVATE KEY-----\n"; | |
| 139 | let masks = variants(pem); | |
| 140 | // Printed one line at a time, as `cat key.pem` does. | |
| 141 | for line in pem.lines() { | |
| 142 | assert_eq!(apply(line, &masks), "***", "{line}"); | |
| 143 | } | |
| 144 | assert_eq!(apply(&format!("key: {pem}"), &masks).matches("***").count(), 1, "the whole value first"); | |
| 145 | } | |
| 146 | ||
| 147 | #[test] | |
| 148 | fn base64_and_json_forms_are_masked() { | |
| 149 | let secret = "s3cr3t-v4lue!"; | |
| 150 | let masks = variants(secret); | |
| 151 | // `echo -n $SECRET | base64`. | |
| 152 | assert_eq!(apply(&base64(secret.as_bytes()), &masks), "***"); | |
| 153 | // Inside a longer encoding, at any offset: an HTTP basic header. | |
| 154 | for prefix in ["", "u:", "us:", "use:"] { | |
| 155 | let header = base64(format!("{prefix}{secret}").as_bytes()); | |
| 156 | assert!(apply(&header, &masks).contains("***"), "{prefix}"); | |
| 157 | assert!(!apply(&header, &masks).contains(&base64(secret.as_bytes())[4..12]), "{prefix}"); | |
| 158 | } | |
| 159 | // JSON-escaped, as a value printed inside JSON. | |
| 160 | let quoted = "pa\"ss\\word\nline"; | |
| 161 | let printed = serde_json::json!({ "password": quoted }).to_string(); | |
| 162 | assert_eq!(apply(&printed, &variants(quoted)), "{\"password\":\"***\"}"); | |
| 163 | } | |
| 164 | ||
| 165 | #[test] | |
| 166 | fn short_values_and_blanks() { | |
| 167 | assert!(variants("").is_empty()); | |
| 168 | assert!(variants(" \n").is_empty()); | |
| 169 | assert!(variants("x").is_empty(), "a single character is not masked"); | |
| 170 | assert_eq!(apply("pin is 42", &variants("42")), "pin is ***"); | |
| 171 | assert_eq!(apply("abc", &variants("abc")), "***"); | |
| 172 | } | |
| 173 | ||
| 174 | #[test] | |
| 175 | fn outputs_that_reveal_a_secret_are_found() { | |
| 176 | let masks = all_variants(["topsecret", "other\nvalue"]); | |
| 177 | assert!(reveals("x=topsecret", &masks)); | |
| 178 | assert!(reveals("value", &masks)); | |
| 179 | assert!(reveals(&base64(b"topsecret"), &masks)); | |
| 180 | assert!(!reveals("nothing here", &masks)); | |
| 181 | } | |
| 182 | } |