| 1 | //! Downloading workflow logs, at GitHub's addresses: a run's (or one |
| 2 | //! attempt's) as a zip archive, and one job's as plain text. |
| 3 | //! |
| 4 | //! - `GET /repos/{owner}/{repo}/actions/runs/{id}/logs` |
| 5 | //! - `GET /repos/{owner}/{repo}/actions/runs/{id}/attempts/{attempt}/logs` |
| 6 | //! - `GET /repos/{owner}/{repo}/actions/jobs/{job}/logs?format=text` |
| 7 | //! |
| 8 | //! The archive holds `{n}_{job}.txt`, each job's whole log, and a folder |
| 9 | //! per job with `{step}_{step name}.txt` for each step, as GitHub's does. |
| 10 | //! Who may read the run may download its logs; a token needs the scope |
| 11 | //! `get_job_logs` needs. |
| 12 | |
| 13 | use g1t_contracts::actions::{JobLogText, JobLogTextArgs, RunLogsArgs}; |
| 14 | use g1t_contracts::repos::RepoPath; |
| 15 | use g1t_contracts::{FailureCode, Outcome, Viewer}; |
| 16 | use serde_json::json; |
| 17 | use worker::{Response, Result}; |
| 18 | |
| 19 | use crate::operations::Op; |
| 20 | use crate::operations::Services; |
| 21 | use crate::{audit, fail, failure}; |
| 22 | |
| 23 | /// What a download path asks for. |
| 24 | #[derive(Debug, PartialEq)] |
| 25 | pub enum Wanted<'a> { |
| 26 | Run { owner: &'a str, repo: &'a str, id: &'a str, attempt: Option<u64> }, |
| 27 | Job { owner: &'a str, repo: &'a str, job: &'a str }, |
| 28 | } |
| 29 | |
| 30 | /// The download a `GET` path (and its query) asks for, if it is one. |
| 31 | pub fn wanted<'a>(path: &'a str, text: bool) -> Option<Wanted<'a>> { |
| 32 | let parts: Vec<&str> = path.strip_prefix("/repos/")?.trim_end_matches('/').split('/').collect(); |
| 33 | match parts.as_slice() { |
| 34 | [owner, repo, "actions", "runs", id, "logs"] => Some(Wanted::Run { owner, repo, id, attempt: None }), |
| 35 | [owner, repo, "actions", "runs", id, "attempts", attempt, "logs"] => { |
| 36 | Some(Wanted::Run { owner, repo, id, attempt: Some(attempt.parse().ok()?) }) |
| 37 | } |
| 38 | [owner, repo, "actions", "jobs", job, "logs"] if text => Some(Wanted::Job { owner, repo, job }), |
| 39 | _ => None, |
| 40 | } |
| 41 | } |
| 42 | |
| 43 | /// A name safe as a file name in an archive: no slashes or characters |
| 44 | /// Windows refuses, at most 100 characters. |
| 45 | pub fn file_name(name: &str) -> String { |
| 46 | let cleaned: String = name |
| 47 | .chars() |
| 48 | .map(|c| if matches!(c, '/' | '\\' | ':' | '*' | '?' | '"' | '<' | '>' | '|') || c.is_control() { '_' } else { c }) |
| 49 | .take(100) |
| 50 | .collect(); |
| 51 | let trimmed = cleaned.trim().trim_matches('.'); |
| 52 | if trimmed.is_empty() { "job".to_owned() } else { trimmed.to_owned() } |
| 53 | } |
| 54 | |
| 55 | /// A job's log as one text, in order. |
| 56 | pub fn job_text(job: &JobLogText) -> String { |
| 57 | if job.omitted { |
| 58 | return "This job's log was left out: the run's logs are larger than one download holds. Download it on its own.\n".to_owned(); |
| 59 | } |
| 60 | job.chunks.iter().map(|chunk| chunk.text.as_str()).collect() |
| 61 | } |
| 62 | |
| 63 | /// The files of a run's log archive, in order. |
| 64 | pub fn archive_files(jobs: &[JobLogText]) -> Vec<(String, String)> { |
| 65 | let mut files = Vec::new(); |
| 66 | for (index, job) in jobs.iter().enumerate() { |
| 67 | let name = file_name(&job.name); |
| 68 | files.push((format!("{}_{name}.txt", index + 1), job_text(job))); |
| 69 | if job.omitted { |
| 70 | continue; |
| 71 | } |
| 72 | let mut steps: Vec<u32> = job.chunks.iter().map(|chunk| chunk.step).collect(); |
| 73 | steps.sort_unstable(); |
| 74 | steps.dedup(); |
| 75 | for step in steps { |
| 76 | let title = if step == 0 { |
| 77 | "Set up job".to_owned() |
| 78 | } else { |
| 79 | job.steps.iter().find(|s| s.number == step).map_or_else(|| format!("Step {step}"), |s| s.name.clone()) |
| 80 | }; |
| 81 | let text: String = job.chunks.iter().filter(|chunk| chunk.step == step).map(|chunk| chunk.text.as_str()).collect(); |
| 82 | files.push((format!("{name}/{}_{}.txt", step, file_name(&title)), text)); |
| 83 | } |
| 84 | } |
| 85 | files |
| 86 | } |
| 87 | |
| 88 | const CRC_POLY: u32 = 0xedb8_8320; |
| 89 | |
| 90 | fn crc32(data: &[u8]) -> u32 { |
| 91 | let mut crc = 0xffff_ffffu32; |
| 92 | for byte in data { |
| 93 | crc ^= u32::from(*byte); |
| 94 | for _ in 0..8 { |
| 95 | crc = if crc & 1 == 1 { (crc >> 1) ^ CRC_POLY } else { crc >> 1 }; |
| 96 | } |
| 97 | } |
| 98 | !crc |
| 99 | } |
| 100 | |
| 101 | /// A zip archive of `files`, stored (not compressed), with UTF-8 names. |
| 102 | /// Logs are small next to the 4 GB zip64 would be needed for. |
| 103 | pub fn zip(files: &[(String, String)]) -> Vec<u8> { |
| 104 | let mut out: Vec<u8> = Vec::new(); |
| 105 | let mut central: Vec<u8> = Vec::new(); |
| 106 | for (name, text) in files { |
| 107 | let data = text.as_bytes(); |
| 108 | let crc = crc32(data); |
| 109 | let offset = out.len() as u32; |
| 110 | let size = data.len() as u32; |
| 111 | let name = name.as_bytes(); |
| 112 | // Local file header: version 2.0, UTF-8 names (bit 11), stored. |
| 113 | out.extend_from_slice(&0x0403_4b50u32.to_le_bytes()); |
| 114 | out.extend_from_slice(&20u16.to_le_bytes()); |
| 115 | out.extend_from_slice(&0x0800u16.to_le_bytes()); |
| 116 | out.extend_from_slice(&0u16.to_le_bytes()); |
| 117 | out.extend_from_slice(&0u16.to_le_bytes()); |
| 118 | out.extend_from_slice(&0x21u16.to_le_bytes()); // 1980-01-01 |
| 119 | out.extend_from_slice(&crc.to_le_bytes()); |
| 120 | out.extend_from_slice(&size.to_le_bytes()); |
| 121 | out.extend_from_slice(&size.to_le_bytes()); |
| 122 | out.extend_from_slice(&(name.len() as u16).to_le_bytes()); |
| 123 | out.extend_from_slice(&0u16.to_le_bytes()); |
| 124 | out.extend_from_slice(name); |
| 125 | out.extend_from_slice(data); |
| 126 | // Its central directory entry. |
| 127 | central.extend_from_slice(&0x0201_4b50u32.to_le_bytes()); |
| 128 | central.extend_from_slice(&20u16.to_le_bytes()); |
| 129 | central.extend_from_slice(&20u16.to_le_bytes()); |
| 130 | central.extend_from_slice(&0x0800u16.to_le_bytes()); |
| 131 | central.extend_from_slice(&0u16.to_le_bytes()); |
| 132 | central.extend_from_slice(&0u16.to_le_bytes()); |
| 133 | central.extend_from_slice(&0x21u16.to_le_bytes()); |
| 134 | central.extend_from_slice(&crc.to_le_bytes()); |
| 135 | central.extend_from_slice(&size.to_le_bytes()); |
| 136 | central.extend_from_slice(&size.to_le_bytes()); |
| 137 | central.extend_from_slice(&(name.len() as u16).to_le_bytes()); |
| 138 | central.extend_from_slice(&[0u8; 12]); |
| 139 | central.extend_from_slice(&offset.to_le_bytes()); |
| 140 | central.extend_from_slice(name); |
| 141 | } |
| 142 | let start = out.len() as u32; |
| 143 | let count = files.len() as u16; |
| 144 | out.extend_from_slice(¢ral); |
| 145 | out.extend_from_slice(&0x0605_4b50u32.to_le_bytes()); |
| 146 | out.extend_from_slice(&[0u8; 4]); |
| 147 | out.extend_from_slice(&count.to_le_bytes()); |
| 148 | out.extend_from_slice(&count.to_le_bytes()); |
| 149 | out.extend_from_slice(&(central.len() as u32).to_le_bytes()); |
| 150 | out.extend_from_slice(&start.to_le_bytes()); |
| 151 | out.extend_from_slice(&0u16.to_le_bytes()); |
| 152 | out |
| 153 | } |
| 154 | |
| 155 | fn attachment(bytes: Vec<u8>, content_type: &str, file: &str) -> Result<Response> { |
| 156 | let mut response = Response::from_bytes(bytes)?; |
| 157 | let headers = response.headers_mut(); |
| 158 | headers.set("content-type", content_type)?; |
| 159 | headers.set("content-disposition", &format!("attachment; filename=\"{}\"", file.replace('"', "")))?; |
| 160 | headers.set("cache-control", "no-store")?; |
| 161 | Ok(response) |
| 162 | } |
| 163 | |
| 164 | /// Answers a download `wanted` names, for `viewer`. |
| 165 | pub async fn download(services: &Services, viewer: &Viewer, wanted: Wanted<'_>) -> Result<Response> { |
| 166 | let (owner, repo) = match &wanted { |
| 167 | Wanted::Run { owner, repo, .. } | Wanted::Job { owner, repo, .. } => (*owner, *repo), |
| 168 | }; |
| 169 | // A workflow job's token reaches its own repository only, and any |
| 170 | // token needs what reading a job's log needs. |
| 171 | if viewer.as_ref().and_then(|user| user.token.as_deref()).is_some_and(|token| !token.reaches(&format!("{owner}/{repo}"))) { |
| 172 | return fail(FailureCode::NotFound, "There is no such repository."); |
| 173 | } |
| 174 | let input = json!({ "repo": format!("{owner}/{repo}") }); |
| 175 | if let Some(scope) = audit::missing_scope(Op::GetJobLogs, viewer, &input) { |
| 176 | return Ok(Response::from_json(&json!({ |
| 177 | "error": { "code": FailureCode::Forbidden, "message": "This token cannot read workflow logs.", "needed_scope": scope.as_str() } |
| 178 | }))? |
| 179 | .with_status(403)); |
| 180 | } |
| 181 | let path = RepoPath { namespace: owner.to_owned(), name: repo.to_owned() }; |
| 182 | match wanted { |
| 183 | Wanted::Run { id, attempt, .. } => { |
| 184 | let logs: Outcome<Vec<JobLogText>> = |
| 185 | g1t_kit::call(&services.actions, "run_logs", &RunLogsArgs { repo: path, viewer: viewer.clone(), id: id.to_owned(), attempt }).await?; |
| 186 | match logs { |
| 187 | Outcome::Ok(jobs) => { |
| 188 | let file = match attempt { |
| 189 | Some(n) => format!("logs_{id}_attempt_{n}.zip"), |
| 190 | None => format!("logs_{id}.zip"), |
| 191 | }; |
| 192 | attachment(zip(&archive_files(&jobs)), "application/zip", &file) |
| 193 | } |
| 194 | Outcome::Fail(refused) => failure(&refused), |
| 195 | } |
| 196 | } |
| 197 | Wanted::Job { job, .. } => { |
| 198 | let log: Outcome<JobLogText> = |
| 199 | g1t_kit::call(&services.actions, "job_log_text", &JobLogTextArgs { repo: path, viewer: viewer.clone(), job: job.to_owned() }).await?; |
| 200 | match log { |
| 201 | Outcome::Ok(log) => attachment(job_text(&log).into_bytes(), "text/plain; charset=utf-8", &format!("{}.txt", file_name(&log.name))), |
| 202 | Outcome::Fail(refused) => failure(&refused), |
| 203 | } |
| 204 | } |
| 205 | } |
| 206 | } |
| 207 | |
| 208 | #[cfg(test)] |
| 209 | mod tests { |
| 210 | use g1t_contracts::actions::{JobLogText, LogChunk, StepState}; |
| 211 | |
| 212 | use super::*; |
| 213 | |
| 214 | fn job(name: &str, chunks: &[(u32, &str)]) -> JobLogText { |
| 215 | JobLogText { |
| 216 | job_id: "job_1".into(), |
| 217 | name: name.into(), |
| 218 | steps: vec![StepState { number: 1, name: "Run cargo test".into(), ..StepState::default() }], |
| 219 | chunks: chunks.iter().enumerate().map(|(i, (step, text))| LogChunk { seq: i as u64 + 1, step: *step, text: (*text).into() }).collect(), |
| 220 | done: true, |
| 221 | omitted: false, |
| 222 | } |
| 223 | } |
| 224 | |
| 225 | #[test] |
| 226 | fn download_paths_are_githubs() { |
| 227 | assert_eq!( |
| 228 | wanted("/repos/acme/web/actions/runs/run_1/logs", false), |
| 229 | Some(Wanted::Run { owner: "acme", repo: "web", id: "run_1", attempt: None }) |
| 230 | ); |
| 231 | assert_eq!( |
| 232 | wanted("/repos/acme/web/actions/runs/run_1/attempts/2/logs", false), |
| 233 | Some(Wanted::Run { owner: "acme", repo: "web", id: "run_1", attempt: Some(2) }) |
| 234 | ); |
| 235 | assert_eq!(wanted("/repos/acme/web/actions/runs/run_1/attempts/x/logs", false), None); |
| 236 | // A job's log is text when asked for; otherwise the JSON operation answers. |
| 237 | assert_eq!(wanted("/repos/acme/web/actions/jobs/job_1/logs", false), None); |
| 238 | assert_eq!(wanted("/repos/acme/web/actions/jobs/job_1/logs", true), Some(Wanted::Job { owner: "acme", repo: "web", job: "job_1" })); |
| 239 | assert_eq!(wanted("/repos/acme/web/actions/runs/run_1", false), None); |
| 240 | } |
| 241 | |
| 242 | #[test] |
| 243 | fn names_are_safe_in_an_archive() { |
| 244 | assert_eq!(file_name("test (ubuntu-latest, 20)"), "test (ubuntu-latest, 20)"); |
| 245 | assert_eq!(file_name("build / a:b"), "build _ a_b"); |
| 246 | assert_eq!(file_name(".."), "job"); |
| 247 | assert_eq!(file_name(&"x".repeat(300)).len(), 100); |
| 248 | } |
| 249 | |
| 250 | #[test] |
| 251 | fn an_archive_has_each_job_whole_and_by_step() { |
| 252 | let jobs = [job("test", &[(0, "set up\n"), (1, "running\n"), (1, "ok\n")]), JobLogText { omitted: true, ..job("deploy/x", &[]) }]; |
| 253 | let files = archive_files(&jobs); |
| 254 | let names: Vec<&str> = files.iter().map(|(name, _)| name.as_str()).collect(); |
| 255 | assert_eq!(names, ["1_test.txt", "test/0_Set up job.txt", "test/1_Run cargo test.txt", "2_deploy_x.txt"]); |
| 256 | assert_eq!(files[0].1, "set up\nrunning\nok\n"); |
| 257 | assert_eq!(files[2].1, "running\nok\n"); |
| 258 | assert!(files[3].1.contains("left out")); |
| 259 | } |
| 260 | |
| 261 | #[test] |
| 262 | fn the_archive_is_a_zip() { |
| 263 | let bytes = zip(&[("a.txt".into(), "hello".into()), ("dir/b.txt".into(), String::new())]); |
| 264 | assert_eq!(&bytes[..4], &[0x50, 0x4b, 0x03, 0x04]); |
| 265 | // The end of the central directory names both files. |
| 266 | let end = bytes.len() - 22; |
| 267 | assert_eq!(&bytes[end..end + 4], &[0x50, 0x4b, 0x05, 0x06]); |
| 268 | assert_eq!(u16::from_le_bytes([bytes[end + 10], bytes[end + 11]]), 2); |
| 269 | assert_eq!(crc32(b"hello"), 0x3610_a686); |
| 270 | assert_eq!(crc32(b""), 0); |
| 271 | } |
| 272 | } |