Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Merge Actions runs: summaries, attempts and re-runs, graceful cancel, log downloads, badges (actions 0009) | 1 | //! Downloading workflow logs, at GitHub's addresses: a run's (or one |
| 2 | //! attempt's) as a zip archive, and one job's as plain text. | |
| 3 | //! | |
| 4 | //! - `GET /repos/{owner}/{repo}/actions/runs/{id}/logs` | |
| 5 | //! - `GET /repos/{owner}/{repo}/actions/runs/{id}/attempts/{attempt}/logs` | |
| 6 | //! - `GET /repos/{owner}/{repo}/actions/jobs/{job}/logs?format=text` | |
| 7 | //! | |
| 8 | //! The archive holds `{n}_{job}.txt`, each job's whole log, and a folder | |
| 9 | //! per job with `{step}_{step name}.txt` for each step, as GitHub's does. | |
| 10 | //! Who may read the run may download its logs; a token needs the scope | |
| 11 | //! `get_job_logs` needs. | |
| 12 | ||
| 13 | use g1t_contracts::actions::{JobLogText, JobLogTextArgs, RunLogsArgs}; | |
| 14 | use g1t_contracts::repos::RepoPath; | |
| 15 | use g1t_contracts::{FailureCode, Outcome, Viewer}; | |
| 16 | use serde_json::json; | |
| 17 | use worker::{Response, Result}; | |
| 18 | ||
| 19 | use crate::operations::Op; | |
| 20 | use crate::operations::Services; | |
| 21 | use crate::{audit, fail, failure}; | |
| 22 | ||
| 23 | /// What a download path asks for. | |
| 24 | #[derive(Debug, PartialEq)] | |
| 25 | pub enum Wanted<'a> { | |
| 26 | Run { owner: &'a str, repo: &'a str, id: &'a str, attempt: Option<u64> }, | |
| 27 | Job { owner: &'a str, repo: &'a str, job: &'a str }, | |
| 28 | } | |
| 29 | ||
| 30 | /// The download a `GET` path (and its query) asks for, if it is one. | |
| 31 | pub fn wanted<'a>(path: &'a str, text: bool) -> Option<Wanted<'a>> { | |
| 32 | let parts: Vec<&str> = path.strip_prefix("/repos/")?.trim_end_matches('/').split('/').collect(); | |
| 33 | match parts.as_slice() { | |
| 34 | [owner, repo, "actions", "runs", id, "logs"] => Some(Wanted::Run { owner, repo, id, attempt: None }), | |
| 35 | [owner, repo, "actions", "runs", id, "attempts", attempt, "logs"] => { | |
| 36 | Some(Wanted::Run { owner, repo, id, attempt: Some(attempt.parse().ok()?) }) | |
| 37 | } | |
| 38 | [owner, repo, "actions", "jobs", job, "logs"] if text => Some(Wanted::Job { owner, repo, job }), | |
| 39 | _ => None, | |
| 40 | } | |
| 41 | } | |
| 42 | ||
| 43 | /// A name safe as a file name in an archive: no slashes or characters | |
| 44 | /// Windows refuses, at most 100 characters. | |
| 45 | pub fn file_name(name: &str) -> String { | |
| 46 | let cleaned: String = name | |
| 47 | .chars() | |
| 48 | .map(|c| if matches!(c, '/' | '\\' | ':' | '*' | '?' | '"' | '<' | '>' | '|') || c.is_control() { '_' } else { c }) | |
| 49 | .take(100) | |
| 50 | .collect(); | |
| 51 | let trimmed = cleaned.trim().trim_matches('.'); | |
| 52 | if trimmed.is_empty() { "job".to_owned() } else { trimmed.to_owned() } | |
| 53 | } | |
| 54 | ||
| 55 | /// A job's log as one text, in order. | |
| 56 | pub fn job_text(job: &JobLogText) -> String { | |
| 57 | if job.omitted { | |
| 58 | return "This job's log was left out: the run's logs are larger than one download holds. Download it on its own.\n".to_owned(); | |
| 59 | } | |
| 60 | job.chunks.iter().map(|chunk| chunk.text.as_str()).collect() | |
| 61 | } | |
| 62 | ||
| 63 | /// The files of a run's log archive, in order. | |
| 64 | pub fn archive_files(jobs: &[JobLogText]) -> Vec<(String, String)> { | |
| 65 | let mut files = Vec::new(); | |
| 66 | for (index, job) in jobs.iter().enumerate() { | |
| 67 | let name = file_name(&job.name); | |
| 68 | files.push((format!("{}_{name}.txt", index + 1), job_text(job))); | |
| 69 | if job.omitted { | |
| 70 | continue; | |
| 71 | } | |
| 72 | let mut steps: Vec<u32> = job.chunks.iter().map(|chunk| chunk.step).collect(); | |
| 73 | steps.sort_unstable(); | |
| 74 | steps.dedup(); | |
| 75 | for step in steps { | |
| 76 | let title = if step == 0 { | |
| 77 | "Set up job".to_owned() | |
| 78 | } else { | |
| 79 | job.steps.iter().find(|s| s.number == step).map_or_else(|| format!("Step {step}"), |s| s.name.clone()) | |
| 80 | }; | |
| 81 | let text: String = job.chunks.iter().filter(|chunk| chunk.step == step).map(|chunk| chunk.text.as_str()).collect(); | |
| 82 | files.push((format!("{name}/{}_{}.txt", step, file_name(&title)), text)); | |
| 83 | } | |
| 84 | } | |
| 85 | files | |
| 86 | } | |
| 87 | ||
| 88 | const CRC_POLY: u32 = 0xedb8_8320; | |
| 89 | ||
| 90 | fn crc32(data: &[u8]) -> u32 { | |
| 91 | let mut crc = 0xffff_ffffu32; | |
| 92 | for byte in data { | |
| 93 | crc ^= u32::from(*byte); | |
| 94 | for _ in 0..8 { | |
| 95 | crc = if crc & 1 == 1 { (crc >> 1) ^ CRC_POLY } else { crc >> 1 }; | |
| 96 | } | |
| 97 | } | |
| 98 | !crc | |
| 99 | } | |
| 100 | ||
| 101 | /// A zip archive of `files`, stored (not compressed), with UTF-8 names. | |
| 102 | /// Logs are small next to the 4 GB zip64 would be needed for. | |
| 103 | pub fn zip(files: &[(String, String)]) -> Vec<u8> { | |
| 104 | let mut out: Vec<u8> = Vec::new(); | |
| 105 | let mut central: Vec<u8> = Vec::new(); | |
| 106 | for (name, text) in files { | |
| 107 | let data = text.as_bytes(); | |
| 108 | let crc = crc32(data); | |
| 109 | let offset = out.len() as u32; | |
| 110 | let size = data.len() as u32; | |
| 111 | let name = name.as_bytes(); | |
| 112 | // Local file header: version 2.0, UTF-8 names (bit 11), stored. | |
| 113 | out.extend_from_slice(&0x0403_4b50u32.to_le_bytes()); | |
| 114 | out.extend_from_slice(&20u16.to_le_bytes()); | |
| 115 | out.extend_from_slice(&0x0800u16.to_le_bytes()); | |
| 116 | out.extend_from_slice(&0u16.to_le_bytes()); | |
| 117 | out.extend_from_slice(&0u16.to_le_bytes()); | |
| 118 | out.extend_from_slice(&0x21u16.to_le_bytes()); // 1980-01-01 | |
| 119 | out.extend_from_slice(&crc.to_le_bytes()); | |
| 120 | out.extend_from_slice(&size.to_le_bytes()); | |
| 121 | out.extend_from_slice(&size.to_le_bytes()); | |
| 122 | out.extend_from_slice(&(name.len() as u16).to_le_bytes()); | |
| 123 | out.extend_from_slice(&0u16.to_le_bytes()); | |
| 124 | out.extend_from_slice(name); | |
| 125 | out.extend_from_slice(data); | |
| 126 | // Its central directory entry. | |
| 127 | central.extend_from_slice(&0x0201_4b50u32.to_le_bytes()); | |
| 128 | central.extend_from_slice(&20u16.to_le_bytes()); | |
| 129 | central.extend_from_slice(&20u16.to_le_bytes()); | |
| 130 | central.extend_from_slice(&0x0800u16.to_le_bytes()); | |
| 131 | central.extend_from_slice(&0u16.to_le_bytes()); | |
| 132 | central.extend_from_slice(&0u16.to_le_bytes()); | |
| 133 | central.extend_from_slice(&0x21u16.to_le_bytes()); | |
| 134 | central.extend_from_slice(&crc.to_le_bytes()); | |
| 135 | central.extend_from_slice(&size.to_le_bytes()); | |
| 136 | central.extend_from_slice(&size.to_le_bytes()); | |
| 137 | central.extend_from_slice(&(name.len() as u16).to_le_bytes()); | |
| 138 | central.extend_from_slice(&[0u8; 12]); | |
| 139 | central.extend_from_slice(&offset.to_le_bytes()); | |
| 140 | central.extend_from_slice(name); | |
| 141 | } | |
| 142 | let start = out.len() as u32; | |
| 143 | let count = files.len() as u16; | |
| 144 | out.extend_from_slice(¢ral); | |
| 145 | out.extend_from_slice(&0x0605_4b50u32.to_le_bytes()); | |
| 146 | out.extend_from_slice(&[0u8; 4]); | |
| 147 | out.extend_from_slice(&count.to_le_bytes()); | |
| 148 | out.extend_from_slice(&count.to_le_bytes()); | |
| 149 | out.extend_from_slice(&(central.len() as u32).to_le_bytes()); | |
| 150 | out.extend_from_slice(&start.to_le_bytes()); | |
| 151 | out.extend_from_slice(&0u16.to_le_bytes()); | |
| 152 | out | |
| 153 | } | |
| 154 | ||
| 155 | fn attachment(bytes: Vec<u8>, content_type: &str, file: &str) -> Result<Response> { | |
| 156 | let mut response = Response::from_bytes(bytes)?; | |
| 157 | let headers = response.headers_mut(); | |
| 158 | headers.set("content-type", content_type)?; | |
| 159 | headers.set("content-disposition", &format!("attachment; filename=\"{}\"", file.replace('"', "")))?; | |
| 160 | headers.set("cache-control", "no-store")?; | |
| 161 | Ok(response) | |
| 162 | } | |
| 163 | ||
| 164 | /// Answers a download `wanted` names, for `viewer`. | |
| 165 | pub async fn download(services: &Services, viewer: &Viewer, wanted: Wanted<'_>) -> Result<Response> { | |
| 166 | let (owner, repo) = match &wanted { | |
| 167 | Wanted::Run { owner, repo, .. } | Wanted::Job { owner, repo, .. } => (*owner, *repo), | |
| 168 | }; | |
| 169 | // A workflow job's token reaches its own repository only, and any | |
| 170 | // token needs what reading a job's log needs. | |
| 171 | if viewer.as_ref().and_then(|user| user.token.as_deref()).is_some_and(|token| !token.reaches(&format!("{owner}/{repo}"))) { | |
| 172 | return fail(FailureCode::NotFound, "There is no such repository."); | |
| 173 | } | |
| 174 | let input = json!({ "repo": format!("{owner}/{repo}") }); | |
| 175 | if let Some(scope) = audit::missing_scope(Op::GetJobLogs, viewer, &input) { | |
| 176 | return Ok(Response::from_json(&json!({ | |
| 177 | "error": { "code": FailureCode::Forbidden, "message": "This token cannot read workflow logs.", "needed_scope": scope.as_str() } | |
| 178 | }))? | |
| 179 | .with_status(403)); | |
| 180 | } | |
| 181 | let path = RepoPath { namespace: owner.to_owned(), name: repo.to_owned() }; | |
| 182 | match wanted { | |
| 183 | Wanted::Run { id, attempt, .. } => { | |
| 184 | let logs: Outcome<Vec<JobLogText>> = | |
| 185 | g1t_kit::call(&services.actions, "run_logs", &RunLogsArgs { repo: path, viewer: viewer.clone(), id: id.to_owned(), attempt }).await?; | |
| 186 | match logs { | |
| 187 | Outcome::Ok(jobs) => { | |
| 188 | let file = match attempt { | |
| 189 | Some(n) => format!("logs_{id}_attempt_{n}.zip"), | |
| 190 | None => format!("logs_{id}.zip"), | |
| 191 | }; | |
| 192 | attachment(zip(&archive_files(&jobs)), "application/zip", &file) | |
| 193 | } | |
| 194 | Outcome::Fail(refused) => failure(&refused), | |
| 195 | } | |
| 196 | } | |
| 197 | Wanted::Job { job, .. } => { | |
| 198 | let log: Outcome<JobLogText> = | |
| 199 | g1t_kit::call(&services.actions, "job_log_text", &JobLogTextArgs { repo: path, viewer: viewer.clone(), job: job.to_owned() }).await?; | |
| 200 | match log { | |
| 201 | Outcome::Ok(log) => attachment(job_text(&log).into_bytes(), "text/plain; charset=utf-8", &format!("{}.txt", file_name(&log.name))), | |
| 202 | Outcome::Fail(refused) => failure(&refused), | |
| 203 | } | |
| 204 | } | |
| 205 | } | |
| 206 | } | |
| 207 | ||
| 208 | #[cfg(test)] | |
| 209 | mod tests { | |
| 210 | use g1t_contracts::actions::{JobLogText, LogChunk, StepState}; | |
| 211 | ||
| 212 | use super::*; | |
| 213 | ||
| 214 | fn job(name: &str, chunks: &[(u32, &str)]) -> JobLogText { | |
| 215 | JobLogText { | |
| 216 | job_id: "job_1".into(), | |
| 217 | name: name.into(), | |
| 218 | steps: vec![StepState { number: 1, name: "Run cargo test".into(), ..StepState::default() }], | |
| 219 | chunks: chunks.iter().enumerate().map(|(i, (step, text))| LogChunk { seq: i as u64 + 1, step: *step, text: (*text).into() }).collect(), | |
| 220 | done: true, | |
| 221 | omitted: false, | |
| 222 | } | |
| 223 | } | |
| 224 | ||
| 225 | #[test] | |
| 226 | fn download_paths_are_githubs() { | |
| 227 | assert_eq!( | |
| 228 | wanted("/repos/acme/web/actions/runs/run_1/logs", false), | |
| 229 | Some(Wanted::Run { owner: "acme", repo: "web", id: "run_1", attempt: None }) | |
| 230 | ); | |
| 231 | assert_eq!( | |
| 232 | wanted("/repos/acme/web/actions/runs/run_1/attempts/2/logs", false), | |
| 233 | Some(Wanted::Run { owner: "acme", repo: "web", id: "run_1", attempt: Some(2) }) | |
| 234 | ); | |
| 235 | assert_eq!(wanted("/repos/acme/web/actions/runs/run_1/attempts/x/logs", false), None); | |
| 236 | // A job's log is text when asked for; otherwise the JSON operation answers. | |
| 237 | assert_eq!(wanted("/repos/acme/web/actions/jobs/job_1/logs", false), None); | |
| 238 | assert_eq!(wanted("/repos/acme/web/actions/jobs/job_1/logs", true), Some(Wanted::Job { owner: "acme", repo: "web", job: "job_1" })); | |
| 239 | assert_eq!(wanted("/repos/acme/web/actions/runs/run_1", false), None); | |
| 240 | } | |
| 241 | ||
| 242 | #[test] | |
| 243 | fn names_are_safe_in_an_archive() { | |
| 244 | assert_eq!(file_name("test (ubuntu-latest, 20)"), "test (ubuntu-latest, 20)"); | |
| 245 | assert_eq!(file_name("build / a:b"), "build _ a_b"); | |
| 246 | assert_eq!(file_name(".."), "job"); | |
| 247 | assert_eq!(file_name(&"x".repeat(300)).len(), 100); | |
| 248 | } | |
| 249 | ||
| 250 | #[test] | |
| 251 | fn an_archive_has_each_job_whole_and_by_step() { | |
| 252 | let jobs = [job("test", &[(0, "set up\n"), (1, "running\n"), (1, "ok\n")]), JobLogText { omitted: true, ..job("deploy/x", &[]) }]; | |
| 253 | let files = archive_files(&jobs); | |
| 254 | let names: Vec<&str> = files.iter().map(|(name, _)| name.as_str()).collect(); | |
| 255 | assert_eq!(names, ["1_test.txt", "test/0_Set up job.txt", "test/1_Run cargo test.txt", "2_deploy_x.txt"]); | |
| 256 | assert_eq!(files[0].1, "set up\nrunning\nok\n"); | |
| 257 | assert_eq!(files[2].1, "running\nok\n"); | |
| 258 | assert!(files[3].1.contains("left out")); | |
| 259 | } | |
| 260 | ||
| 261 | #[test] | |
| 262 | fn the_archive_is_a_zip() { | |
| 263 | let bytes = zip(&[("a.txt".into(), "hello".into()), ("dir/b.txt".into(), String::new())]); | |
| 264 | assert_eq!(&bytes[..4], &[0x50, 0x4b, 0x03, 0x04]); | |
| 265 | // The end of the central directory names both files. | |
| 266 | let end = bytes.len() - 22; | |
| 267 | assert_eq!(&bytes[end..end + 4], &[0x50, 0x4b, 0x05, 0x06]); | |
| 268 | assert_eq!(u16::from_le_bytes([bytes[end + 10], bytes[end + 11]]), 2); | |
| 269 | assert_eq!(crc32(b"hello"), 0x3610_a686); | |
| 270 | assert_eq!(crc32(b""), 0); | |
| 271 | } | |
| 272 | } |
This file's history is long; its oldest lines are credited to the oldest commit read.