Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 1 | //! Reading the objects in a git pack, as a push sends them, so that what a |
| 2 | //! push adds can be looked at before it is stored. | |
| 3 | //! | |
| 4 | //! A pushed pack is usually thin: some objects are deltas against objects | |
| 5 | //! the repository already has. Those are left pending until the caller | |
| 6 | //! supplies their bases with [`Pack::supply`]. | |
| Catching up with main takes seconds when the two sides touched different files | 7 | //! |
| 8 | //! Writing is the small part g1t needs for a merge it makes itself: a pack | |
| 9 | //! of whole objects ([`write_pack`]), or one fetched from elsewhere with a | |
| 10 | //! few objects added ([`extend_pack`]). | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 11 | |
| 12 | use std::collections::HashMap; | |
| 13 | ||
| 14 | use miniz_oxide::inflate::TINFLStatus; | |
| 15 | use miniz_oxide::inflate::core::{DecompressorOxide, decompress, inflate_flags}; | |
| 16 | use sha1::{Digest, Sha1}; | |
| 17 | ||
| 18 | /// Beyond this much inflated content the pack is not read: a push that | |
| 19 | /// large is let through unread rather than risk the worker's memory. | |
| 20 | pub const MAX_INFLATED: usize = 48 * 1024 * 1024; | |
| 21 | ||
| 22 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 23 | pub enum ObjectKind { | |
| 24 | Commit, | |
| 25 | Tree, | |
| 26 | Blob, | |
| 27 | Tag, | |
| 28 | } | |
| 29 | ||
| 30 | impl ObjectKind { | |
| 31 | fn from_type(code: u8) -> Option<ObjectKind> { | |
| 32 | Some(match code { | |
| 33 | 1 => ObjectKind::Commit, | |
| 34 | 2 => ObjectKind::Tree, | |
| 35 | 3 => ObjectKind::Blob, | |
| 36 | 4 => ObjectKind::Tag, | |
| 37 | _ => return None, | |
| 38 | }) | |
| 39 | } | |
| 40 | ||
| Catching up with main takes seconds when the two sides touched different files | 41 | /// The type code a pack gives objects of this kind. |
| 42 | fn code(self) -> u8 { | |
| 43 | match self { | |
| 44 | ObjectKind::Commit => 1, | |
| 45 | ObjectKind::Tree => 2, | |
| 46 | ObjectKind::Blob => 3, | |
| 47 | ObjectKind::Tag => 4, | |
| 48 | } | |
| 49 | } | |
| 50 | ||
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 51 | fn name(self) -> &'static str { |
| 52 | match self { | |
| 53 | ObjectKind::Commit => "commit", | |
| 54 | ObjectKind::Tree => "tree", | |
| 55 | ObjectKind::Blob => "blob", | |
| 56 | ObjectKind::Tag => "tag", | |
| 57 | } | |
| 58 | } | |
| 59 | } | |
| 60 | ||
| 61 | /// A git object's id: the SHA-1 of its header and content, in hex. | |
| 62 | pub fn object_id(kind: ObjectKind, data: &[u8]) -> String { | |
| 63 | let mut hasher = Sha1::new(); | |
| 64 | hasher.update(format!("{} {}\0", kind.name(), data.len()).as_bytes()); | |
| 65 | hasher.update(data); | |
| 66 | hasher.finalize().iter().map(|byte| format!("{byte:02x}")).collect() | |
| 67 | } | |
| 68 | ||
| 69 | enum Base { | |
| 70 | /// An earlier object in the pack, by its offset. | |
| 71 | Offset(usize), | |
| 72 | /// Any object, by id. | |
| 73 | Id(String), | |
| 74 | } | |
| 75 | ||
| 76 | struct Delta { | |
| 77 | base: Base, | |
| 78 | data: Vec<u8>, | |
| 79 | } | |
| 80 | ||
| 81 | /// The objects of a pack, by id. | |
| 82 | #[derive(Default)] | |
| 83 | pub struct Pack { | |
| 84 | objects: HashMap<String, (ObjectKind, Vec<u8>)>, | |
| 85 | /// Ids of the objects at each offset, once resolved. | |
| 86 | at_offset: HashMap<usize, String>, | |
| 87 | pending: Vec<(usize, Delta)>, | |
| 88 | /// Commits, in the order the pack holds them. | |
| 89 | commits: Vec<String>, | |
| 90 | } | |
| 91 | ||
| 92 | /// Where the pack starts in a receive-pack request: after the commands | |
| 93 | /// and anything else sent as pkt-lines. | |
| 94 | pub fn pack_start(body: &[u8]) -> Option<usize> { | |
| 95 | let mut at = 0; | |
| 96 | loop { | |
| 97 | if body.get(at..at + 4) == Some(b"PACK") { | |
| 98 | return Some(at); | |
| 99 | } | |
| 100 | let length = std::str::from_utf8(body.get(at..at + 4)?) | |
| 101 | .ok() | |
| 102 | .and_then(|hex| usize::from_str_radix(hex, 16).ok())?; | |
| 103 | // A flush packet is four bytes; any other line counts its own length. | |
| 104 | at += if length == 0 { 4 } else { length.max(4) }; | |
| 105 | } | |
| 106 | } | |
| 107 | ||
| 108 | fn inflate(input: &[u8], size: usize) -> Result<(Vec<u8>, usize), String> { | |
| 109 | let mut out = vec![0u8; size.max(1)]; | |
| 110 | let mut state = DecompressorOxide::new(); | |
| 111 | let flags = inflate_flags::TINFL_FLAG_PARSE_ZLIB_HEADER | inflate_flags::TINFL_FLAG_USING_NON_WRAPPING_OUTPUT_BUF; | |
| 112 | let (status, consumed, written) = decompress(&mut state, input, &mut out, 0, flags); | |
| 113 | match status { | |
| 114 | TINFLStatus::Done => { | |
| 115 | out.truncate(written); | |
| 116 | if written != size { | |
| 117 | return Err(format!("an object inflated to {written} bytes, not {size}")); | |
| 118 | } | |
| 119 | Ok((out, consumed)) | |
| 120 | } | |
| 121 | other => Err(format!("an object could not be inflated: {other:?}")), | |
| 122 | } | |
| 123 | } | |
| 124 | ||
| 125 | fn varint(data: &[u8], at: &mut usize) -> Option<usize> { | |
| 126 | let (mut value, mut shift) = (0usize, 0); | |
| 127 | loop { | |
| 128 | let byte = *data.get(*at)?; | |
| 129 | *at += 1; | |
| 130 | value |= ((byte & 0x7f) as usize) << shift; | |
| 131 | shift += 7; | |
| 132 | if byte & 0x80 == 0 || shift > 56 { | |
| 133 | return Some(value); | |
| 134 | } | |
| 135 | } | |
| 136 | } | |
| 137 | ||
| Catching up with main takes seconds when the two sides touched different files | 138 | /// An entry's header in a pack: its type and its inflated size. |
| 139 | fn header(code: u8, size: usize) -> Vec<u8> { | |
| 140 | let mut out = Vec::new(); | |
| 141 | let mut byte = (code << 4) | (size & 15) as u8; | |
| 142 | let mut rest = size >> 4; | |
| 143 | while rest > 0 { | |
| 144 | out.push(byte | 0x80); | |
| 145 | byte = (rest & 0x7f) as u8; | |
| 146 | rest >>= 7; | |
| 147 | } | |
| 148 | out.push(byte); | |
| 149 | out | |
| 150 | } | |
| 151 | ||
| 152 | fn write_entries(pack: &mut Vec<u8>, objects: &[(ObjectKind, Vec<u8>)]) { | |
| 153 | for (kind, data) in objects { | |
| 154 | pack.extend(header(kind.code(), data.len())); | |
| 155 | pack.extend(miniz_oxide::deflate::compress_to_vec_zlib(data, 6)); | |
| 156 | } | |
| 157 | } | |
| 158 | ||
| 159 | fn seal(mut pack: Vec<u8>) -> Vec<u8> { | |
| 160 | let checksum = Sha1::digest(&pack); | |
| 161 | pack.extend_from_slice(&checksum); | |
| 162 | pack | |
| 163 | } | |
| 164 | ||
| 165 | /// A version 2 pack holding `objects` whole, with no deltas: what git's | |
| 166 | /// receive-pack takes, when the objects are few and new. | |
| 167 | pub fn write_pack(objects: &[(ObjectKind, Vec<u8>)]) -> Vec<u8> { | |
| 168 | let mut pack = b"PACK".to_vec(); | |
| 169 | pack.extend_from_slice(&2u32.to_be_bytes()); | |
| 170 | pack.extend_from_slice(&(objects.len() as u32).to_be_bytes()); | |
| 171 | write_entries(&mut pack, objects); | |
| 172 | seal(pack) | |
| 173 | } | |
| 174 | ||
| 175 | /// `pack` with `objects` added after its own, as one pack. Its entries keep | |
| 176 | /// their offsets, since the header stays the same length, so its deltas | |
| 177 | /// still find their bases. `pack` must not be thin. | |
| 178 | pub fn extend_pack(pack: &[u8], objects: &[(ObjectKind, Vec<u8>)]) -> Result<Vec<u8>, String> { | |
| 179 | if pack.len() < 32 || &pack[..4] != b"PACK" { | |
| 180 | return Err("not a pack".into()); | |
| 181 | } | |
| 182 | let (body, trailer) = pack.split_at(pack.len() - 20); | |
| 183 | if Sha1::digest(body).as_slice() != trailer { | |
| 184 | return Err("the pack's checksum does not match".into()); | |
| 185 | } | |
| 186 | let count = u32::from_be_bytes([pack[8], pack[9], pack[10], pack[11]]) as usize; | |
| 187 | let total = u32::try_from(count + objects.len()).map_err(|_| "the pack is too large")?; | |
| 188 | let mut out = body.to_vec(); | |
| 189 | out[8..12].copy_from_slice(&total.to_be_bytes()); | |
| 190 | write_entries(&mut out, objects); | |
| 191 | Ok(seal(out)) | |
| 192 | } | |
| 193 | ||
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 194 | /// Applies a git delta to its base. |
| 195 | pub fn apply_delta(base: &[u8], delta: &[u8]) -> Result<Vec<u8>, String> { | |
| 196 | let mut at = 0; | |
| 197 | let bad = || "a delta is malformed".to_owned(); | |
| 198 | let source = varint(delta, &mut at).ok_or_else(bad)?; | |
| 199 | if source != base.len() { | |
| 200 | return Err("a delta does not fit its base".into()); | |
| 201 | } | |
| 202 | let target = varint(delta, &mut at).ok_or_else(bad)?; | |
| 203 | let mut out = Vec::with_capacity(target); | |
| 204 | while at < delta.len() { | |
| 205 | let op = delta[at]; | |
| 206 | at += 1; | |
| 207 | if op & 0x80 != 0 { | |
| 208 | let mut offset = 0usize; | |
| 209 | let mut size = 0usize; | |
| 210 | for bit in 0..4 { | |
| 211 | if op & (1 << bit) != 0 { | |
| 212 | offset |= (*delta.get(at).ok_or_else(bad)? as usize) << (8 * bit); | |
| 213 | at += 1; | |
| 214 | } | |
| 215 | } | |
| 216 | for bit in 0..3 { | |
| 217 | if op & (0x10 << bit) != 0 { | |
| 218 | size |= (*delta.get(at).ok_or_else(bad)? as usize) << (8 * bit); | |
| 219 | at += 1; | |
| 220 | } | |
| 221 | } | |
| 222 | if size == 0 { | |
| 223 | size = 0x10000; | |
| 224 | } | |
| 225 | out.extend_from_slice(base.get(offset..offset + size).ok_or_else(bad)?); | |
| 226 | } else if op != 0 { | |
| 227 | out.extend_from_slice(delta.get(at..at + op as usize).ok_or_else(bad)?); | |
| 228 | at += op as usize; | |
| 229 | } else { | |
| 230 | return Err(bad()); | |
| 231 | } | |
| 232 | } | |
| 233 | if out.len() != target { | |
| 234 | return Err(bad()); | |
| 235 | } | |
| 236 | Ok(out) | |
| 237 | } | |
| 238 | ||
| 239 | impl Pack { | |
| 240 | /// Reads every object in `pack`, resolving the deltas whose bases are | |
| 241 | /// in it. | |
| 242 | pub fn parse(pack: &[u8]) -> Result<Pack, String> { | |
| 243 | if pack.len() < 12 || &pack[..4] != b"PACK" { | |
| 244 | return Err("not a pack".into()); | |
| 245 | } | |
| 246 | let count = u32::from_be_bytes([pack[8], pack[9], pack[10], pack[11]]) as usize; | |
| 247 | let mut at = 12; | |
| 248 | let mut result = Pack::default(); | |
| 249 | let mut inflated = 0usize; | |
| 250 | for _ in 0..count { | |
| 251 | let start = at; | |
| 252 | let mut byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 253 | at += 1; | |
| 254 | let code = (byte >> 4) & 7; | |
| 255 | let mut size = (byte & 15) as usize; | |
| 256 | let mut shift = 4; | |
| 257 | while byte & 0x80 != 0 { | |
| 258 | byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 259 | at += 1; | |
| 260 | size |= ((byte & 0x7f) as usize) << shift; | |
| 261 | shift += 7; | |
| 262 | } | |
| 263 | inflated += size; | |
| 264 | if inflated > MAX_INFLATED { | |
| 265 | return Err("the pack is too large to read".into()); | |
| 266 | } | |
| 267 | let base = match code { | |
| 268 | 6 => { | |
| 269 | let mut byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 270 | at += 1; | |
| 271 | let mut offset = (byte & 0x7f) as usize; | |
| 272 | while byte & 0x80 != 0 { | |
| 273 | byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 274 | at += 1; | |
| 275 | offset = ((offset + 1) << 7) | (byte & 0x7f) as usize; | |
| 276 | } | |
| 277 | Some(Base::Offset(start.checked_sub(offset).ok_or("a delta points before the pack")?)) | |
| 278 | } | |
| 279 | 7 => { | |
| 280 | let id = pack.get(at..at + 20).ok_or("the pack ends early")?; | |
| 281 | at += 20; | |
| 282 | Some(Base::Id(id.iter().map(|byte| format!("{byte:02x}")).collect())) | |
| 283 | } | |
| 284 | _ => None, | |
| 285 | }; | |
| 286 | let (data, consumed) = inflate(&pack[at..], size)?; | |
| 287 | at += consumed; | |
| 288 | match base { | |
| 289 | Some(base) => result.pending.push((start, Delta { base, data })), | |
| 290 | None => { | |
| 291 | let kind = ObjectKind::from_type(code).ok_or("an object has an unknown type")?; | |
| 292 | result.insert(start, kind, data); | |
| 293 | } | |
| 294 | } | |
| 295 | } | |
| 296 | result.resolve(); | |
| 297 | Ok(result) | |
| 298 | } | |
| 299 | ||
| 300 | fn insert(&mut self, offset: usize, kind: ObjectKind, data: Vec<u8>) { | |
| 301 | let id = object_id(kind, &data); | |
| 302 | if kind == ObjectKind::Commit { | |
| 303 | self.commits.push(id.clone()); | |
| 304 | } | |
| 305 | self.at_offset.insert(offset, id.clone()); | |
| 306 | self.objects.insert(id, (kind, data)); | |
| 307 | } | |
| 308 | ||
| 309 | /// Resolves every pending delta whose base is known by now. | |
| 310 | fn resolve(&mut self) { | |
| 311 | loop { | |
| 312 | let mut progress = false; | |
| 313 | let pending = std::mem::take(&mut self.pending); | |
| 314 | for (offset, delta) in pending { | |
| 315 | let base_id = match &delta.base { | |
| 316 | Base::Offset(base) => self.at_offset.get(base).cloned(), | |
| 317 | Base::Id(id) => Some(id.clone()), | |
| 318 | }; | |
| 319 | let resolved = base_id | |
| 320 | .and_then(|id| self.objects.get(&id)) | |
| 321 | .map(|(kind, base)| (*kind, apply_delta(base, &delta.data))); | |
| 322 | match resolved { | |
| 323 | Some((kind, Ok(data))) => { | |
| 324 | self.insert(offset, kind, data); | |
| 325 | progress = true; | |
| 326 | } | |
| 327 | // A delta that does not apply is dropped. | |
| 328 | Some((_, Err(_))) => progress = true, | |
| 329 | None => self.pending.push((offset, delta)), | |
| 330 | } | |
| 331 | } | |
| 332 | if !progress || self.pending.is_empty() { | |
| 333 | return; | |
| 334 | } | |
| 335 | } | |
| 336 | } | |
| 337 | ||
| 338 | /// Objects the pack's deltas are based on that it does not hold: what | |
| 339 | /// the repository has to supply. | |
| 340 | pub fn missing_bases(&self) -> Vec<String> { | |
| 341 | let mut ids: Vec<String> = self | |
| 342 | .pending | |
| 343 | .iter() | |
| 344 | .filter_map(|(_, delta)| match &delta.base { | |
| 345 | Base::Id(id) if !self.objects.contains_key(id) => Some(id.clone()), | |
| 346 | _ => None, | |
| 347 | }) | |
| 348 | .collect(); | |
| 349 | ids.sort(); | |
| 350 | ids.dedup(); | |
| 351 | ids | |
| 352 | } | |
| 353 | ||
| 354 | /// Supplies a base object from the repository, and resolves what | |
| 355 | /// depends on it. | |
| 356 | pub fn supply(&mut self, id: &str, kind: ObjectKind, data: Vec<u8>) { | |
| 357 | self.objects.insert(id.to_owned(), (kind, data)); | |
| 358 | self.resolve(); | |
| 359 | } | |
| 360 | ||
| 361 | /// Deltas still unresolved. | |
| 362 | pub fn unresolved(&self) -> usize { | |
| 363 | self.pending.len() | |
| 364 | } | |
| 365 | ||
| 366 | pub fn get(&self, id: &str) -> Option<(ObjectKind, &[u8])> { | |
| 367 | self.objects.get(id).map(|(kind, data)| (*kind, data.as_slice())) | |
| 368 | } | |
| 369 | ||
| 370 | pub fn contains(&self, id: &str) -> bool { | |
| 371 | self.objects.contains_key(id) | |
| 372 | } | |
| 373 | ||
| 374 | /// The commits in the pack: what the push adds. | |
| 375 | pub fn commits(&self) -> &[String] { | |
| 376 | &self.commits | |
| 377 | } | |
| 378 | ||
| 379 | pub fn commit(&self, id: &str) -> Option<CommitInfo> { | |
| 380 | match self.get(id)? { | |
| 381 | (ObjectKind::Commit, data) => Some(parse_commit(data)), | |
| 382 | _ => None, | |
| 383 | } | |
| 384 | } | |
| 385 | ||
| 386 | pub fn tree(&self, id: &str) -> Option<Vec<TreeItem>> { | |
| 387 | match self.get(id)? { | |
| 388 | (ObjectKind::Tree, data) => Some(parse_tree(data)), | |
| 389 | _ => None, | |
| 390 | } | |
| 391 | } | |
| 392 | ||
| 393 | pub fn blob(&self, id: &str) -> Option<&[u8]> { | |
| 394 | match self.get(id)? { | |
| 395 | (ObjectKind::Blob, data) => Some(data), | |
| 396 | _ => None, | |
| 397 | } | |
| 398 | } | |
| A push is checked once and side by side: its pack is read and its bases fetched once for the rules, the workflow gate and the secret scan, which run together, other services are asked while the pack is read, and cache writes and rule records finish after git has its answer | 399 | |
| 400 | /// What the pack's own commits and trees say the objects they name | |
| 401 | /// are: a commit's tree and parents, a tree's entries (submodules | |
| 402 | /// left out). How a missing base can be asked for as what it is. | |
| 403 | pub fn named_kinds(&self) -> HashMap<String, ObjectKind> { | |
| 404 | let mut kinds = HashMap::new(); | |
| 405 | for (kind, data) in self.objects.values() { | |
| 406 | match kind { | |
| 407 | ObjectKind::Commit => { | |
| 408 | let commit = parse_commit(data); | |
| 409 | kinds.insert(commit.tree, ObjectKind::Tree); | |
| 410 | kinds.extend(commit.parents.into_iter().map(|parent| (parent, ObjectKind::Commit))); | |
| 411 | } | |
| 412 | ObjectKind::Tree => { | |
| 413 | for item in parse_tree(data) { | |
| 414 | if item.is_tree() { | |
| 415 | kinds.insert(item.id, ObjectKind::Tree); | |
| 416 | } else if item.mode != "160000" { | |
| 417 | kinds.insert(item.id, ObjectKind::Blob); | |
| 418 | } | |
| 419 | } | |
| 420 | } | |
| 421 | _ => {} | |
| 422 | } | |
| 423 | } | |
| 424 | kinds | |
| 425 | } | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 426 | } |
| 427 | ||
| Invite-only launch: sign in with GitHub, repository access and lifecycle, many emails, a new look | 428 | /// What a commit says about its place in history, and whose it is. |
| 429 | #[derive(Debug, Default, PartialEq, Eq)] | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 430 | pub struct CommitInfo { |
| 431 | pub tree: String, | |
| 432 | pub parents: Vec<String>, | |
| Invite-only launch: sign in with GitHub, repository access and lifecycle, many emails, a new look | 433 | /// The address on the `author` line, as written. |
| 434 | pub author_email: Option<String>, | |
| 435 | /// The address on the `committer` line, as written. | |
| 436 | pub committer_email: Option<String>, | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 437 | } |
| 438 | ||
| Invite-only launch: sign in with GitHub, repository access and lifecycle, many emails, a new look | 439 | /// The address in a signature line's value: `Name <address> 1700000000 +0000`. |
| 440 | fn signature_email(value: &str) -> Option<String> { | |
| 441 | let start = value.rfind('<')?; | |
| 442 | let end = start + value[start..].find('>')?; | |
| 443 | Some(value[start + 1..end].trim().to_owned()) | |
| 444 | } | |
| 445 | ||
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 446 | pub fn parse_commit(data: &[u8]) -> CommitInfo { |
| 447 | let text = String::from_utf8_lossy(data); | |
| Invite-only launch: sign in with GitHub, repository access and lifecycle, many emails, a new look | 448 | let mut info = CommitInfo::default(); |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 449 | for line in text.lines() { |
| 450 | if line.is_empty() { | |
| 451 | break; | |
| 452 | } | |
| 453 | if let Some(tree) = line.strip_prefix("tree ") { | |
| 454 | info.tree = tree.trim().to_owned(); | |
| 455 | } else if let Some(parent) = line.strip_prefix("parent ") { | |
| 456 | info.parents.push(parent.trim().to_owned()); | |
| Invite-only launch: sign in with GitHub, repository access and lifecycle, many emails, a new look | 457 | } else if let Some(author) = line.strip_prefix("author ") { |
| 458 | info.author_email = signature_email(author); | |
| 459 | } else if let Some(committer) = line.strip_prefix("committer ") { | |
| 460 | info.committer_email = signature_email(committer); | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 461 | } |
| 462 | } | |
| 463 | info | |
| 464 | } | |
| 465 | ||
| 466 | /// One entry of a tree. | |
| 467 | #[derive(Clone, Debug, PartialEq, Eq)] | |
| 468 | pub struct TreeItem { | |
| 469 | /// `100644`, `100755`, `120000`, `40000` or `160000`. | |
| 470 | pub mode: String, | |
| 471 | pub name: String, | |
| 472 | pub id: String, | |
| 473 | } | |
| 474 | ||
| 475 | impl TreeItem { | |
| 476 | pub fn is_tree(&self) -> bool { | |
| 477 | self.mode == "40000" | |
| 478 | } | |
| 479 | ||
| 480 | /// A regular or executable file; not a link or a submodule. | |
| 481 | pub fn is_file(&self) -> bool { | |
| 482 | self.mode.starts_with("100") | |
| 483 | } | |
| 484 | } | |
| 485 | ||
| 486 | pub fn parse_tree(data: &[u8]) -> Vec<TreeItem> { | |
| 487 | let mut items = Vec::new(); | |
| 488 | let mut at = 0; | |
| 489 | while at < data.len() { | |
| 490 | let Some(space) = data[at..].iter().position(|byte| *byte == b' ') else { | |
| 491 | break; | |
| 492 | }; | |
| 493 | let Some(nul) = data[at + space..].iter().position(|byte| *byte == 0) else { | |
| 494 | break; | |
| 495 | }; | |
| 496 | let mode = String::from_utf8_lossy(&data[at..at + space]).into_owned(); | |
| 497 | let name = String::from_utf8_lossy(&data[at + space + 1..at + space + nul]).into_owned(); | |
| 498 | let id_at = at + space + nul + 1; | |
| 499 | let Some(id) = data.get(id_at..id_at + 20) else { | |
| 500 | break; | |
| 501 | }; | |
| 502 | items.push(TreeItem { mode, name, id: id.iter().map(|byte| format!("{byte:02x}")).collect() }); | |
| 503 | at = id_at + 20; | |
| 504 | } | |
| 505 | items | |
| 506 | } | |
| 507 | ||
| 508 | /// A tree's bytes from its entries, as git writes them: what a delta | |
| 509 | /// against a tree the repository has needs as its base. | |
| 510 | pub fn encode_tree(items: &[TreeItem]) -> Vec<u8> { | |
| 511 | let mut out = Vec::new(); | |
| 512 | for item in items { | |
| 513 | out.extend_from_slice(item.mode.as_bytes()); | |
| 514 | out.push(b' '); | |
| 515 | out.extend_from_slice(item.name.as_bytes()); | |
| 516 | out.push(0); | |
| 517 | for pair in item.id.as_bytes().chunks(2) { | |
| 518 | out.push(u8::from_str_radix(std::str::from_utf8(pair).unwrap_or("00"), 16).unwrap_or(0)); | |
| 519 | } | |
| 520 | } | |
| 521 | out | |
| 522 | } | |
| 523 | ||
| 524 | #[cfg(test)] | |
| 525 | pub(crate) mod tests { | |
| 526 | use super::*; | |
| 527 | use miniz_oxide::deflate::compress_to_vec_zlib; | |
| 528 | ||
| 529 | /// A pack of whole objects, plus ref-deltas given as (base id, delta). | |
| 530 | pub fn build_pack(objects: &[(ObjectKind, Vec<u8>)], ref_deltas: &[(String, Vec<u8>)]) -> Vec<u8> { | |
| 531 | let mut pack = b"PACK".to_vec(); | |
| 532 | pack.extend_from_slice(&2u32.to_be_bytes()); | |
| 533 | pack.extend_from_slice(&((objects.len() + ref_deltas.len()) as u32).to_be_bytes()); | |
| 534 | for (kind, data) in objects { | |
| Catching up with main takes seconds when the two sides touched different files | 535 | pack.extend(header(kind.code(), data.len())); |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 536 | pack.extend(compress_to_vec_zlib(data, 6)); |
| 537 | } | |
| 538 | for (base, delta) in ref_deltas { | |
| 539 | pack.extend(header(7, delta.len())); | |
| 540 | for pair in base.as_bytes().chunks(2) { | |
| 541 | pack.push(u8::from_str_radix(std::str::from_utf8(pair).unwrap(), 16).unwrap()); | |
| 542 | } | |
| 543 | pack.extend(compress_to_vec_zlib(delta, 6)); | |
| 544 | } | |
| 545 | pack.extend_from_slice(&[0u8; 20]); | |
| 546 | pack | |
| 547 | } | |
| 548 | ||
| 549 | /// A delta that keeps the first `keep` bytes of `base` and appends `tail`. | |
| 550 | pub fn delta(base: &[u8], keep: usize, tail: &[u8]) -> Vec<u8> { | |
| 551 | let mut out = Vec::new(); | |
| 552 | let put = |mut value: usize, out: &mut Vec<u8>| loop { | |
| 553 | let byte = (value & 0x7f) as u8; | |
| 554 | value >>= 7; | |
| 555 | if value == 0 { | |
| 556 | out.push(byte); | |
| 557 | break; | |
| 558 | } | |
| 559 | out.push(byte | 0x80); | |
| 560 | }; | |
| 561 | put(base.len(), &mut out); | |
| 562 | put(keep + tail.len(), &mut out); | |
| 563 | // Copy from offset 0, `keep` bytes (one size byte). | |
| 564 | out.push(0x80 | 0x10); | |
| 565 | out.push(keep as u8); | |
| 566 | out.push(tail.len() as u8); | |
| 567 | out.extend_from_slice(tail); | |
| 568 | out | |
| 569 | } | |
| 570 | ||
| 571 | #[test] | |
| 572 | fn ids_match_git() { | |
| 573 | assert_eq!(object_id(ObjectKind::Blob, b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391"); | |
| 574 | assert_eq!(object_id(ObjectKind::Blob, b"hello\n"), "ce013625030ba8dba906f756967f9e9ca394464a"); | |
| 575 | } | |
| 576 | ||
| 577 | #[test] | |
| 578 | fn whole_objects_and_deltas_are_read() { | |
| 579 | let base = b"first line\n".to_vec(); | |
| 580 | let base_id = object_id(ObjectKind::Blob, &base); | |
| 581 | let pack = build_pack(&[(ObjectKind::Blob, base.clone())], &[(base_id.clone(), delta(&base, base.len(), b"second\n"))]); | |
| 582 | let parsed = Pack::parse(&pack).unwrap(); | |
| 583 | let grown = b"first line\nsecond\n"; | |
| 584 | assert_eq!(parsed.blob(&object_id(ObjectKind::Blob, grown)), Some(&grown[..])); | |
| 585 | assert!(parsed.missing_bases().is_empty()); | |
| 586 | } | |
| 587 | ||
| 588 | #[test] | |
| 589 | fn a_thin_pack_waits_for_its_base() { | |
| 590 | let base = b"kept in the repository\n".to_vec(); | |
| 591 | let base_id = object_id(ObjectKind::Blob, &base); | |
| 592 | let pack = build_pack(&[], &[(base_id.clone(), delta(&base, 4, b" and more\n"))]); | |
| 593 | let mut parsed = Pack::parse(&pack).unwrap(); | |
| 594 | assert_eq!(parsed.missing_bases(), vec![base_id.clone()]); | |
| 595 | parsed.supply(&base_id, ObjectKind::Blob, base); | |
| 596 | assert_eq!(parsed.unresolved(), 0); | |
| 597 | assert!(parsed.blob(&object_id(ObjectKind::Blob, b"kept and more\n")).is_some()); | |
| 598 | } | |
| 599 | ||
| 600 | #[test] | |
| A push is checked once and side by side: its pack is read and its bases fetched once for the rules, the workflow gate and the secret scan, which run together, other services are asked while the pack is read, and cache writes and rule records finish after git has its answer | 601 | fn a_pack_says_what_the_objects_it_names_are() { |
| 602 | let blob_id = object_id(ObjectKind::Blob, b"x"); | |
| 603 | let sub_id = "1".repeat(40); | |
| 604 | let module_id = "2".repeat(40); | |
| 605 | let tree = encode_tree(&[ | |
| 606 | TreeItem { mode: "100644".into(), name: "a.txt".into(), id: blob_id.clone() }, | |
| 607 | TreeItem { mode: "40000".into(), name: "src".into(), id: sub_id.clone() }, | |
| 608 | TreeItem { mode: "160000".into(), name: "vendor".into(), id: module_id.clone() }, | |
| 609 | ]); | |
| 610 | let tree_id = object_id(ObjectKind::Tree, &tree); | |
| 611 | let parent = "3".repeat(40); | |
| 612 | let commit = format!("tree {tree_id}\nparent {parent}\nauthor A <a@x> 1 +0000\n\nm\n"); | |
| 613 | let pack = Pack::parse(&build_pack(&[(ObjectKind::Tree, tree), (ObjectKind::Commit, commit.into_bytes())], &[])).unwrap(); | |
| 614 | let kinds = pack.named_kinds(); | |
| 615 | assert_eq!(kinds.get(&blob_id), Some(&ObjectKind::Blob)); | |
| 616 | assert_eq!(kinds.get(&sub_id), Some(&ObjectKind::Tree)); | |
| 617 | assert_eq!(kinds.get(&tree_id), Some(&ObjectKind::Tree)); | |
| 618 | assert_eq!(kinds.get(&parent), Some(&ObjectKind::Commit)); | |
| 619 | // A submodule's commit lives in another repository. | |
| 620 | assert_eq!(kinds.get(&module_id), None); | |
| 621 | } | |
| 622 | ||
| 623 | #[test] | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 624 | fn commits_and_trees_are_parsed_and_trees_rebuilt() { |
| 625 | let blob_id = object_id(ObjectKind::Blob, b"x"); | |
| 626 | let tree = encode_tree(&[ | |
| 627 | TreeItem { mode: "100644".into(), name: "a.txt".into(), id: blob_id.clone() }, | |
| 628 | TreeItem { mode: "40000".into(), name: "src".into(), id: blob_id.clone() }, | |
| 629 | ]); | |
| 630 | let items = parse_tree(&tree); | |
| 631 | assert_eq!(items.len(), 2); | |
| 632 | assert!(items[0].is_file() && items[1].is_tree()); | |
| 633 | assert_eq!(encode_tree(&items), tree); | |
| 634 | let commit = format!("tree {}\nparent aaaa\nparent bbbb\nauthor x\n\nmessage\nparent no\n", object_id(ObjectKind::Tree, &tree)); | |
| 635 | let info = parse_commit(commit.as_bytes()); | |
| 636 | assert_eq!(info.parents, ["aaaa", "bbbb"]); | |
| 637 | assert_eq!(info.tree.len(), 40); | |
| Invite-only launch: sign in with GitHub, repository access and lifecycle, many emails, a new look | 638 | assert_eq!(info.author_email, None); |
| 639 | let signed = parse_commit( | |
| 640 | b"tree t | |
| 641 | author Ada L <Ada@Example.com> 1700000000 +0000 | |
| 642 | committer Bot <bot@x.io> 1700000000 +0000 | |
| 643 | ||
| 644 | author <no@x.io> | |
| 645 | ", | |
| 646 | ); | |
| 647 | assert_eq!(signed.author_email.as_deref(), Some("Ada@Example.com")); | |
| 648 | assert_eq!(signed.committer_email.as_deref(), Some("bot@x.io")); | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 649 | let pack = build_pack(&[(ObjectKind::Commit, commit.into_bytes())], &[]); |
| 650 | assert_eq!(Pack::parse(&pack).unwrap().commits().len(), 1); | |
| 651 | } | |
| 652 | ||
| 653 | #[test] | |
| 654 | fn the_pack_is_found_after_the_commands() { | |
| 655 | let line = b"old new refs/heads/PACKAGING\0report-status\n"; | |
| 656 | let commands = [format!("{:04x}", line.len() + 4).into_bytes(), line.to_vec(), b"0000".to_vec()].concat(); | |
| 657 | let body = [commands.clone(), b"PACK\0\0\0\x02".to_vec()].concat(); | |
| 658 | assert_eq!(pack_start(&body), Some(commands.len())); | |
| 659 | assert_eq!(pack_start(&commands), None); | |
| 660 | assert!(Pack::parse(b"nope").is_err()); | |
| 661 | } | |
| Catching up with main takes seconds when the two sides touched different files | 662 | |
| 663 | #[test] | |
| 664 | fn written_packs_are_sealed_and_extend() { | |
| 665 | let blob = b"hello | |
| 666 | ".to_vec(); | |
| 667 | let pack = write_pack(&[(ObjectKind::Blob, blob.clone())]); | |
| 668 | let (body, trailer) = pack.split_at(pack.len() - 20); | |
| 669 | assert_eq!(Sha1::digest(body).as_slice(), trailer); | |
| 670 | let more = extend_pack(&pack, &[(ObjectKind::Blob, b"more | |
| 671 | ".to_vec())]).unwrap(); | |
| 672 | assert_eq!(&more[8..12], &2u32.to_be_bytes()); | |
| 673 | let read = Pack::parse(&more).unwrap(); | |
| 674 | assert!(read.blob("ce013625030ba8dba906f756967f9e9ca394464a").is_some()); | |
| 675 | assert!(read.blob(&object_id(ObjectKind::Blob, b"more | |
| 676 | ")).is_some()); | |
| 677 | let mut broken = pack.clone(); | |
| 678 | broken[12] ^= 1; | |
| 679 | assert!(extend_pack(&broken, &[]).is_err()); | |
| 680 | } | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 681 | } |