Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 1 | //! Reading the objects in a git pack, as a push sends them, so that what a |
| 2 | //! push adds can be looked at before it is stored. | |
| 3 | //! | |
| 4 | //! A pushed pack is usually thin: some objects are deltas against objects | |
| 5 | //! the repository already has. Those are left pending until the caller | |
| 6 | //! supplies their bases with [`Pack::supply`]. | |
| Catching up with main takes seconds when the two sides touched different files | 7 | //! |
| 8 | //! Writing is the small part g1t needs for a merge it makes itself: a pack | |
| 9 | //! of whole objects ([`write_pack`]), or one fetched from elsewhere with a | |
| 10 | //! few objects added ([`extend_pack`]). | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 11 | |
| 12 | use std::collections::HashMap; | |
| 13 | ||
| 14 | use miniz_oxide::inflate::TINFLStatus; | |
| 15 | use miniz_oxide::inflate::core::{DecompressorOxide, decompress, inflate_flags}; | |
| 16 | use sha1::{Digest, Sha1}; | |
| 17 | ||
| 18 | /// Beyond this much inflated content the pack is not read: a push that | |
| 19 | /// large is let through unread rather than risk the worker's memory. | |
| 20 | pub const MAX_INFLATED: usize = 48 * 1024 * 1024; | |
| 21 | ||
| 22 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] | |
| 23 | pub enum ObjectKind { | |
| 24 | Commit, | |
| 25 | Tree, | |
| 26 | Blob, | |
| 27 | Tag, | |
| 28 | } | |
| 29 | ||
| 30 | impl ObjectKind { | |
| 31 | fn from_type(code: u8) -> Option<ObjectKind> { | |
| 32 | Some(match code { | |
| 33 | 1 => ObjectKind::Commit, | |
| 34 | 2 => ObjectKind::Tree, | |
| 35 | 3 => ObjectKind::Blob, | |
| 36 | 4 => ObjectKind::Tag, | |
| 37 | _ => return None, | |
| 38 | }) | |
| 39 | } | |
| 40 | ||
| Catching up with main takes seconds when the two sides touched different files | 41 | /// The type code a pack gives objects of this kind. |
| 42 | fn code(self) -> u8 { | |
| 43 | match self { | |
| 44 | ObjectKind::Commit => 1, | |
| 45 | ObjectKind::Tree => 2, | |
| 46 | ObjectKind::Blob => 3, | |
| 47 | ObjectKind::Tag => 4, | |
| 48 | } | |
| 49 | } | |
| 50 | ||
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 51 | fn name(self) -> &'static str { |
| 52 | match self { | |
| 53 | ObjectKind::Commit => "commit", | |
| 54 | ObjectKind::Tree => "tree", | |
| 55 | ObjectKind::Blob => "blob", | |
| 56 | ObjectKind::Tag => "tag", | |
| 57 | } | |
| 58 | } | |
| 59 | } | |
| 60 | ||
| 61 | /// A git object's id: the SHA-1 of its header and content, in hex. | |
| 62 | pub fn object_id(kind: ObjectKind, data: &[u8]) -> String { | |
| 63 | let mut hasher = Sha1::new(); | |
| 64 | hasher.update(format!("{} {}\0", kind.name(), data.len()).as_bytes()); | |
| 65 | hasher.update(data); | |
| 66 | hasher.finalize().iter().map(|byte| format!("{byte:02x}")).collect() | |
| 67 | } | |
| 68 | ||
| 69 | enum Base { | |
| 70 | /// An earlier object in the pack, by its offset. | |
| 71 | Offset(usize), | |
| 72 | /// Any object, by id. | |
| 73 | Id(String), | |
| 74 | } | |
| 75 | ||
| 76 | struct Delta { | |
| 77 | base: Base, | |
| 78 | data: Vec<u8>, | |
| 79 | } | |
| 80 | ||
| 81 | /// The objects of a pack, by id. | |
| 82 | #[derive(Default)] | |
| 83 | pub struct Pack { | |
| 84 | objects: HashMap<String, (ObjectKind, Vec<u8>)>, | |
| 85 | /// Ids of the objects at each offset, once resolved. | |
| 86 | at_offset: HashMap<usize, String>, | |
| 87 | pending: Vec<(usize, Delta)>, | |
| 88 | /// Commits, in the order the pack holds them. | |
| 89 | commits: Vec<String>, | |
| 90 | } | |
| 91 | ||
| 92 | /// Where the pack starts in a receive-pack request: after the commands | |
| 93 | /// and anything else sent as pkt-lines. | |
| 94 | pub fn pack_start(body: &[u8]) -> Option<usize> { | |
| 95 | let mut at = 0; | |
| 96 | loop { | |
| 97 | if body.get(at..at + 4) == Some(b"PACK") { | |
| 98 | return Some(at); | |
| 99 | } | |
| 100 | let length = std::str::from_utf8(body.get(at..at + 4)?) | |
| 101 | .ok() | |
| 102 | .and_then(|hex| usize::from_str_radix(hex, 16).ok())?; | |
| 103 | // A flush packet is four bytes; any other line counts its own length. | |
| 104 | at += if length == 0 { 4 } else { length.max(4) }; | |
| 105 | } | |
| 106 | } | |
| 107 | ||
| 108 | fn inflate(input: &[u8], size: usize) -> Result<(Vec<u8>, usize), String> { | |
| 109 | let mut out = vec![0u8; size.max(1)]; | |
| 110 | let mut state = DecompressorOxide::new(); | |
| 111 | let flags = inflate_flags::TINFL_FLAG_PARSE_ZLIB_HEADER | inflate_flags::TINFL_FLAG_USING_NON_WRAPPING_OUTPUT_BUF; | |
| 112 | let (status, consumed, written) = decompress(&mut state, input, &mut out, 0, flags); | |
| 113 | match status { | |
| 114 | TINFLStatus::Done => { | |
| 115 | out.truncate(written); | |
| 116 | if written != size { | |
| 117 | return Err(format!("an object inflated to {written} bytes, not {size}")); | |
| 118 | } | |
| 119 | Ok((out, consumed)) | |
| 120 | } | |
| 121 | other => Err(format!("an object could not be inflated: {other:?}")), | |
| 122 | } | |
| 123 | } | |
| 124 | ||
| 125 | fn varint(data: &[u8], at: &mut usize) -> Option<usize> { | |
| 126 | let (mut value, mut shift) = (0usize, 0); | |
| 127 | loop { | |
| 128 | let byte = *data.get(*at)?; | |
| 129 | *at += 1; | |
| 130 | value |= ((byte & 0x7f) as usize) << shift; | |
| 131 | shift += 7; | |
| 132 | if byte & 0x80 == 0 || shift > 56 { | |
| 133 | return Some(value); | |
| 134 | } | |
| 135 | } | |
| 136 | } | |
| 137 | ||
| Catching up with main takes seconds when the two sides touched different files | 138 | /// An entry's header in a pack: its type and its inflated size. |
| 139 | fn header(code: u8, size: usize) -> Vec<u8> { | |
| 140 | let mut out = Vec::new(); | |
| 141 | let mut byte = (code << 4) | (size & 15) as u8; | |
| 142 | let mut rest = size >> 4; | |
| 143 | while rest > 0 { | |
| 144 | out.push(byte | 0x80); | |
| 145 | byte = (rest & 0x7f) as u8; | |
| 146 | rest >>= 7; | |
| 147 | } | |
| 148 | out.push(byte); | |
| 149 | out | |
| 150 | } | |
| 151 | ||
| 152 | fn write_entries(pack: &mut Vec<u8>, objects: &[(ObjectKind, Vec<u8>)]) { | |
| 153 | for (kind, data) in objects { | |
| 154 | pack.extend(header(kind.code(), data.len())); | |
| 155 | pack.extend(miniz_oxide::deflate::compress_to_vec_zlib(data, 6)); | |
| 156 | } | |
| 157 | } | |
| 158 | ||
| 159 | fn seal(mut pack: Vec<u8>) -> Vec<u8> { | |
| 160 | let checksum = Sha1::digest(&pack); | |
| 161 | pack.extend_from_slice(&checksum); | |
| 162 | pack | |
| 163 | } | |
| 164 | ||
| 165 | /// A version 2 pack holding `objects` whole, with no deltas: what git's | |
| 166 | /// receive-pack takes, when the objects are few and new. | |
| 167 | pub fn write_pack(objects: &[(ObjectKind, Vec<u8>)]) -> Vec<u8> { | |
| 168 | let mut pack = b"PACK".to_vec(); | |
| 169 | pack.extend_from_slice(&2u32.to_be_bytes()); | |
| 170 | pack.extend_from_slice(&(objects.len() as u32).to_be_bytes()); | |
| 171 | write_entries(&mut pack, objects); | |
| 172 | seal(pack) | |
| 173 | } | |
| 174 | ||
| 175 | /// `pack` with `objects` added after its own, as one pack. Its entries keep | |
| 176 | /// their offsets, since the header stays the same length, so its deltas | |
| 177 | /// still find their bases. `pack` must not be thin. | |
| 178 | pub fn extend_pack(pack: &[u8], objects: &[(ObjectKind, Vec<u8>)]) -> Result<Vec<u8>, String> { | |
| 179 | if pack.len() < 32 || &pack[..4] != b"PACK" { | |
| 180 | return Err("not a pack".into()); | |
| 181 | } | |
| 182 | let (body, trailer) = pack.split_at(pack.len() - 20); | |
| 183 | if Sha1::digest(body).as_slice() != trailer { | |
| 184 | return Err("the pack's checksum does not match".into()); | |
| 185 | } | |
| 186 | let count = u32::from_be_bytes([pack[8], pack[9], pack[10], pack[11]]) as usize; | |
| 187 | let total = u32::try_from(count + objects.len()).map_err(|_| "the pack is too large")?; | |
| 188 | let mut out = body.to_vec(); | |
| 189 | out[8..12].copy_from_slice(&total.to_be_bytes()); | |
| 190 | write_entries(&mut out, objects); | |
| 191 | Ok(seal(out)) | |
| 192 | } | |
| 193 | ||
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 194 | /// Applies a git delta to its base. |
| 195 | pub fn apply_delta(base: &[u8], delta: &[u8]) -> Result<Vec<u8>, String> { | |
| 196 | let mut at = 0; | |
| 197 | let bad = || "a delta is malformed".to_owned(); | |
| 198 | let source = varint(delta, &mut at).ok_or_else(bad)?; | |
| 199 | if source != base.len() { | |
| 200 | return Err("a delta does not fit its base".into()); | |
| 201 | } | |
| 202 | let target = varint(delta, &mut at).ok_or_else(bad)?; | |
| 203 | let mut out = Vec::with_capacity(target); | |
| 204 | while at < delta.len() { | |
| 205 | let op = delta[at]; | |
| 206 | at += 1; | |
| 207 | if op & 0x80 != 0 { | |
| 208 | let mut offset = 0usize; | |
| 209 | let mut size = 0usize; | |
| 210 | for bit in 0..4 { | |
| 211 | if op & (1 << bit) != 0 { | |
| 212 | offset |= (*delta.get(at).ok_or_else(bad)? as usize) << (8 * bit); | |
| 213 | at += 1; | |
| 214 | } | |
| 215 | } | |
| 216 | for bit in 0..3 { | |
| 217 | if op & (0x10 << bit) != 0 { | |
| 218 | size |= (*delta.get(at).ok_or_else(bad)? as usize) << (8 * bit); | |
| 219 | at += 1; | |
| 220 | } | |
| 221 | } | |
| 222 | if size == 0 { | |
| 223 | size = 0x10000; | |
| 224 | } | |
| 225 | out.extend_from_slice(base.get(offset..offset + size).ok_or_else(bad)?); | |
| 226 | } else if op != 0 { | |
| 227 | out.extend_from_slice(delta.get(at..at + op as usize).ok_or_else(bad)?); | |
| 228 | at += op as usize; | |
| 229 | } else { | |
| 230 | return Err(bad()); | |
| 231 | } | |
| 232 | } | |
| 233 | if out.len() != target { | |
| 234 | return Err(bad()); | |
| 235 | } | |
| 236 | Ok(out) | |
| 237 | } | |
| 238 | ||
| 239 | impl Pack { | |
| 240 | /// Reads every object in `pack`, resolving the deltas whose bases are | |
| 241 | /// in it. | |
| 242 | pub fn parse(pack: &[u8]) -> Result<Pack, String> { | |
| 243 | if pack.len() < 12 || &pack[..4] != b"PACK" { | |
| 244 | return Err("not a pack".into()); | |
| 245 | } | |
| 246 | let count = u32::from_be_bytes([pack[8], pack[9], pack[10], pack[11]]) as usize; | |
| 247 | let mut at = 12; | |
| 248 | let mut result = Pack::default(); | |
| 249 | let mut inflated = 0usize; | |
| 250 | for _ in 0..count { | |
| 251 | let start = at; | |
| 252 | let mut byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 253 | at += 1; | |
| 254 | let code = (byte >> 4) & 7; | |
| 255 | let mut size = (byte & 15) as usize; | |
| 256 | let mut shift = 4; | |
| 257 | while byte & 0x80 != 0 { | |
| 258 | byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 259 | at += 1; | |
| 260 | size |= ((byte & 0x7f) as usize) << shift; | |
| 261 | shift += 7; | |
| 262 | } | |
| 263 | inflated += size; | |
| 264 | if inflated > MAX_INFLATED { | |
| 265 | return Err("the pack is too large to read".into()); | |
| 266 | } | |
| 267 | let base = match code { | |
| 268 | 6 => { | |
| 269 | let mut byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 270 | at += 1; | |
| 271 | let mut offset = (byte & 0x7f) as usize; | |
| 272 | while byte & 0x80 != 0 { | |
| 273 | byte = *pack.get(at).ok_or("the pack ends early")?; | |
| 274 | at += 1; | |
| 275 | offset = ((offset + 1) << 7) | (byte & 0x7f) as usize; | |
| 276 | } | |
| 277 | Some(Base::Offset(start.checked_sub(offset).ok_or("a delta points before the pack")?)) | |
| 278 | } | |
| 279 | 7 => { | |
| 280 | let id = pack.get(at..at + 20).ok_or("the pack ends early")?; | |
| 281 | at += 20; | |
| 282 | Some(Base::Id(id.iter().map(|byte| format!("{byte:02x}")).collect())) | |
| 283 | } | |
| 284 | _ => None, | |
| 285 | }; | |
| 286 | let (data, consumed) = inflate(&pack[at..], size)?; | |
| 287 | at += consumed; | |
| 288 | match base { | |
| 289 | Some(base) => result.pending.push((start, Delta { base, data })), | |
| 290 | None => { | |
| 291 | let kind = ObjectKind::from_type(code).ok_or("an object has an unknown type")?; | |
| 292 | result.insert(start, kind, data); | |
| 293 | } | |
| 294 | } | |
| 295 | } | |
| 296 | result.resolve(); | |
| 297 | Ok(result) | |
| 298 | } | |
| 299 | ||
| 300 | fn insert(&mut self, offset: usize, kind: ObjectKind, data: Vec<u8>) { | |
| 301 | let id = object_id(kind, &data); | |
| 302 | if kind == ObjectKind::Commit { | |
| 303 | self.commits.push(id.clone()); | |
| 304 | } | |
| 305 | self.at_offset.insert(offset, id.clone()); | |
| 306 | self.objects.insert(id, (kind, data)); | |
| 307 | } | |
| 308 | ||
| 309 | /// Resolves every pending delta whose base is known by now. | |
| 310 | fn resolve(&mut self) { | |
| 311 | loop { | |
| 312 | let mut progress = false; | |
| 313 | let pending = std::mem::take(&mut self.pending); | |
| 314 | for (offset, delta) in pending { | |
| 315 | let base_id = match &delta.base { | |
| 316 | Base::Offset(base) => self.at_offset.get(base).cloned(), | |
| 317 | Base::Id(id) => Some(id.clone()), | |
| 318 | }; | |
| 319 | let resolved = base_id | |
| 320 | .and_then(|id| self.objects.get(&id)) | |
| 321 | .map(|(kind, base)| (*kind, apply_delta(base, &delta.data))); | |
| 322 | match resolved { | |
| 323 | Some((kind, Ok(data))) => { | |
| 324 | self.insert(offset, kind, data); | |
| 325 | progress = true; | |
| 326 | } | |
| 327 | // A delta that does not apply is dropped. | |
| 328 | Some((_, Err(_))) => progress = true, | |
| 329 | None => self.pending.push((offset, delta)), | |
| 330 | } | |
| 331 | } | |
| 332 | if !progress || self.pending.is_empty() { | |
| 333 | return; | |
| 334 | } | |
| 335 | } | |
| 336 | } | |
| 337 | ||
| 338 | /// Objects the pack's deltas are based on that it does not hold: what | |
| 339 | /// the repository has to supply. | |
| 340 | pub fn missing_bases(&self) -> Vec<String> { | |
| 341 | let mut ids: Vec<String> = self | |
| 342 | .pending | |
| 343 | .iter() | |
| 344 | .filter_map(|(_, delta)| match &delta.base { | |
| 345 | Base::Id(id) if !self.objects.contains_key(id) => Some(id.clone()), | |
| 346 | _ => None, | |
| 347 | }) | |
| 348 | .collect(); | |
| 349 | ids.sort(); | |
| 350 | ids.dedup(); | |
| 351 | ids | |
| 352 | } | |
| 353 | ||
| 354 | /// Supplies a base object from the repository, and resolves what | |
| 355 | /// depends on it. | |
| 356 | pub fn supply(&mut self, id: &str, kind: ObjectKind, data: Vec<u8>) { | |
| 357 | self.objects.insert(id.to_owned(), (kind, data)); | |
| 358 | self.resolve(); | |
| 359 | } | |
| 360 | ||
| 361 | /// Deltas still unresolved. | |
| 362 | pub fn unresolved(&self) -> usize { | |
| 363 | self.pending.len() | |
| 364 | } | |
| 365 | ||
| 366 | pub fn get(&self, id: &str) -> Option<(ObjectKind, &[u8])> { | |
| 367 | self.objects.get(id).map(|(kind, data)| (*kind, data.as_slice())) | |
| 368 | } | |
| 369 | ||
| 370 | pub fn contains(&self, id: &str) -> bool { | |
| 371 | self.objects.contains_key(id) | |
| 372 | } | |
| 373 | ||
| 374 | /// The commits in the pack: what the push adds. | |
| 375 | pub fn commits(&self) -> &[String] { | |
| 376 | &self.commits | |
| 377 | } | |
| 378 | ||
| 379 | pub fn commit(&self, id: &str) -> Option<CommitInfo> { | |
| 380 | match self.get(id)? { | |
| 381 | (ObjectKind::Commit, data) => Some(parse_commit(data)), | |
| 382 | _ => None, | |
| 383 | } | |
| 384 | } | |
| 385 | ||
| 386 | pub fn tree(&self, id: &str) -> Option<Vec<TreeItem>> { | |
| 387 | match self.get(id)? { | |
| 388 | (ObjectKind::Tree, data) => Some(parse_tree(data)), | |
| 389 | _ => None, | |
| 390 | } | |
| 391 | } | |
| 392 | ||
| 393 | pub fn blob(&self, id: &str) -> Option<&[u8]> { | |
| 394 | match self.get(id)? { | |
| 395 | (ObjectKind::Blob, data) => Some(data), | |
| 396 | _ => None, | |
| 397 | } | |
| 398 | } | |
| 399 | } | |
| 400 | ||
| 401 | /// What a commit says about its place in history. | |
| 402 | #[derive(Debug, PartialEq, Eq)] | |
| 403 | pub struct CommitInfo { | |
| 404 | pub tree: String, | |
| 405 | pub parents: Vec<String>, | |
| 406 | } | |
| 407 | ||
| 408 | pub fn parse_commit(data: &[u8]) -> CommitInfo { | |
| 409 | let text = String::from_utf8_lossy(data); | |
| 410 | let mut info = CommitInfo { tree: String::new(), parents: Vec::new() }; | |
| 411 | for line in text.lines() { | |
| 412 | if line.is_empty() { | |
| 413 | break; | |
| 414 | } | |
| 415 | if let Some(tree) = line.strip_prefix("tree ") { | |
| 416 | info.tree = tree.trim().to_owned(); | |
| 417 | } else if let Some(parent) = line.strip_prefix("parent ") { | |
| 418 | info.parents.push(parent.trim().to_owned()); | |
| 419 | } | |
| 420 | } | |
| 421 | info | |
| 422 | } | |
| 423 | ||
| 424 | /// One entry of a tree. | |
| 425 | #[derive(Clone, Debug, PartialEq, Eq)] | |
| 426 | pub struct TreeItem { | |
| 427 | /// `100644`, `100755`, `120000`, `40000` or `160000`. | |
| 428 | pub mode: String, | |
| 429 | pub name: String, | |
| 430 | pub id: String, | |
| 431 | } | |
| 432 | ||
| 433 | impl TreeItem { | |
| 434 | pub fn is_tree(&self) -> bool { | |
| 435 | self.mode == "40000" | |
| 436 | } | |
| 437 | ||
| 438 | /// A regular or executable file; not a link or a submodule. | |
| 439 | pub fn is_file(&self) -> bool { | |
| 440 | self.mode.starts_with("100") | |
| 441 | } | |
| 442 | } | |
| 443 | ||
| 444 | pub fn parse_tree(data: &[u8]) -> Vec<TreeItem> { | |
| 445 | let mut items = Vec::new(); | |
| 446 | let mut at = 0; | |
| 447 | while at < data.len() { | |
| 448 | let Some(space) = data[at..].iter().position(|byte| *byte == b' ') else { | |
| 449 | break; | |
| 450 | }; | |
| 451 | let Some(nul) = data[at + space..].iter().position(|byte| *byte == 0) else { | |
| 452 | break; | |
| 453 | }; | |
| 454 | let mode = String::from_utf8_lossy(&data[at..at + space]).into_owned(); | |
| 455 | let name = String::from_utf8_lossy(&data[at + space + 1..at + space + nul]).into_owned(); | |
| 456 | let id_at = at + space + nul + 1; | |
| 457 | let Some(id) = data.get(id_at..id_at + 20) else { | |
| 458 | break; | |
| 459 | }; | |
| 460 | items.push(TreeItem { mode, name, id: id.iter().map(|byte| format!("{byte:02x}")).collect() }); | |
| 461 | at = id_at + 20; | |
| 462 | } | |
| 463 | items | |
| 464 | } | |
| 465 | ||
| 466 | /// A tree's bytes from its entries, as git writes them: what a delta | |
| 467 | /// against a tree the repository has needs as its base. | |
| 468 | pub fn encode_tree(items: &[TreeItem]) -> Vec<u8> { | |
| 469 | let mut out = Vec::new(); | |
| 470 | for item in items { | |
| 471 | out.extend_from_slice(item.mode.as_bytes()); | |
| 472 | out.push(b' '); | |
| 473 | out.extend_from_slice(item.name.as_bytes()); | |
| 474 | out.push(0); | |
| 475 | for pair in item.id.as_bytes().chunks(2) { | |
| 476 | out.push(u8::from_str_radix(std::str::from_utf8(pair).unwrap_or("00"), 16).unwrap_or(0)); | |
| 477 | } | |
| 478 | } | |
| 479 | out | |
| 480 | } | |
| 481 | ||
| 482 | #[cfg(test)] | |
| 483 | pub(crate) mod tests { | |
| 484 | use super::*; | |
| 485 | use miniz_oxide::deflate::compress_to_vec_zlib; | |
| 486 | ||
| 487 | /// A pack of whole objects, plus ref-deltas given as (base id, delta). | |
| 488 | pub fn build_pack(objects: &[(ObjectKind, Vec<u8>)], ref_deltas: &[(String, Vec<u8>)]) -> Vec<u8> { | |
| 489 | let mut pack = b"PACK".to_vec(); | |
| 490 | pack.extend_from_slice(&2u32.to_be_bytes()); | |
| 491 | pack.extend_from_slice(&((objects.len() + ref_deltas.len()) as u32).to_be_bytes()); | |
| 492 | for (kind, data) in objects { | |
| Catching up with main takes seconds when the two sides touched different files | 493 | pack.extend(header(kind.code(), data.len())); |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 494 | pack.extend(compress_to_vec_zlib(data, 6)); |
| 495 | } | |
| 496 | for (base, delta) in ref_deltas { | |
| 497 | pack.extend(header(7, delta.len())); | |
| 498 | for pair in base.as_bytes().chunks(2) { | |
| 499 | pack.push(u8::from_str_radix(std::str::from_utf8(pair).unwrap(), 16).unwrap()); | |
| 500 | } | |
| 501 | pack.extend(compress_to_vec_zlib(delta, 6)); | |
| 502 | } | |
| 503 | pack.extend_from_slice(&[0u8; 20]); | |
| 504 | pack | |
| 505 | } | |
| 506 | ||
| 507 | /// A delta that keeps the first `keep` bytes of `base` and appends `tail`. | |
| 508 | pub fn delta(base: &[u8], keep: usize, tail: &[u8]) -> Vec<u8> { | |
| 509 | let mut out = Vec::new(); | |
| 510 | let put = |mut value: usize, out: &mut Vec<u8>| loop { | |
| 511 | let byte = (value & 0x7f) as u8; | |
| 512 | value >>= 7; | |
| 513 | if value == 0 { | |
| 514 | out.push(byte); | |
| 515 | break; | |
| 516 | } | |
| 517 | out.push(byte | 0x80); | |
| 518 | }; | |
| 519 | put(base.len(), &mut out); | |
| 520 | put(keep + tail.len(), &mut out); | |
| 521 | // Copy from offset 0, `keep` bytes (one size byte). | |
| 522 | out.push(0x80 | 0x10); | |
| 523 | out.push(keep as u8); | |
| 524 | out.push(tail.len() as u8); | |
| 525 | out.extend_from_slice(tail); | |
| 526 | out | |
| 527 | } | |
| 528 | ||
| 529 | #[test] | |
| 530 | fn ids_match_git() { | |
| 531 | assert_eq!(object_id(ObjectKind::Blob, b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391"); | |
| 532 | assert_eq!(object_id(ObjectKind::Blob, b"hello\n"), "ce013625030ba8dba906f756967f9e9ca394464a"); | |
| 533 | } | |
| 534 | ||
| 535 | #[test] | |
| 536 | fn whole_objects_and_deltas_are_read() { | |
| 537 | let base = b"first line\n".to_vec(); | |
| 538 | let base_id = object_id(ObjectKind::Blob, &base); | |
| 539 | let pack = build_pack(&[(ObjectKind::Blob, base.clone())], &[(base_id.clone(), delta(&base, base.len(), b"second\n"))]); | |
| 540 | let parsed = Pack::parse(&pack).unwrap(); | |
| 541 | let grown = b"first line\nsecond\n"; | |
| 542 | assert_eq!(parsed.blob(&object_id(ObjectKind::Blob, grown)), Some(&grown[..])); | |
| 543 | assert!(parsed.missing_bases().is_empty()); | |
| 544 | } | |
| 545 | ||
| 546 | #[test] | |
| 547 | fn a_thin_pack_waits_for_its_base() { | |
| 548 | let base = b"kept in the repository\n".to_vec(); | |
| 549 | let base_id = object_id(ObjectKind::Blob, &base); | |
| 550 | let pack = build_pack(&[], &[(base_id.clone(), delta(&base, 4, b" and more\n"))]); | |
| 551 | let mut parsed = Pack::parse(&pack).unwrap(); | |
| 552 | assert_eq!(parsed.missing_bases(), vec![base_id.clone()]); | |
| 553 | parsed.supply(&base_id, ObjectKind::Blob, base); | |
| 554 | assert_eq!(parsed.unresolved(), 0); | |
| 555 | assert!(parsed.blob(&object_id(ObjectKind::Blob, b"kept and more\n")).is_some()); | |
| 556 | } | |
| 557 | ||
| 558 | #[test] | |
| 559 | fn commits_and_trees_are_parsed_and_trees_rebuilt() { | |
| 560 | let blob_id = object_id(ObjectKind::Blob, b"x"); | |
| 561 | let tree = encode_tree(&[ | |
| 562 | TreeItem { mode: "100644".into(), name: "a.txt".into(), id: blob_id.clone() }, | |
| 563 | TreeItem { mode: "40000".into(), name: "src".into(), id: blob_id.clone() }, | |
| 564 | ]); | |
| 565 | let items = parse_tree(&tree); | |
| 566 | assert_eq!(items.len(), 2); | |
| 567 | assert!(items[0].is_file() && items[1].is_tree()); | |
| 568 | assert_eq!(encode_tree(&items), tree); | |
| 569 | let commit = format!("tree {}\nparent aaaa\nparent bbbb\nauthor x\n\nmessage\nparent no\n", object_id(ObjectKind::Tree, &tree)); | |
| 570 | let info = parse_commit(commit.as_bytes()); | |
| 571 | assert_eq!(info.parents, ["aaaa", "bbbb"]); | |
| 572 | assert_eq!(info.tree.len(), 40); | |
| 573 | let pack = build_pack(&[(ObjectKind::Commit, commit.into_bytes())], &[]); | |
| 574 | assert_eq!(Pack::parse(&pack).unwrap().commits().len(), 1); | |
| 575 | } | |
| 576 | ||
| 577 | #[test] | |
| 578 | fn the_pack_is_found_after_the_commands() { | |
| 579 | let line = b"old new refs/heads/PACKAGING\0report-status\n"; | |
| 580 | let commands = [format!("{:04x}", line.len() + 4).into_bytes(), line.to_vec(), b"0000".to_vec()].concat(); | |
| 581 | let body = [commands.clone(), b"PACK\0\0\0\x02".to_vec()].concat(); | |
| 582 | assert_eq!(pack_start(&body), Some(commands.len())); | |
| 583 | assert_eq!(pack_start(&commands), None); | |
| 584 | assert!(Pack::parse(b"nope").is_err()); | |
| 585 | } | |
| Catching up with main takes seconds when the two sides touched different files | 586 | |
| 587 | #[test] | |
| 588 | fn written_packs_are_sealed_and_extend() { | |
| 589 | let blob = b"hello | |
| 590 | ".to_vec(); | |
| 591 | let pack = write_pack(&[(ObjectKind::Blob, blob.clone())]); | |
| 592 | let (body, trailer) = pack.split_at(pack.len() - 20); | |
| 593 | assert_eq!(Sha1::digest(body).as_slice(), trailer); | |
| 594 | let more = extend_pack(&pack, &[(ObjectKind::Blob, b"more | |
| 595 | ".to_vec())]).unwrap(); | |
| 596 | assert_eq!(&more[8..12], &2u32.to_be_bytes()); | |
| 597 | let read = Pack::parse(&more).unwrap(); | |
| 598 | assert!(read.blob("ce013625030ba8dba906f756967f9e9ca394464a").is_some()); | |
| 599 | assert!(read.blob(&object_id(ObjectKind::Blob, b"more | |
| 600 | ")).is_some()); | |
| 601 | let mut broken = pack.clone(); | |
| 602 | broken[12] ^= 1; | |
| 603 | assert!(extend_pack(&broken, &[]).is_err()); | |
| 604 | } | |
| Agents get guardrails, run credentials, an audit log, a context hub, repository instructions and mentions; security upkeep; snake_case API | 605 | } |