| 1 | //! Reading the objects in a git pack, as a push sends them, so that what a |
| 2 | //! push adds can be looked at before it is stored. |
| 3 | //! |
| 4 | //! A pushed pack is usually thin: some objects are deltas against objects |
| 5 | //! the repository already has. Those are left pending until the caller |
| 6 | //! supplies their bases with [`Pack::supply`]. |
| 7 | //! |
| 8 | //! Writing is the small part g1t needs for a merge it makes itself: a pack |
| 9 | //! of whole objects ([`write_pack`]), or one fetched from elsewhere with a |
| 10 | //! few objects added ([`extend_pack`]). |
| 11 | |
| 12 | use std::collections::HashMap; |
| 13 | |
| 14 | use miniz_oxide::inflate::TINFLStatus; |
| 15 | use miniz_oxide::inflate::core::{DecompressorOxide, decompress, inflate_flags}; |
| 16 | use sha1::{Digest, Sha1}; |
| 17 | |
| 18 | /// Beyond this much inflated content the pack is not read: a push that |
| 19 | /// large is let through unread rather than risk the worker's memory. |
| 20 | pub const MAX_INFLATED: usize = 48 * 1024 * 1024; |
| 21 | |
| 22 | #[derive(Clone, Copy, Debug, PartialEq, Eq)] |
| 23 | pub enum ObjectKind { |
| 24 | Commit, |
| 25 | Tree, |
| 26 | Blob, |
| 27 | Tag, |
| 28 | } |
| 29 | |
| 30 | impl ObjectKind { |
| 31 | fn from_type(code: u8) -> Option<ObjectKind> { |
| 32 | Some(match code { |
| 33 | 1 => ObjectKind::Commit, |
| 34 | 2 => ObjectKind::Tree, |
| 35 | 3 => ObjectKind::Blob, |
| 36 | 4 => ObjectKind::Tag, |
| 37 | _ => return None, |
| 38 | }) |
| 39 | } |
| 40 | |
| 41 | /// The type code a pack gives objects of this kind. |
| 42 | fn code(self) -> u8 { |
| 43 | match self { |
| 44 | ObjectKind::Commit => 1, |
| 45 | ObjectKind::Tree => 2, |
| 46 | ObjectKind::Blob => 3, |
| 47 | ObjectKind::Tag => 4, |
| 48 | } |
| 49 | } |
| 50 | |
| 51 | fn name(self) -> &'static str { |
| 52 | match self { |
| 53 | ObjectKind::Commit => "commit", |
| 54 | ObjectKind::Tree => "tree", |
| 55 | ObjectKind::Blob => "blob", |
| 56 | ObjectKind::Tag => "tag", |
| 57 | } |
| 58 | } |
| 59 | } |
| 60 | |
| 61 | /// A git object's id: the SHA-1 of its header and content, in hex. |
| 62 | pub fn object_id(kind: ObjectKind, data: &[u8]) -> String { |
| 63 | let mut hasher = Sha1::new(); |
| 64 | hasher.update(format!("{} {}\0", kind.name(), data.len()).as_bytes()); |
| 65 | hasher.update(data); |
| 66 | hasher.finalize().iter().map(|byte| format!("{byte:02x}")).collect() |
| 67 | } |
| 68 | |
| 69 | enum Base { |
| 70 | /// An earlier object in the pack, by its offset. |
| 71 | Offset(usize), |
| 72 | /// Any object, by id. |
| 73 | Id(String), |
| 74 | } |
| 75 | |
| 76 | struct Delta { |
| 77 | base: Base, |
| 78 | data: Vec<u8>, |
| 79 | } |
| 80 | |
| 81 | /// The objects of a pack, by id. |
| 82 | #[derive(Default)] |
| 83 | pub struct Pack { |
| 84 | objects: HashMap<String, (ObjectKind, Vec<u8>)>, |
| 85 | /// Ids of the objects at each offset, once resolved. |
| 86 | at_offset: HashMap<usize, String>, |
| 87 | pending: Vec<(usize, Delta)>, |
| 88 | /// Commits, in the order the pack holds them. |
| 89 | commits: Vec<String>, |
| 90 | } |
| 91 | |
| 92 | /// Where the pack starts in a receive-pack request: after the commands |
| 93 | /// and anything else sent as pkt-lines. |
| 94 | pub fn pack_start(body: &[u8]) -> Option<usize> { |
| 95 | let mut at = 0; |
| 96 | loop { |
| 97 | if body.get(at..at + 4) == Some(b"PACK") { |
| 98 | return Some(at); |
| 99 | } |
| 100 | let length = std::str::from_utf8(body.get(at..at + 4)?) |
| 101 | .ok() |
| 102 | .and_then(|hex| usize::from_str_radix(hex, 16).ok())?; |
| 103 | // A flush packet is four bytes; any other line counts its own length. |
| 104 | at += if length == 0 { 4 } else { length.max(4) }; |
| 105 | } |
| 106 | } |
| 107 | |
| 108 | fn inflate(input: &[u8], size: usize) -> Result<(Vec<u8>, usize), String> { |
| 109 | let mut out = vec![0u8; size.max(1)]; |
| 110 | let mut state = DecompressorOxide::new(); |
| 111 | let flags = inflate_flags::TINFL_FLAG_PARSE_ZLIB_HEADER | inflate_flags::TINFL_FLAG_USING_NON_WRAPPING_OUTPUT_BUF; |
| 112 | let (status, consumed, written) = decompress(&mut state, input, &mut out, 0, flags); |
| 113 | match status { |
| 114 | TINFLStatus::Done => { |
| 115 | out.truncate(written); |
| 116 | if written != size { |
| 117 | return Err(format!("an object inflated to {written} bytes, not {size}")); |
| 118 | } |
| 119 | Ok((out, consumed)) |
| 120 | } |
| 121 | other => Err(format!("an object could not be inflated: {other:?}")), |
| 122 | } |
| 123 | } |
| 124 | |
| 125 | fn varint(data: &[u8], at: &mut usize) -> Option<usize> { |
| 126 | let (mut value, mut shift) = (0usize, 0); |
| 127 | loop { |
| 128 | let byte = *data.get(*at)?; |
| 129 | *at += 1; |
| 130 | value |= ((byte & 0x7f) as usize) << shift; |
| 131 | shift += 7; |
| 132 | if byte & 0x80 == 0 || shift > 56 { |
| 133 | return Some(value); |
| 134 | } |
| 135 | } |
| 136 | } |
| 137 | |
| 138 | /// An entry's header in a pack: its type and its inflated size. |
| 139 | fn header(code: u8, size: usize) -> Vec<u8> { |
| 140 | let mut out = Vec::new(); |
| 141 | let mut byte = (code << 4) | (size & 15) as u8; |
| 142 | let mut rest = size >> 4; |
| 143 | while rest > 0 { |
| 144 | out.push(byte | 0x80); |
| 145 | byte = (rest & 0x7f) as u8; |
| 146 | rest >>= 7; |
| 147 | } |
| 148 | out.push(byte); |
| 149 | out |
| 150 | } |
| 151 | |
| 152 | fn write_entries(pack: &mut Vec<u8>, objects: &[(ObjectKind, Vec<u8>)]) { |
| 153 | for (kind, data) in objects { |
| 154 | pack.extend(header(kind.code(), data.len())); |
| 155 | pack.extend(miniz_oxide::deflate::compress_to_vec_zlib(data, 6)); |
| 156 | } |
| 157 | } |
| 158 | |
| 159 | fn seal(mut pack: Vec<u8>) -> Vec<u8> { |
| 160 | let checksum = Sha1::digest(&pack); |
| 161 | pack.extend_from_slice(&checksum); |
| 162 | pack |
| 163 | } |
| 164 | |
| 165 | /// A version 2 pack holding `objects` whole, with no deltas: what git's |
| 166 | /// receive-pack takes, when the objects are few and new. |
| 167 | pub fn write_pack(objects: &[(ObjectKind, Vec<u8>)]) -> Vec<u8> { |
| 168 | let mut pack = b"PACK".to_vec(); |
| 169 | pack.extend_from_slice(&2u32.to_be_bytes()); |
| 170 | pack.extend_from_slice(&(objects.len() as u32).to_be_bytes()); |
| 171 | write_entries(&mut pack, objects); |
| 172 | seal(pack) |
| 173 | } |
| 174 | |
| 175 | /// `pack` with `objects` added after its own, as one pack. Its entries keep |
| 176 | /// their offsets, since the header stays the same length, so its deltas |
| 177 | /// still find their bases. `pack` must not be thin. |
| 178 | pub fn extend_pack(pack: &[u8], objects: &[(ObjectKind, Vec<u8>)]) -> Result<Vec<u8>, String> { |
| 179 | if pack.len() < 32 || &pack[..4] != b"PACK" { |
| 180 | return Err("not a pack".into()); |
| 181 | } |
| 182 | let (body, trailer) = pack.split_at(pack.len() - 20); |
| 183 | if Sha1::digest(body).as_slice() != trailer { |
| 184 | return Err("the pack's checksum does not match".into()); |
| 185 | } |
| 186 | let count = u32::from_be_bytes([pack[8], pack[9], pack[10], pack[11]]) as usize; |
| 187 | let total = u32::try_from(count + objects.len()).map_err(|_| "the pack is too large")?; |
| 188 | let mut out = body.to_vec(); |
| 189 | out[8..12].copy_from_slice(&total.to_be_bytes()); |
| 190 | write_entries(&mut out, objects); |
| 191 | Ok(seal(out)) |
| 192 | } |
| 193 | |
| 194 | /// Applies a git delta to its base. |
| 195 | pub fn apply_delta(base: &[u8], delta: &[u8]) -> Result<Vec<u8>, String> { |
| 196 | let mut at = 0; |
| 197 | let bad = || "a delta is malformed".to_owned(); |
| 198 | let source = varint(delta, &mut at).ok_or_else(bad)?; |
| 199 | if source != base.len() { |
| 200 | return Err("a delta does not fit its base".into()); |
| 201 | } |
| 202 | let target = varint(delta, &mut at).ok_or_else(bad)?; |
| 203 | let mut out = Vec::with_capacity(target); |
| 204 | while at < delta.len() { |
| 205 | let op = delta[at]; |
| 206 | at += 1; |
| 207 | if op & 0x80 != 0 { |
| 208 | let mut offset = 0usize; |
| 209 | let mut size = 0usize; |
| 210 | for bit in 0..4 { |
| 211 | if op & (1 << bit) != 0 { |
| 212 | offset |= (*delta.get(at).ok_or_else(bad)? as usize) << (8 * bit); |
| 213 | at += 1; |
| 214 | } |
| 215 | } |
| 216 | for bit in 0..3 { |
| 217 | if op & (0x10 << bit) != 0 { |
| 218 | size |= (*delta.get(at).ok_or_else(bad)? as usize) << (8 * bit); |
| 219 | at += 1; |
| 220 | } |
| 221 | } |
| 222 | if size == 0 { |
| 223 | size = 0x10000; |
| 224 | } |
| 225 | out.extend_from_slice(base.get(offset..offset + size).ok_or_else(bad)?); |
| 226 | } else if op != 0 { |
| 227 | out.extend_from_slice(delta.get(at..at + op as usize).ok_or_else(bad)?); |
| 228 | at += op as usize; |
| 229 | } else { |
| 230 | return Err(bad()); |
| 231 | } |
| 232 | } |
| 233 | if out.len() != target { |
| 234 | return Err(bad()); |
| 235 | } |
| 236 | Ok(out) |
| 237 | } |
| 238 | |
| 239 | impl Pack { |
| 240 | /// Reads every object in `pack`, resolving the deltas whose bases are |
| 241 | /// in it. |
| 242 | pub fn parse(pack: &[u8]) -> Result<Pack, String> { |
| 243 | if pack.len() < 12 || &pack[..4] != b"PACK" { |
| 244 | return Err("not a pack".into()); |
| 245 | } |
| 246 | let count = u32::from_be_bytes([pack[8], pack[9], pack[10], pack[11]]) as usize; |
| 247 | let mut at = 12; |
| 248 | let mut result = Pack::default(); |
| 249 | let mut inflated = 0usize; |
| 250 | for _ in 0..count { |
| 251 | let start = at; |
| 252 | let mut byte = *pack.get(at).ok_or("the pack ends early")?; |
| 253 | at += 1; |
| 254 | let code = (byte >> 4) & 7; |
| 255 | let mut size = (byte & 15) as usize; |
| 256 | let mut shift = 4; |
| 257 | while byte & 0x80 != 0 { |
| 258 | byte = *pack.get(at).ok_or("the pack ends early")?; |
| 259 | at += 1; |
| 260 | size |= ((byte & 0x7f) as usize) << shift; |
| 261 | shift += 7; |
| 262 | } |
| 263 | inflated += size; |
| 264 | if inflated > MAX_INFLATED { |
| 265 | return Err("the pack is too large to read".into()); |
| 266 | } |
| 267 | let base = match code { |
| 268 | 6 => { |
| 269 | let mut byte = *pack.get(at).ok_or("the pack ends early")?; |
| 270 | at += 1; |
| 271 | let mut offset = (byte & 0x7f) as usize; |
| 272 | while byte & 0x80 != 0 { |
| 273 | byte = *pack.get(at).ok_or("the pack ends early")?; |
| 274 | at += 1; |
| 275 | offset = ((offset + 1) << 7) | (byte & 0x7f) as usize; |
| 276 | } |
| 277 | Some(Base::Offset(start.checked_sub(offset).ok_or("a delta points before the pack")?)) |
| 278 | } |
| 279 | 7 => { |
| 280 | let id = pack.get(at..at + 20).ok_or("the pack ends early")?; |
| 281 | at += 20; |
| 282 | Some(Base::Id(id.iter().map(|byte| format!("{byte:02x}")).collect())) |
| 283 | } |
| 284 | _ => None, |
| 285 | }; |
| 286 | let (data, consumed) = inflate(&pack[at..], size)?; |
| 287 | at += consumed; |
| 288 | match base { |
| 289 | Some(base) => result.pending.push((start, Delta { base, data })), |
| 290 | None => { |
| 291 | let kind = ObjectKind::from_type(code).ok_or("an object has an unknown type")?; |
| 292 | result.insert(start, kind, data); |
| 293 | } |
| 294 | } |
| 295 | } |
| 296 | result.resolve(); |
| 297 | Ok(result) |
| 298 | } |
| 299 | |
| 300 | fn insert(&mut self, offset: usize, kind: ObjectKind, data: Vec<u8>) { |
| 301 | let id = object_id(kind, &data); |
| 302 | if kind == ObjectKind::Commit { |
| 303 | self.commits.push(id.clone()); |
| 304 | } |
| 305 | self.at_offset.insert(offset, id.clone()); |
| 306 | self.objects.insert(id, (kind, data)); |
| 307 | } |
| 308 | |
| 309 | /// Resolves every pending delta whose base is known by now. |
| 310 | fn resolve(&mut self) { |
| 311 | loop { |
| 312 | let mut progress = false; |
| 313 | let pending = std::mem::take(&mut self.pending); |
| 314 | for (offset, delta) in pending { |
| 315 | let base_id = match &delta.base { |
| 316 | Base::Offset(base) => self.at_offset.get(base).cloned(), |
| 317 | Base::Id(id) => Some(id.clone()), |
| 318 | }; |
| 319 | let resolved = base_id |
| 320 | .and_then(|id| self.objects.get(&id)) |
| 321 | .map(|(kind, base)| (*kind, apply_delta(base, &delta.data))); |
| 322 | match resolved { |
| 323 | Some((kind, Ok(data))) => { |
| 324 | self.insert(offset, kind, data); |
| 325 | progress = true; |
| 326 | } |
| 327 | // A delta that does not apply is dropped. |
| 328 | Some((_, Err(_))) => progress = true, |
| 329 | None => self.pending.push((offset, delta)), |
| 330 | } |
| 331 | } |
| 332 | if !progress || self.pending.is_empty() { |
| 333 | return; |
| 334 | } |
| 335 | } |
| 336 | } |
| 337 | |
| 338 | /// Objects the pack's deltas are based on that it does not hold: what |
| 339 | /// the repository has to supply. |
| 340 | pub fn missing_bases(&self) -> Vec<String> { |
| 341 | let mut ids: Vec<String> = self |
| 342 | .pending |
| 343 | .iter() |
| 344 | .filter_map(|(_, delta)| match &delta.base { |
| 345 | Base::Id(id) if !self.objects.contains_key(id) => Some(id.clone()), |
| 346 | _ => None, |
| 347 | }) |
| 348 | .collect(); |
| 349 | ids.sort(); |
| 350 | ids.dedup(); |
| 351 | ids |
| 352 | } |
| 353 | |
| 354 | /// Supplies a base object from the repository, and resolves what |
| 355 | /// depends on it. |
| 356 | pub fn supply(&mut self, id: &str, kind: ObjectKind, data: Vec<u8>) { |
| 357 | self.objects.insert(id.to_owned(), (kind, data)); |
| 358 | self.resolve(); |
| 359 | } |
| 360 | |
| 361 | /// Deltas still unresolved. |
| 362 | pub fn unresolved(&self) -> usize { |
| 363 | self.pending.len() |
| 364 | } |
| 365 | |
| 366 | pub fn get(&self, id: &str) -> Option<(ObjectKind, &[u8])> { |
| 367 | self.objects.get(id).map(|(kind, data)| (*kind, data.as_slice())) |
| 368 | } |
| 369 | |
| 370 | pub fn contains(&self, id: &str) -> bool { |
| 371 | self.objects.contains_key(id) |
| 372 | } |
| 373 | |
| 374 | /// The commits in the pack: what the push adds. |
| 375 | pub fn commits(&self) -> &[String] { |
| 376 | &self.commits |
| 377 | } |
| 378 | |
| 379 | pub fn commit(&self, id: &str) -> Option<CommitInfo> { |
| 380 | match self.get(id)? { |
| 381 | (ObjectKind::Commit, data) => Some(parse_commit(data)), |
| 382 | _ => None, |
| 383 | } |
| 384 | } |
| 385 | |
| 386 | pub fn tree(&self, id: &str) -> Option<Vec<TreeItem>> { |
| 387 | match self.get(id)? { |
| 388 | (ObjectKind::Tree, data) => Some(parse_tree(data)), |
| 389 | _ => None, |
| 390 | } |
| 391 | } |
| 392 | |
| 393 | pub fn blob(&self, id: &str) -> Option<&[u8]> { |
| 394 | match self.get(id)? { |
| 395 | (ObjectKind::Blob, data) => Some(data), |
| 396 | _ => None, |
| 397 | } |
| 398 | } |
| 399 | } |
| 400 | |
| 401 | /// What a commit says about its place in history, and whose it is. |
| 402 | #[derive(Debug, Default, PartialEq, Eq)] |
| 403 | pub struct CommitInfo { |
| 404 | pub tree: String, |
| 405 | pub parents: Vec<String>, |
| 406 | /// The address on the `author` line, as written. |
| 407 | pub author_email: Option<String>, |
| 408 | /// The address on the `committer` line, as written. |
| 409 | pub committer_email: Option<String>, |
| 410 | } |
| 411 | |
| 412 | /// The address in a signature line's value: `Name <address> 1700000000 +0000`. |
| 413 | fn signature_email(value: &str) -> Option<String> { |
| 414 | let start = value.rfind('<')?; |
| 415 | let end = start + value[start..].find('>')?; |
| 416 | Some(value[start + 1..end].trim().to_owned()) |
| 417 | } |
| 418 | |
| 419 | pub fn parse_commit(data: &[u8]) -> CommitInfo { |
| 420 | let text = String::from_utf8_lossy(data); |
| 421 | let mut info = CommitInfo::default(); |
| 422 | for line in text.lines() { |
| 423 | if line.is_empty() { |
| 424 | break; |
| 425 | } |
| 426 | if let Some(tree) = line.strip_prefix("tree ") { |
| 427 | info.tree = tree.trim().to_owned(); |
| 428 | } else if let Some(parent) = line.strip_prefix("parent ") { |
| 429 | info.parents.push(parent.trim().to_owned()); |
| 430 | } else if let Some(author) = line.strip_prefix("author ") { |
| 431 | info.author_email = signature_email(author); |
| 432 | } else if let Some(committer) = line.strip_prefix("committer ") { |
| 433 | info.committer_email = signature_email(committer); |
| 434 | } |
| 435 | } |
| 436 | info |
| 437 | } |
| 438 | |
| 439 | /// One entry of a tree. |
| 440 | #[derive(Clone, Debug, PartialEq, Eq)] |
| 441 | pub struct TreeItem { |
| 442 | /// `100644`, `100755`, `120000`, `40000` or `160000`. |
| 443 | pub mode: String, |
| 444 | pub name: String, |
| 445 | pub id: String, |
| 446 | } |
| 447 | |
| 448 | impl TreeItem { |
| 449 | pub fn is_tree(&self) -> bool { |
| 450 | self.mode == "40000" |
| 451 | } |
| 452 | |
| 453 | /// A regular or executable file; not a link or a submodule. |
| 454 | pub fn is_file(&self) -> bool { |
| 455 | self.mode.starts_with("100") |
| 456 | } |
| 457 | } |
| 458 | |
| 459 | pub fn parse_tree(data: &[u8]) -> Vec<TreeItem> { |
| 460 | let mut items = Vec::new(); |
| 461 | let mut at = 0; |
| 462 | while at < data.len() { |
| 463 | let Some(space) = data[at..].iter().position(|byte| *byte == b' ') else { |
| 464 | break; |
| 465 | }; |
| 466 | let Some(nul) = data[at + space..].iter().position(|byte| *byte == 0) else { |
| 467 | break; |
| 468 | }; |
| 469 | let mode = String::from_utf8_lossy(&data[at..at + space]).into_owned(); |
| 470 | let name = String::from_utf8_lossy(&data[at + space + 1..at + space + nul]).into_owned(); |
| 471 | let id_at = at + space + nul + 1; |
| 472 | let Some(id) = data.get(id_at..id_at + 20) else { |
| 473 | break; |
| 474 | }; |
| 475 | items.push(TreeItem { mode, name, id: id.iter().map(|byte| format!("{byte:02x}")).collect() }); |
| 476 | at = id_at + 20; |
| 477 | } |
| 478 | items |
| 479 | } |
| 480 | |
| 481 | /// A tree's bytes from its entries, as git writes them: what a delta |
| 482 | /// against a tree the repository has needs as its base. |
| 483 | pub fn encode_tree(items: &[TreeItem]) -> Vec<u8> { |
| 484 | let mut out = Vec::new(); |
| 485 | for item in items { |
| 486 | out.extend_from_slice(item.mode.as_bytes()); |
| 487 | out.push(b' '); |
| 488 | out.extend_from_slice(item.name.as_bytes()); |
| 489 | out.push(0); |
| 490 | for pair in item.id.as_bytes().chunks(2) { |
| 491 | out.push(u8::from_str_radix(std::str::from_utf8(pair).unwrap_or("00"), 16).unwrap_or(0)); |
| 492 | } |
| 493 | } |
| 494 | out |
| 495 | } |
| 496 | |
| 497 | #[cfg(test)] |
| 498 | pub(crate) mod tests { |
| 499 | use super::*; |
| 500 | use miniz_oxide::deflate::compress_to_vec_zlib; |
| 501 | |
| 502 | /// A pack of whole objects, plus ref-deltas given as (base id, delta). |
| 503 | pub fn build_pack(objects: &[(ObjectKind, Vec<u8>)], ref_deltas: &[(String, Vec<u8>)]) -> Vec<u8> { |
| 504 | let mut pack = b"PACK".to_vec(); |
| 505 | pack.extend_from_slice(&2u32.to_be_bytes()); |
| 506 | pack.extend_from_slice(&((objects.len() + ref_deltas.len()) as u32).to_be_bytes()); |
| 507 | for (kind, data) in objects { |
| 508 | pack.extend(header(kind.code(), data.len())); |
| 509 | pack.extend(compress_to_vec_zlib(data, 6)); |
| 510 | } |
| 511 | for (base, delta) in ref_deltas { |
| 512 | pack.extend(header(7, delta.len())); |
| 513 | for pair in base.as_bytes().chunks(2) { |
| 514 | pack.push(u8::from_str_radix(std::str::from_utf8(pair).unwrap(), 16).unwrap()); |
| 515 | } |
| 516 | pack.extend(compress_to_vec_zlib(delta, 6)); |
| 517 | } |
| 518 | pack.extend_from_slice(&[0u8; 20]); |
| 519 | pack |
| 520 | } |
| 521 | |
| 522 | /// A delta that keeps the first `keep` bytes of `base` and appends `tail`. |
| 523 | pub fn delta(base: &[u8], keep: usize, tail: &[u8]) -> Vec<u8> { |
| 524 | let mut out = Vec::new(); |
| 525 | let put = |mut value: usize, out: &mut Vec<u8>| loop { |
| 526 | let byte = (value & 0x7f) as u8; |
| 527 | value >>= 7; |
| 528 | if value == 0 { |
| 529 | out.push(byte); |
| 530 | break; |
| 531 | } |
| 532 | out.push(byte | 0x80); |
| 533 | }; |
| 534 | put(base.len(), &mut out); |
| 535 | put(keep + tail.len(), &mut out); |
| 536 | // Copy from offset 0, `keep` bytes (one size byte). |
| 537 | out.push(0x80 | 0x10); |
| 538 | out.push(keep as u8); |
| 539 | out.push(tail.len() as u8); |
| 540 | out.extend_from_slice(tail); |
| 541 | out |
| 542 | } |
| 543 | |
| 544 | #[test] |
| 545 | fn ids_match_git() { |
| 546 | assert_eq!(object_id(ObjectKind::Blob, b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391"); |
| 547 | assert_eq!(object_id(ObjectKind::Blob, b"hello\n"), "ce013625030ba8dba906f756967f9e9ca394464a"); |
| 548 | } |
| 549 | |
| 550 | #[test] |
| 551 | fn whole_objects_and_deltas_are_read() { |
| 552 | let base = b"first line\n".to_vec(); |
| 553 | let base_id = object_id(ObjectKind::Blob, &base); |
| 554 | let pack = build_pack(&[(ObjectKind::Blob, base.clone())], &[(base_id.clone(), delta(&base, base.len(), b"second\n"))]); |
| 555 | let parsed = Pack::parse(&pack).unwrap(); |
| 556 | let grown = b"first line\nsecond\n"; |
| 557 | assert_eq!(parsed.blob(&object_id(ObjectKind::Blob, grown)), Some(&grown[..])); |
| 558 | assert!(parsed.missing_bases().is_empty()); |
| 559 | } |
| 560 | |
| 561 | #[test] |
| 562 | fn a_thin_pack_waits_for_its_base() { |
| 563 | let base = b"kept in the repository\n".to_vec(); |
| 564 | let base_id = object_id(ObjectKind::Blob, &base); |
| 565 | let pack = build_pack(&[], &[(base_id.clone(), delta(&base, 4, b" and more\n"))]); |
| 566 | let mut parsed = Pack::parse(&pack).unwrap(); |
| 567 | assert_eq!(parsed.missing_bases(), vec![base_id.clone()]); |
| 568 | parsed.supply(&base_id, ObjectKind::Blob, base); |
| 569 | assert_eq!(parsed.unresolved(), 0); |
| 570 | assert!(parsed.blob(&object_id(ObjectKind::Blob, b"kept and more\n")).is_some()); |
| 571 | } |
| 572 | |
| 573 | #[test] |
| 574 | fn commits_and_trees_are_parsed_and_trees_rebuilt() { |
| 575 | let blob_id = object_id(ObjectKind::Blob, b"x"); |
| 576 | let tree = encode_tree(&[ |
| 577 | TreeItem { mode: "100644".into(), name: "a.txt".into(), id: blob_id.clone() }, |
| 578 | TreeItem { mode: "40000".into(), name: "src".into(), id: blob_id.clone() }, |
| 579 | ]); |
| 580 | let items = parse_tree(&tree); |
| 581 | assert_eq!(items.len(), 2); |
| 582 | assert!(items[0].is_file() && items[1].is_tree()); |
| 583 | assert_eq!(encode_tree(&items), tree); |
| 584 | let commit = format!("tree {}\nparent aaaa\nparent bbbb\nauthor x\n\nmessage\nparent no\n", object_id(ObjectKind::Tree, &tree)); |
| 585 | let info = parse_commit(commit.as_bytes()); |
| 586 | assert_eq!(info.parents, ["aaaa", "bbbb"]); |
| 587 | assert_eq!(info.tree.len(), 40); |
| 588 | assert_eq!(info.author_email, None); |
| 589 | let signed = parse_commit( |
| 590 | b"tree t |
| 591 | author Ada L <Ada@Example.com> 1700000000 +0000 |
| 592 | committer Bot <bot@x.io> 1700000000 +0000 |
| 593 | |
| 594 | author <no@x.io> |
| 595 | ", |
| 596 | ); |
| 597 | assert_eq!(signed.author_email.as_deref(), Some("Ada@Example.com")); |
| 598 | assert_eq!(signed.committer_email.as_deref(), Some("bot@x.io")); |
| 599 | let pack = build_pack(&[(ObjectKind::Commit, commit.into_bytes())], &[]); |
| 600 | assert_eq!(Pack::parse(&pack).unwrap().commits().len(), 1); |
| 601 | } |
| 602 | |
| 603 | #[test] |
| 604 | fn the_pack_is_found_after_the_commands() { |
| 605 | let line = b"old new refs/heads/PACKAGING\0report-status\n"; |
| 606 | let commands = [format!("{:04x}", line.len() + 4).into_bytes(), line.to_vec(), b"0000".to_vec()].concat(); |
| 607 | let body = [commands.clone(), b"PACK\0\0\0\x02".to_vec()].concat(); |
| 608 | assert_eq!(pack_start(&body), Some(commands.len())); |
| 609 | assert_eq!(pack_start(&commands), None); |
| 610 | assert!(Pack::parse(b"nope").is_err()); |
| 611 | } |
| 612 | |
| 613 | #[test] |
| 614 | fn written_packs_are_sealed_and_extend() { |
| 615 | let blob = b"hello |
| 616 | ".to_vec(); |
| 617 | let pack = write_pack(&[(ObjectKind::Blob, blob.clone())]); |
| 618 | let (body, trailer) = pack.split_at(pack.len() - 20); |
| 619 | assert_eq!(Sha1::digest(body).as_slice(), trailer); |
| 620 | let more = extend_pack(&pack, &[(ObjectKind::Blob, b"more |
| 621 | ".to_vec())]).unwrap(); |
| 622 | assert_eq!(&more[8..12], &2u32.to_be_bytes()); |
| 623 | let read = Pack::parse(&more).unwrap(); |
| 624 | assert!(read.blob("ce013625030ba8dba906f756967f9e9ca394464a").is_some()); |
| 625 | assert!(read.blob(&object_id(ObjectKind::Blob, b"more |
| 626 | ")).is_some()); |
| 627 | let mut broken = pack.clone(); |
| 628 | broken[12] ^= 1; |
| 629 | assert!(extend_pack(&broken, &[]).is_err()); |
| 630 | } |
| 631 | } |