Commit

Download ZIP reads blobs eight batches at a time and deflates files together

syntaqxcommitted Parent03c23b4Browse files
2 files+19−100/2 viewed
+4−2
3232 const parts: Uint8Array[] = [];
3333 const central: Uint8Array[] = [];
3434 let offset = 0;
35− for (const file of files) {
35+ // Every file deflated at once: the streams interleave rather than queue.
36+ const deflatedAll = await Promise.all(files.map((file) => (file.data.length > 0 ? deflate(file.data) : file.data)));
37+ for (const [index, file] of files.entries()) {
3638 const name = encoder.encode(file.path);
3739 const crc = crc32(file.data);
38− const deflated = file.data.length > 0 ? await deflate(file.data) : file.data;
40+ const deflated = deflatedAll[index]!;
3941 const [method, body] = deflated.length < file.data.length ? [8, deflated] : [0, file.data];
4042 const header = (central: boolean) => {
4143 const out = new Uint8Array((central ? 46 : 30) + name.length);
+15−8
1313 const MAX_FILES = 10_000;
1414 // The bytes, their base64 and the zip are all held at once, in a Worker of 128 MB.
1515 const MAX_BYTES = 24 * 1024 * 1024;
16−/** Blobs read per call to the repository. */
16+/** Blobs read per call to the repository, and calls at once. */
1717 const PER_READ = 100;
18+const READS_AT_ONCE = 8;
1819
1920 function refused(status: number, message: string): Response {
2021 return new Response(`${message}\n`, { status, headers: { "content-type": "text/plain; charset=utf-8", "cache-control": "no-store" } });
3738 const unique = [...new Set(files.map((file) => file.hash))];
3839 const bytes = new Map<string, Uint8Array>();
3940 let total = 0;
40− for (let at = 0; at < unique.length; at += PER_READ) {
41− for (const blob of await repos.rawBlobs(repo.value.id, unique.slice(at, at + PER_READ), MAX_BYTES)) {
42− total += blob.size;
43− if (total > MAX_BYTES || (blob.data == null && blob.size > 0)) {
44− return refused(413, `It is over ${MAX_BYTES / 1024 / 1024} MB, too large for a download. Clone it instead: ${clone}`);
41+ let tooLarge = false;
42+ const batches: string[][] = [];
43+ for (let at = 0; at < unique.length; at += PER_READ) batches.push(unique.slice(at, at + PER_READ));
44+ // A few reads at a time, each taking the next batch, until all are read or it is too large.
45+ const reader = async () => {
46+ for (let batch = batches.shift(); batch && !tooLarge; batch = batches.shift()) {
47+ for (const blob of await repos.rawBlobs(repo.value.id, batch, MAX_BYTES)) {
48+ total += blob.size;
49+ if (total > MAX_BYTES || (blob.data == null && blob.size > 0)) tooLarge = true;
50+ else bytes.set(blob.hash, Uint8Array.from(atob(blob.data ?? ""), (c) => c.charCodeAt(0)));
4551 }
46− bytes.set(blob.hash, Uint8Array.from(atob(blob.data ?? ""), (c) => c.charCodeAt(0)));
4752 }
48− }
53+ };
54+ await Promise.all(Array.from({ length: READS_AT_ONCE }, reader));
55+ if (tooLarge) return refused(413, `It is over ${MAX_BYTES / 1024 / 1024} MB, too large for a download. Clone it instead: ${clone}`);
4956 // A branch name can hold slashes; the folder and file name cannot.
5057 const label = `${repo.value.name}-${ref.replaceAll("/", "-")}`;
5158 const archive = await zip(files.map((file) => ({ path: `${label}/${file.path}`, data: bytes.get(file.hash) ?? new Uint8Array() })));