Merge project overview: one branch_drift call, spliced histories, cached tags, 6 repos calls instead of 25
15 files+857−2330/15 viewed
| 14 | 14 | } from "@g1t/contracts"; | |
| 15 | 15 | ||
| 16 | 16 | import type { ViewerAccess } from "./access"; | |
| 17 | − | import { repos, work } from "./services.server"; | |
| 17 | + | import { projects, repos, work } from "./services.server"; | |
| 18 | 18 | import { getViewer } from "./session.server"; | |
| 19 | 19 | ||
| 20 | 20 | type Context = Parameters<typeof getViewer>[0]; | |
| ⋯ | |||
| 68 | 68 | return found; | |
| 69 | 69 | } | |
| 70 | 70 | ||
| 71 | + | /** | |
| 72 | + | * One lookup of a project per request: the project's layout and its | |
| 73 | + | * overview load at the same time and both need it. | |
| 74 | + | */ | |
| 75 | + | const projectLookups = new WeakMap<object, Map<string, ReturnType<typeof projects.get>>>(); | |
| 76 | + | ||
| 77 | + | export function projectFor(context: Context, params: RepoParams): ReturnType<typeof projects.get> { | |
| 78 | + | const key = `${params.owner}/${params.repo}`.toLowerCase(); | |
| 79 | + | let seen = projectLookups.get(context); | |
| 80 | + | if (!seen) projectLookups.set(context, (seen = new Map())); | |
| 81 | + | let found = seen.get(key); | |
| 82 | + | if (!found) { | |
| 83 | + | found = projects.get(params.owner ?? "", params.repo ?? "", getViewer(context)); | |
| 84 | + | seen.set(key, found); | |
| 85 | + | } | |
| 86 | + | return found; | |
| 87 | + | } | |
| 88 | + | ||
| 71 | 89 | /** The viewer's role on a repository and what it lets them do. */ | |
| 72 | 90 | export function accessFor(viewer: Viewer, repo: Repo): ViewerAccess { | |
| 73 | 91 | return { | |
| 4 | 4 | * it with its checks, and its preview. The overview shows the newest few; | |
| 5 | 5 | * the Branches page shows them all. | |
| 6 | 6 | */ | |
| 7 | − | import type { Branch, Commit, Pull, RepoPath, Viewer } from "@g1t/contracts"; | |
| 7 | + | import type { Branch, BranchDrifts, Pull, RepoPath, Viewer } from "@g1t/contracts"; | |
| 8 | 8 | ||
| 9 | 9 | import type { ActiveBranch } from "../components/branches"; | |
| 10 | − | import { bounded, drift } from "./branches"; | |
| 11 | − | import { immutable } from "./immutable.server"; | |
| 10 | + | import { activeBranches, branchesToRead, summary } from "./branches"; | |
| 12 | 11 | import { repos } from "./services.server"; | |
| 13 | − | ||
| 14 | − | /** | |
| 15 | − | * How deep each history is read, in turn, to find where a branch and the | |
| 16 | − | * default branch meet: most branches meet it within the first; a branch | |
| 17 | − | * left long ago needs the default branch's history further back; one far | |
| 18 | − | * from both reads both deeply. Past the last, the counts are not shown. | |
| 19 | − | */ | |
| 20 | − | const DEPTHS: ReadonlyArray<{ branch: number; main: number }> = [ | |
| 21 | − | { branch: 40, main: 120 }, | |
| 22 | − | { branch: 40, main: 1000 }, | |
| 23 | − | { branch: 1000, main: 1000 }, | |
| 24 | − | ]; | |
| 25 | − | /** Branches counted at once. */ | |
| 26 | − | const COUNTING = 8; | |
| 27 | 12 | ||
| 28 | 13 | type Preview = { branch?: string | null; number?: number | null; url: string }; | |
| 29 | − | /** What never changes for a branch head and a default branch head. */ | |
| 30 | − | type Measured = { commit: ActiveBranch["commit"]; drift: ActiveBranch["drift"] }; | |
| 31 | 14 | ||
| 32 | − | const summary = (commit: Commit | undefined): ActiveBranch["commit"] => | |
| 33 | − | commit ? { hash: commit.hash, message: commit.message.split("\n")[0] ?? "", author: commit.author.name, at: commit.authoredAt } : null; | |
| 34 | − | ||
| 35 | 15 | /** | |
| 36 | 16 | * The branches other than the default, at most `read` of them read (those | |
| 37 | − | * with an open pull request first), newest commit first. `main` is the | |
| 38 | − | * default branch's head commit with its first line. | |
| 17 | + | * with an open pull request first), newest commit first, and the default | |
| 18 | + | * branch's head commit with its first line. | |
| 19 | + | * | |
| 20 | + | * One call to repos (`branch_drift`, services/repos/src/drift.rs) measures | |
| 21 | + | * every branch. It keeps each answer by the pair of head commits, so only | |
| 22 | + | * branches that moved, or every branch once the default branch moved, cost | |
| 23 | + | * a walk, and that walk reads the default branch's history once for all of | |
| 24 | + | * them. Before 2026-10-08 this was a `log` call per branch per depth from | |
| 25 | + | * here, up to twenty on an overview. | |
| 39 | 26 | */ | |
| 40 | 27 | export async function readBranches( | |
| 41 | 28 | path: RepoPath, | |
| ⋯ | |||
| 43 | 30 | input: { defaultBranch: string; branches: Branch[]; pulls: Pull[]; previews: Preview[] }, | |
| 44 | 31 | read: number, | |
| 45 | 32 | ): Promise<{ main: string; total: number; shown: ActiveBranch[]; head: ActiveBranch["commit"] }> { | |
| 46 | − | const soft = <T,>(promise: Promise<T>): Promise<T | null> => promise.catch(() => null); | |
| 47 | − | const main = input.defaultBranch; | |
| 48 | − | const pullOn = new Map(input.pulls.filter((pull) => pull.branch).map((pull) => [pull.branch as string, pull])); | |
| 49 | − | const others = input.branches.filter((branch) => branch.name !== main); | |
| 50 | − | const reading = [...others.filter((b) => pullOn.has(b.name)), ...others.filter((b) => !pullOn.has(b.name))].slice(0, read); | |
| 51 | − | // By commit hash, not name: history from a commit never changes, so | |
| 52 | − | // repos keeps it (services/repos/src/store.rs) and only new heads cost a walk. | |
| 53 | − | const mainHead = input.branches.find((branch) => branch.name === main)?.hash ?? null; | |
| 54 | − | const log = async (hash: string, depth: number): Promise<Commit[] | null> => { | |
| 55 | − | const found = await soft(repos.log(path, viewer, hash, depth)); | |
| 56 | − | return found?.ok && found.value.length > 0 ? found.value : null; | |
| 57 | − | }; | |
| 58 | − | // The default branch's history, read once per depth for every branch, | |
| 59 | − | // and not read deeper when a shallower read already reached its start. | |
| 60 | − | const mainLogs = new Map<number, Promise<Commit[] | null>>(); | |
| 61 | − | const mainLog = (depth: number): Promise<Commit[] | null> => { | |
| 62 | − | if (!mainHead) return Promise.resolve(null); | |
| 63 | − | let kept = mainLogs.get(depth); | |
| 64 | − | if (!kept) { | |
| 65 | − | const shallower = Math.max(0, ...[...mainLogs.keys()].filter((read) => read < depth)); | |
| 66 | − | const before = mainLogs.get(shallower); | |
| 67 | − | kept = before | |
| 68 | − | ? before.then((read) => (read && read.length < shallower ? read : log(mainHead, depth))) | |
| 69 | − | : log(mainHead, depth); | |
| 70 | − | mainLogs.set(depth, kept); | |
| 71 | − | } | |
| 72 | − | return kept; | |
| 73 | − | }; | |
| 74 | − | // Reads deeper only while the two histories have not met. Null when a | |
| 75 | − | // read failed, so a failure is not kept as the answer. | |
| 76 | − | const measure = async (head: string, main: string): Promise<Measured | null> => { | |
| 77 | − | let branch: Commit[] | null = null; | |
| 78 | − | let readTo = 0; | |
| 79 | − | for (const depth of DEPTHS) { | |
| 80 | − | // Not read again when no deeper, or when it already reached the start. | |
| 81 | − | const again: boolean = branch == null || (depth.branch > readTo && branch.length >= readTo); | |
| 82 | − | const [read, mainRead]: [Commit[] | null, Commit[] | null] = await Promise.all([ | |
| 83 | − | again ? log(head, depth.branch) : branch, | |
| 84 | − | mainLog(depth.main), | |
| 85 | − | ]); | |
| 86 | − | if (!read || !mainRead) return null; | |
| 87 | − | if (again) readTo = depth.branch; | |
| 88 | − | branch = read; | |
| 89 | − | const counted = drift(head, main, [...mainRead, ...read]); | |
| 90 | − | if (counted) return { commit: summary(read[0]), drift: counted }; | |
| 91 | − | } | |
| 92 | − | return { commit: summary(branch?.[0]), drift: null }; | |
| 93 | − | }; | |
| 94 | − | // Kept by the pair of heads: neither history can change, so neither can | |
| 95 | − | // the answer. Without one, the head commit alone. | |
| 96 | − | const measureOrHead = async (branch: Branch): Promise<Measured> => { | |
| 97 | − | const kept = | |
| 98 | − | branch.hash && mainHead | |
| 99 | − | ? await immutable(`branch-drift:${path.namespace}/${path.name}:${mainHead}:${branch.hash}`, () => measure(branch.hash, mainHead)) | |
| 100 | − | : null; | |
| 101 | − | if (kept) return kept; | |
| 102 | − | const head = await log(branch.hash || branch.name, 1); | |
| 103 | − | return { commit: summary(head?.[0]), drift: null }; | |
| 104 | − | }; | |
| 105 | − | const [mainTop, measured] = await Promise.all([ | |
| 106 | − | mainHead ? log(mainHead, 1) : Promise.resolve(null), | |
| 107 | − | bounded(reading, COUNTING, measureOrHead), | |
| 108 | − | ]); | |
| 109 | − | const shown = reading | |
| 110 | − | .map((branch, index): ActiveBranch => { | |
| 111 | − | const pull = pullOn.get(branch.name); | |
| 112 | − | return { | |
| 113 | − | name: branch.name, | |
| 114 | − | commit: measured[index]?.commit ?? null, | |
| 115 | − | drift: measured[index]?.drift ?? null, | |
| 116 | − | pull: pull ? { number: pull.number, title: pull.title, checkStatus: pull.checkStatus, draft: pull.status === "draft" } : null, | |
| 117 | − | preview: input.previews.find((app) => app.branch === branch.name || (pull != null && app.number === pull.number))?.url ?? null, | |
| 118 | − | }; | |
| 119 | − | }) | |
| 120 | − | .sort((a, b) => Date.parse(b.commit?.at ?? "0") - Date.parse(a.commit?.at ?? "0")); | |
| 121 | − | return { main, total: others.length, shown, head: summary(mainTop?.[0]) }; | |
| 33 | + | const { reading, total, mainHead } = branchesToRead(input, read); | |
| 34 | + | const measured: BranchDrifts | null = mainHead | |
| 35 | + | ? await repos | |
| 36 | + | .branchDrift(path, viewer, mainHead, reading.map((branch) => branch.hash).filter(Boolean)) | |
| 37 | + | .then((found) => (found.ok ? found.value : null)) | |
| 38 | + | .catch(() => null) | |
| 39 | + | : null; | |
| 40 | + | return { main: input.defaultBranch, total, shown: activeBranches(reading, measured, input), head: summary(measured?.base) }; | |
| 122 | 41 | } | |
| 1 | 1 | import assert from "node:assert/strict"; | |
| 2 | 2 | import { test } from "node:test"; | |
| 3 | 3 | ||
| 4 | − | import { bounded, drift, type Link } from "./branches.ts"; | |
| 4 | + | import type { BranchDrifts, Commit } from "@g1t/contracts"; | |
| 5 | 5 | ||
| 6 | − | /** A history from `[hash, ...parents]` rows. */ | |
| 7 | − | const graph = (...rows: string[][]): Link[] => rows.map(([hash, ...parents]) => ({ hash: hash as string, parents })); | |
| 6 | + | import { activeBranches, branchesToRead } from "./branches.ts"; | |
| 8 | 7 | ||
| 9 | − | // main: m1 <- m2 <- m3; the branch left at m2 and added b1 <- b2. | |
| 10 | − | const forked = graph(["m3", "m2"], ["m2", "m1"], ["m1"], ["b2", "b1"], ["b1", "m2"]); | |
| 11 | − | ||
| 12 | − | test("a branch two ahead of where main was, with main one further on", () => { | |
| 13 | − | assert.deepEqual(drift("b2", "m3", forked), { ahead: 2, behind: 1 }); | |
| 14 | − | }); | |
| 15 | − | ||
| 16 | − | test("a branch at main's head is level", () => { | |
| 17 | − | assert.deepEqual(drift("m3", "m3", forked), { ahead: 0, behind: 0 }); | |
| 18 | − | }); | |
| 19 | − | ||
| 20 | − | test("a branch merged long ago is nothing ahead and all of main since behind", () => { | |
| 21 | − | const history = graph(["m5", "m4"], ["m4", "m3"], ["m3", "m2"], ["m2", "m1"]); | |
| 22 | − | // Neither history was read to its start, but they meet in what was. | |
| 23 | − | assert.deepEqual(drift("m2", "m5", history), { ahead: 0, behind: 3 }); | |
| 8 | + | const commit = (hash: string, at: string, message = `${hash}\n\nbody`): Commit => ({ | |
| 9 | + | hash, | |
| 10 | + | treeHash: `t${hash}`, | |
| 11 | + | message, | |
| 12 | + | author: { name: "Ada", email: "ada@example.com" }, | |
| 13 | + | parents: [], | |
| 14 | + | authoredAt: at, | |
| 24 | 15 | }); | |
| 25 | 16 | ||
| 26 | − | test("a merge into main counts the merged side once", () => { | |
| 27 | − | // main merged side branch s1 <- s2 at m3; the branch is still at m1. | |
| 28 | − | const history = graph(["m3", "m2", "s2"], ["s2", "s1"], ["s1", "m1"], ["m2", "m1"], ["m1"], ["b1", "m1"]); | |
| 29 | − | assert.deepEqual(drift("b1", "m3", history), { ahead: 1, behind: 4 }); | |
| 30 | − | }); | |
| 17 | + | const branches = [ | |
| 18 | + | { name: "main", hash: "m3" }, | |
| 19 | + | { name: "old", hash: "o1" }, | |
| 20 | + | { name: "fix", hash: "f2" }, | |
| 21 | + | { name: "idea", hash: "i1" }, | |
| 22 | + | ]; | |
| 23 | + | const pulls = [{ branch: "fix", number: 7, title: "Fix it", checkStatus: "passed" as const, status: "open" as const }]; | |
| 31 | 24 | ||
| 32 | − | test("histories that never meet count everything on each, once read to the start", () => { | |
| 33 | − | const history = graph(["b2", "b1"], ["b1"], ["m2", "m1"], ["m1"]); | |
| 34 | − | assert.deepEqual(drift("b2", "m2", history), { ahead: 2, behind: 2 }); | |
| 25 | + | test("branches with an open pull request are read first, the default branch never", () => { | |
| 26 | + | const read = branchesToRead({ defaultBranch: "main", branches, pulls }, 2); | |
| 27 | + | assert.deepEqual(read.reading.map((b) => b.name), ["fix", "old"]); | |
| 28 | + | assert.equal(read.total, 3); | |
| 29 | + | assert.equal(read.mainHead, "m3"); | |
| 35 | 30 | }); | |
| 36 | 31 | ||
| 37 | − | test("no answer when what was read stops before the two meet", () => { | |
| 38 | − | // Main's history was read only to m2, whose parent the branch may share. | |
| 39 | − | const history = graph(["m3", "m2"], ["m2", "m1"], ["b2", "b1"], ["b1", "m0"]); | |
| 40 | − | assert.equal(drift("b2", "m3", history), null); | |
| 32 | + | test("no default branch head, nothing to measure against", () => { | |
| 33 | + | assert.equal(branchesToRead({ defaultBranch: "trunk", branches, pulls: [] }, 10).mainHead, null); | |
| 41 | 34 | }); | |
| 42 | 35 | ||
| 43 | − | test("no answer without either head", () => { | |
| 44 | − | assert.equal(drift("b9", "m3", forked), null); | |
| 45 | − | assert.equal(drift("b2", "m9", forked), null); | |
| 36 | + | test("each branch gets its measured commit and drift, newest first, by head hash", () => { | |
| 37 | + | const measured: BranchDrifts = { | |
| 38 | + | base: commit("m3", "2026-10-08T00:00:00Z"), | |
| 39 | + | branches: [ | |
| 40 | + | { head: "f2", commit: commit("f2", "2026-10-07T00:00:00Z"), drift: { ahead: 2, behind: 1 } }, | |
| 41 | + | { head: "o1", commit: commit("o1", "2026-01-01T00:00:00Z"), drift: null }, | |
| 42 | + | { head: "i1", commit: commit("i1", "2026-10-08T00:00:00Z", "one line"), drift: { ahead: 1, behind: 0 } }, | |
| 43 | + | ], | |
| 44 | + | }; | |
| 45 | + | const { reading } = branchesToRead({ defaultBranch: "main", branches, pulls }, 10); | |
| 46 | + | const shown = activeBranches(reading, measured, { pulls, previews: [{ number: 7, url: "https://fix.g1t.page" }] }); | |
| 47 | + | assert.deepEqual(shown.map((b) => b.name), ["idea", "fix", "old"]); | |
| 48 | + | assert.deepEqual(shown[1], { | |
| 49 | + | name: "fix", | |
| 50 | + | commit: { hash: "f2", message: "f2", author: "Ada", at: "2026-10-07T00:00:00Z" }, | |
| 51 | + | drift: { ahead: 2, behind: 1 }, | |
| 52 | + | pull: { number: 7, title: "Fix it", checkStatus: "passed", draft: false }, | |
| 53 | + | preview: "https://fix.g1t.page", | |
| 54 | + | }); | |
| 55 | + | assert.equal(shown[2]?.drift, null); | |
| 46 | 56 | }); | |
| 47 | 57 | ||
| 48 | − | test("bounded loads everything in order, never more at once than asked", async () => { | |
| 49 | − | let running = 0; | |
| 50 | − | let most = 0; | |
| 51 | − | const out = await bounded([5, 1, 4, 2, 3], 2, async (wait) => { | |
| 52 | − | running++; | |
| 53 | − | most = Math.max(most, running); | |
| 54 | − | await new Promise((done) => setTimeout(done, wait)); | |
| 55 | − | running--; | |
| 56 | − | return wait * 10; | |
| 57 | − | }); | |
| 58 | − | assert.deepEqual(out, [50, 10, 40, 20, 30]); | |
| 59 | − | assert.equal(most, 2); | |
| 58 | + | test("when repos could not answer, the branches still show, without commits or counts", () => { | |
| 59 | + | const { reading } = branchesToRead({ defaultBranch: "main", branches, pulls: [] }, 10); | |
| 60 | + | const shown = activeBranches(reading, null, { pulls: [], previews: [] }); | |
| 61 | + | assert.equal(shown.length, 3); | |
| 62 | + | assert.ok(shown.every((b) => b.commit == null && b.drift == null && b.pull == null)); | |
| 60 | 63 | }); |
| 1 | 1 | /** | |
| 2 | + | * Active branches as the overview and the Branches page show them, from | |
| 3 | + | * the repos service's `branch_drift` answer. The counting itself is done | |
| 4 | + | * there (services/repos/src/drift.rs), kept by the pair of head commits. | |
| 5 | + | */ | |
| 6 | + | import type { Branch, BranchDrifts, Commit, Pull } from "@g1t/contracts"; | |
| 7 | + | ||
| 8 | + | import type { ActiveBranch } from "../components/branches"; | |
| 9 | + | ||
| 10 | + | /** | |
| 2 | 11 | * How far a branch has moved from the default branch: commits it has that | |
| 3 | 12 | * the default branch does not (ahead), and commits the default branch has | |
| 4 | 13 | * that it does not (behind), the way `git rev-list --left-right --count` | |
| 5 | − | * says it. Worked out from what was read of the two histories, merges | |
| 6 | − | * included; when what was read stops short of where they meet, there is no | |
| 7 | − | * answer rather than a guess. | |
| 14 | + | * says it. | |
| 8 | 15 | */ | |
| 9 | 16 | export type Drift = { ahead: number; behind: number }; | |
| 10 | 17 | ||
| 11 | − | /** A commit as far as counting needs it. */ | |
| 12 | − | export type Link = { hash: string; parents: string[] }; | |
| 18 | + | type Preview = { branch?: string | null; number?: number | null; url: string }; | |
| 13 | 19 | ||
| 20 | + | /** A commit as a branch row shows it: its first line. */ | |
| 21 | + | export const summary = (commit: Commit | null | undefined): ActiveBranch["commit"] => | |
| 22 | + | commit ? { hash: commit.hash, message: commit.message.split("\n")[0] ?? "", author: commit.author.name, at: commit.authoredAt } : null; | |
| 23 | + | ||
| 14 | 24 | /** | |
| 15 | − | * `commits` is everything read of either history, in any order, repeats | |
| 16 | − | * allowed. Null when either head is missing from it, or when a commit only | |
| 17 | − | * one side reaches has a parent that was not read: that parent's history | |
| 18 | − | * could change either count. | |
| 25 | + | * The branches other than the default that are read (at most `read`, those | |
| 26 | + | * with an open pull request first), how many there are, and the default | |
| 27 | + | * branch's head commit. | |
| 19 | 28 | */ | |
| 20 | − | export function drift(branch: string, main: string, commits: Iterable<Link>): Drift | null { | |
| 21 | − | const parents = new Map<string, string[]>(); | |
| 22 | − | for (const commit of commits) parents.set(commit.hash, commit.parents); | |
| 23 | − | if (!parents.has(branch) || !parents.has(main)) return null; | |
| 24 | − | const fromBranch = reach(branch, parents); | |
| 25 | − | const fromMain = reach(main, parents); | |
| 26 | − | let ahead = 0; | |
| 27 | − | let behind = 0; | |
| 28 | − | for (const [hash, above] of parents) { | |
| 29 | − | const onBranch = fromBranch.has(hash); | |
| 30 | − | const onMain = fromMain.has(hash); | |
| 31 | − | if (onBranch === onMain) continue; | |
| 32 | − | if (above.some((parent) => !parents.has(parent))) return null; | |
| 33 | − | if (onBranch) ahead++; | |
| 34 | − | else behind++; | |
| 35 | − | } | |
| 36 | − | return { ahead, behind }; | |
| 37 | − | } | |
| 38 | − | ||
| 39 | − | /** Every commit read that `head` descends from, itself included. */ | |
| 40 | − | function reach(head: string, parents: Map<string, string[]>): Set<string> { | |
| 41 | − | const seen = new Set([head]); | |
| 42 | − | const next = [head]; | |
| 43 | − | for (let hash = next.pop(); hash != null; hash = next.pop()) { | |
| 44 | − | for (const parent of parents.get(hash) ?? []) { | |
| 45 | − | if (parents.has(parent) && !seen.has(parent)) { | |
| 46 | − | seen.add(parent); | |
| 47 | − | next.push(parent); | |
| 48 | − | } | |
| 49 | − | } | |
| 50 | − | } | |
| 51 | − | return seen; | |
| 29 | + | export function branchesToRead( | |
| 30 | + | input: { defaultBranch: string; branches: Branch[]; pulls: Pick<Pull, "branch">[] }, | |
| 31 | + | read: number, | |
| 32 | + | ): { reading: Branch[]; total: number; mainHead: string | null } { | |
| 33 | + | const pullOn = new Set(input.pulls.flatMap((pull) => (pull.branch ? [pull.branch] : []))); | |
| 34 | + | const others = input.branches.filter((branch) => branch.name !== input.defaultBranch); | |
| 35 | + | const reading = [...others.filter((b) => pullOn.has(b.name)), ...others.filter((b) => !pullOn.has(b.name))].slice(0, read); | |
| 36 | + | const mainHead = input.branches.find((branch) => branch.name === input.defaultBranch)?.hash || null; | |
| 37 | + | return { reading, total: others.length, mainHead }; | |
| 52 | 38 | } | |
| 53 | 39 | ||
| 54 | − | /** `load` over each of `items`, at most `limit` at a time, answers in order. */ | |
| 55 | − | export async function bounded<T, R>(items: readonly T[], limit: number, load: (item: T) => Promise<R>): Promise<R[]> { | |
| 56 | − | const out = new Array<R>(items.length); | |
| 57 | − | let taken = 0; | |
| 58 | − | const worker = async () => { | |
| 59 | − | while (taken < items.length) { | |
| 60 | − | const index = taken++; | |
| 61 | − | out[index] = await load(items[index] as T); | |
| 62 | − | } | |
| 63 | − | }; | |
| 64 | − | await Promise.all(Array.from({ length: Math.min(limit, items.length) }, worker)); | |
| 65 | − | return out; | |
| 40 | + | /** | |
| 41 | + | * Each branch read, newest commit first, with what repos measured (null | |
| 42 | + | * when that could not be had: no commits, no counts), its pull request and | |
| 43 | + | * its preview. | |
| 44 | + | */ | |
| 45 | + | export function activeBranches( | |
| 46 | + | reading: Branch[], | |
| 47 | + | measured: BranchDrifts | null, | |
| 48 | + | input: { pulls: Pick<Pull, "branch" | "number" | "title" | "checkStatus" | "status">[]; previews: Preview[] }, | |
| 49 | + | ): ActiveBranch[] { | |
| 50 | + | const pullOn = new Map(input.pulls.flatMap((pull) => (pull.branch ? [[pull.branch, pull] as const] : []))); | |
| 51 | + | const byHead = new Map((measured?.branches ?? []).map((found) => [found.head, found])); | |
| 52 | + | return reading | |
| 53 | + | .map((branch): ActiveBranch => { | |
| 54 | + | const pull = pullOn.get(branch.name); | |
| 55 | + | const found = byHead.get(branch.hash); | |
| 56 | + | return { | |
| 57 | + | name: branch.name, | |
| 58 | + | commit: summary(found?.commit), | |
| 59 | + | drift: found?.drift ?? null, | |
| 60 | + | pull: pull ? { number: pull.number, title: pull.title, checkStatus: pull.checkStatus, draft: pull.status === "draft" } : null, | |
| 61 | + | preview: input.previews.find((app) => app.branch === branch.name || (pull != null && app.number === pull.number))?.url ?? null, | |
| 62 | + | }; | |
| 63 | + | }) | |
| 64 | + | .sort((a, b) => Date.parse(b.commit?.at ?? "0") - Date.parse(a.commit?.at ?? "0")); | |
| 66 | 65 | } |
| 64 | 64 | }); | |
| 65 | 65 | ||
| 66 | 66 | test("only known reads are taken not to write", () => { | |
| 67 | − | for (const method of ["get_pull", "list_pulls", "counts", "user_for_session", "explore", "usage", "get", "list", "queue", "pulls_for_repos", "stars", "about", "public_links"]) { | |
| 67 | + | for (const method of ["get_pull", "list_pulls", "counts", "user_for_session", "explore", "usage", "get", "list", "queue", "pulls_for_repos", "stars", "about", "public_links", "branch_drift", "tags", "commit_checks", "shortcuts", "last_commits", "languages"]) { | |
| 68 | 68 | assert.equal(mayWrite(method), false, method); | |
| 69 | 69 | } | |
| 70 | 70 | for (const method of ["merge_pull", "verify_email", "github_finish", "sign_in", "something_new"]) { |
| 100 | 100 | * page called `get` (repos, projects), so every one set the cookie, was | |
| 101 | 101 | * never kept in the public cache, and sent the next 30 s of the person's | |
| 102 | 102 | * reads to the primary. `stars`, `about` and `public_links` (2026-10-08) | |
| 103 | − | * did the same to every project page and Explore for 13 hours. | |
| 103 | + | * did the same to every project page and Explore for 13 hours. `tags`, | |
| 104 | + | * `commit_checks` and `shortcuts` (every overview), and the Files page's | |
| 105 | + | * `last_commits`, `languages`, `contributors` and `license`, still made a | |
| 106 | + | * signed-in view count as a write until 2026-10-08: the cookie, 30 s of | |
| 107 | + | * primary reads, and no sidebar cache after every overview. | |
| 104 | 108 | */ | |
| 105 | 109 | const READS = new Set( | |
| 106 | 110 | ( | |
| ⋯ | |||
| 113 | 117 | "routes run run_context run_cost runner_groups runner_settings runners runs scorecards search search_memories " + | |
| 114 | 118 | "settings statement statement_entries status status_by_id suggest tree usage usage_meters user_by_username " + | |
| 115 | 119 | "user_for_session usernames waiting_workspaces workflows workspace workspace_invites github_enabled " + | |
| 116 | − | "stars about public_links" | |
| 120 | + | "stars about public_links branch_drift tags last_commits languages contributors license releases release " + | |
| 121 | + | "stargazers starred commit_checks shortcuts" | |
| 117 | 122 | ).split(" "), | |
| 118 | 123 | ); | |
| 119 | 124 | ||
| 20 | 20 | import { clearWelcome, welcomes } from "../../lib/invites"; | |
| 21 | 21 | import { notFound } from "../../lib/not-found.server"; | |
| 22 | 22 | import { redirectIfRenamed, redirectIfTransferred } from "../../lib/renamed.server"; | |
| 23 | − | import { accessFor, countsFor, repoFor } from "../../lib/access.server"; | |
| 23 | + | import { accessFor, countsFor, projectFor, repoFor } from "../../lib/access.server"; | |
| 24 | 24 | import { inbox, projects, repos } from "../../lib/services.server"; | |
| 25 | 25 | import { getViewer, roleIn, unwrap } from "../../lib/session.server"; | |
| 26 | 26 | ||
| ⋯ | |||
| 36 | 36 | const [repo, counts, found, watching, shortcuts, stars] = await Promise.all([ | |
| 37 | 37 | repoFor(context, params), | |
| 38 | 38 | countsFor(context, params), | |
| 39 | − | projects.get(params.owner, params.repo, viewer), | |
| 39 | + | projectFor(context, params), | |
| 40 | 40 | // How the person watches it, for the header's Watch menu, as soon as | |
| 41 | 41 | // the repository is known. | |
| 42 | 42 | viewer | |
| 22 | 22 | Rocket, | |
| 23 | 23 | RotateCw, | |
| 24 | 24 | } from "lucide-react"; | |
| 25 | + | import { waitUntil } from "cloudflare:workers"; | |
| 26 | + | import { isbot } from "isbot"; | |
| 25 | 27 | import { type ReactNode, Suspense } from "react"; | |
| 26 | 28 | import { Await, Form, Link } from "react-router"; | |
| 27 | 29 | ||
| ⋯ | |||
| 92 | 94 | import { actions, agents, deployments, events as eventLog, identity, packages, projects, repos, work } from "../../lib/services.server"; | |
| 93 | 95 | import { madeByG1t } from "../../lib/opened-by"; | |
| 94 | 96 | import { assertSameOrigin, getViewer, requireUser } from "../../lib/session.server"; | |
| 95 | − | import { accessTo, countsFor, refusal, repoFor } from "../../lib/access.server"; | |
| 97 | + | import { accessTo, countsFor, projectFor, refusal, repoFor } from "../../lib/access.server"; | |
| 96 | 98 | import { shotVersion } from "./production-screenshot"; | |
| 97 | 99 | import { DeploymentsPanel } from "../../components/deployments-panel"; | |
| 98 | 100 | import { environmentUrl, productionEnvironment } from "../../lib/deployments"; | |
| ⋯ | |||
| 111 | 113 | const COMMITS_SHOWN = 5; | |
| 112 | 114 | /** How long Active branches may take before the section links to Branches instead. */ | |
| 113 | 115 | const BRANCHES_WAIT_MS = 3_500; | |
| 116 | + | /** | |
| 117 | + | * The same for a crawler, which gets the page only once everything in it | |
| 118 | + | * has settled (entry.server.tsx): past this, it gets the link to Branches. | |
| 119 | + | */ | |
| 120 | + | const BRANCHES_WAIT_CRAWLER_MS = 700; | |
| 114 | 121 | ||
| 115 | 122 | /** | |
| 116 | 123 | * The overview streams: the layout's header and tabs (one repository | |
| ⋯ | |||
| 119 | 126 | * same response as it settles. Crawlers wait for all of it | |
| 120 | 127 | * (entry.server.tsx). Before, the first byte waited on the slowest of them. | |
| 121 | 128 | */ | |
| 122 | − | export function loader({ params, context }: Route.LoaderArgs) { | |
| 123 | − | return { overview: overviewData({ params, context }) }; | |
| 129 | + | export function loader({ params, context, request }: Route.LoaderArgs) { | |
| 130 | + | const userAgent = request.headers.get("user-agent"); | |
| 131 | + | return { overview: overviewData({ params, context }, Boolean(userAgent && isbot(userAgent))) }; | |
| 124 | 132 | } | |
| 125 | 133 | ||
| 126 | − | async function overviewData({ params, context }: Pick<Route.LoaderArgs, "params" | "context">) { | |
| 134 | + | async function overviewData({ params, context }: Pick<Route.LoaderArgs, "params" | "context">, crawler: boolean) { | |
| 127 | 135 | const viewer = getViewer(context); | |
| 128 | 136 | const path = { namespace: params.owner, name: params.repo }; | |
| 129 | 137 | const ref = { workspace: params.owner, slug: params.repo }; | |
| ⋯ | |||
| 140 | 148 | const forMembers = <T,>(start: () => Promise<T>): Promise<T | null> => | |
| 141 | 149 | memberP.then((member) => (member ? soft(start()) : null)); | |
| 142 | 150 | const repoP = soft(repoFor(context, params)); | |
| 143 | − | const projectP = soft(projects.get(params.owner, params.repo, viewer)); | |
| 151 | + | const projectP = soft(projectFor(context, params)); | |
| 144 | 152 | // A library or a tool shows its packages where an app shows production, | |
| 145 | 153 | // and every project g1t does not deploy counts a workflow in its checklist. | |
| 146 | 154 | // Deployments from anywhere (g1t.page, g1t Actions, the API), by environment, for anyone who can read it. | |
| ⋯ | |||
| 182 | 190 | return { main, total: read.total, shown: read.shown.slice(0, BRANCHES_SHOWN) }; | |
| 183 | 191 | }, | |
| 184 | 192 | ); | |
| 185 | − | // Active branches read several logs each: streamed, so the rest shows | |
| 186 | − | // first. Bounded well inside the response's stream timeout | |
| 187 | − | // (entry.server.tsx): a promise still pending when the stream ends never | |
| 188 | − | // settles in the browser, and its skeleton would stay. | |
| 193 | + | // Active branches walk history when a branch or the default branch has | |
| 194 | + | // moved: streamed, so the rest shows first. Bounded well inside the | |
| 195 | + | // response's stream timeout (entry.server.tsx): a promise still pending | |
| 196 | + | // when the stream ends never settles in the browser, and its skeleton | |
| 197 | + | // would stay. The walk finishes after the page if it must (waitUntil), | |
| 198 | + | // so repos keeps the answer and the next view has it. Crawlers, which | |
| 199 | + | // wait for the whole page, wait for it only briefly. | |
| 200 | + | waitUntil(branchesP.then(() => undefined, () => undefined)); | |
| 189 | 201 | const branches = Promise.race([ | |
| 190 | 202 | branchesP.catch(() => null), | |
| 191 | − | new Promise<"slow">((resolve) => setTimeout(() => resolve("slow"), BRANCHES_WAIT_MS)), | |
| 203 | + | new Promise<"slow">((resolve) => setTimeout(() => resolve("slow"), crawler ? BRANCHES_WAIT_CRAWLER_MS : BRANCHES_WAIT_MS)), | |
| 192 | 204 | ]); | |
| 193 | 205 | const [{ insider: member, can }, project, settings, list, open, closed, log, counts, deps, runs, queue, issues, memories, recent, mine, domains, root, packageList, workflows, tagList] = await Promise.all([ | |
| 194 | 206 | accessP, | |
| 617 | 617 | pub complete: bool, | |
| 618 | 618 | } | |
| 619 | 619 | ||
| 620 | + | /// `branch_drift`: how far each of `heads` (branch head commits) has moved | |
| 621 | + | /// from `base` (the default branch's head commit), and each one's head | |
| 622 | + | /// commit, in one call. Every answer is kept by the pair of hashes: neither | |
| 623 | + | /// history can change, so neither can it. Returns `Outcome<BranchDrifts>`. | |
| 624 | + | #[derive(Debug, Serialize, Deserialize)] | |
| 625 | + | #[serde(rename_all = "camelCase")] | |
| 626 | + | pub struct BranchDriftArgs { | |
| 627 | + | pub path: RepoPath, | |
| 628 | + | pub viewer: Viewer, | |
| 629 | + | pub base: String, | |
| 630 | + | pub heads: Vec<String>, | |
| 631 | + | } | |
| 632 | + | ||
| 633 | + | /// Commits a branch has that the default branch does not (`ahead`), and | |
| 634 | + | /// the other way round (`behind`), as `git rev-list --left-right --count`. | |
| 635 | + | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 636 | + | pub struct Drift { | |
| 637 | + | pub ahead: u32, | |
| 638 | + | pub behind: u32, | |
| 639 | + | } | |
| 640 | + | ||
| 641 | + | /// One branch head's commit and drift. `drift` is absent when the two | |
| 642 | + | /// histories do not meet within what is read (or could not be read). | |
| 643 | + | #[derive(Clone, Debug, Serialize, Deserialize)] | |
| 644 | + | pub struct BranchDrift { | |
| 645 | + | pub head: String, | |
| 646 | + | pub commit: Option<Commit>, | |
| 647 | + | pub drift: Option<Drift>, | |
| 648 | + | } | |
| 649 | + | ||
| 650 | + | /// `base`'s own commit, and each head's answer in the order asked. | |
| 651 | + | #[derive(Clone, Debug, Serialize, Deserialize)] | |
| 652 | + | pub struct BranchDrifts { | |
| 653 | + | pub base: Option<Commit>, | |
| 654 | + | pub branches: Vec<BranchDrift>, | |
| 655 | + | } | |
| 656 | + | ||
| 620 | 657 | /// `tags`: the repository's tags, newest commit first, each with the | |
| 621 | 658 | /// commit it names. Returns `Outcome<Vec<Tag>>`. | |
| 622 | 659 | #[derive(Debug, Serialize, Deserialize)] |
| 188 | 188 | | Public pages for people signed out | the data centre's cache (`workers/app.ts`, `servePublic`) | fresh 30 s, then served once more while a new copy is made, up to 5 min | GET, no `g1t_session` cookie, an allowlisted path (home, pricing, explore, policies, a project's pages), status 200 or 404, no `Set-Cookie`, nothing private. Reserved first segments and workspace pages (`-`) are never kept. A project's kept page is served only after repos' `visibility` says the repository is still there and public (one indexed read, alongside the cache lookup); a repository made private or deleted is never served from any data centre's copy, and the copy is dropped. The answer says `server-timing: cache;desc="hit, Ns old"`. | | |
| 189 | 189 | | Sidebar data (projects, spend, limit, entitlements) | per isolate (`lib/cache.server.ts`) | 15 s, per person and workspace | skipped during a write and for 30 s after the person's last one; failures not kept; only settled answers kept | | |
| 190 | 190 | | Registration mode | per isolate | 60 s | | | |
| 191 | − | | A commit's log by hash | repos' data-centre cache | for good | history from a commit never changes; Active branches asks by hash | | |
| 191 | + | | A commit's log by hash | repos' data-centre cache | for good | history from a commit never changes. One of 100 commits or more is put together from a 16-commit read and the history kept from any of those commits, when there is one (`store.rs` `spliced_log`): a default branch that moved by a merge costs 16 commits, not 120 or 1,000 | | |
| 192 | + | | A branch's drift from the default branch (Active branches) | repos' data-centre cache (`branch_drift`) | for good | by repository and the pair of head commits; a failed read is not kept | | |
| 193 | + | | A repository's tags | repos' data-centre cache | until the refs move, 5 min at most | as the branch list; not kept when a tag's commit could not be read | | |
| 192 | 194 | | Git objects, trees, refs | repos' caches | see services/repos | | | |
| 193 | 195 | | A branch's log, the branch list, a file by branch and path | repos' data-centre cache | until the repository's refs change (`refs_version`), 5 min at most | only while no handed-out push credential is live; by commit hash for good (docs/ARTIFACTS.md R9) | | |
| 194 | 196 | | A target branch's history, for mergeability | the repos isolate | 60 s, per target head | 100 pull requests checked after a push walk it once (R10) | | |
| ⋯ | |||
| 235 | 237 | | --- | --- | --- | | |
| 236 | 238 | | Any in-app navigation | root (sidebar: 6 calls, then up to 20 `get_by_id` for shared repositories) and the project layout re-ran when moving between pages of a project | root re-runs only when the workspace or project changes, or after a form; the project layout likewise; shared repositories are one `readable` call, in the same round; the sidebar's workspace data is cached for 15 s; open counts are read once per request for both | | |
| 237 | 239 | | Pull request ("Review and respond") | access, then 8 calls, then checks' runs / comparison / session, then up to 5 more deployments lookups for stacked previews | one round of 9 (access-dependent ones start as soon as the repository lookup returns), then the comparison on Changes; the workflow jobs and stacked previews stream in | | |
| 238 | − | | Project overview | access, then 17 calls, one of which (Active branches) read the default branch's last 120 commits and up to 10 branches' last 40 | one round; Active branches streams in with a skeleton, reading logs by commit hash so a branch that has not moved costs nothing | | |
| 240 | + | | Project overview | access, then 17 calls, one of which (Active branches) read the default branch's last 120 commits and up to 10 branches' last 40 | one round; Active branches streams in with a skeleton, in one `branch_drift` call (below) | | |
| 239 | 241 | | Mission control | per project: open pulls, closed pulls and events (3 × up to 10), then `get_by_id` per unknown repository | one `pulls_for_repos` call for every project (one access check, one query), events per project alongside, one `readable` for the rest | | |
| 240 | 242 | | Issue, issues | access, then the rest | one round | | |
| 241 | 243 | ||
| ⋯ | |||
| 287 | 289 | waits for `allReady` for bots), and signed out it is kept in the public | |
| 288 | 290 | cache like any other project page. | |
| 289 | 291 | ||
| 292 | + | ### Active branches (2026-10-08) | |
| 293 | + | ||
| 294 | + | Measured on production, `/flagon-io/g1t` uncached, as a crawler, soon | |
| 295 | + | after pushes: **3,606 ms** to the first byte, `rpc` 3,605 ms over 38 | |
| 296 | + | service calls, **repos 25 calls, 17,953 ms inside**. Nearly all of it was | |
| 297 | + | Active branches: | |
| 298 | + | ||
| 299 | + | - The site measured each branch itself: a `log` call per branch per | |
| 300 | + | depth (40, then 1,000), plus the default branch's (120, then 1,000), plus | |
| 301 | + | one more for the default branch's head, each call with its own access | |
| 302 | + | check and store handle. The answer was kept by the pair of heads in the | |
| 303 | + | site's cache, so every push to a branch, and every merge to the default | |
| 304 | + | branch (which changes every pair), started the walks again, in every | |
| 305 | + | data centre. | |
| 306 | + | - A branch far from the default branch read 1,000 commits of both, and | |
| 307 | + | the default branch's 1,000 again after each merge. | |
| 308 | + | - The page waits 3.5 s for the section, then shows a link to Branches. | |
| 309 | + | The walks were not in `waitUntil`, so when the page stopped waiting | |
| 310 | + | they were dropped with it: an answer that took longer than 3.5 s was | |
| 311 | + | never kept, and the next view started over. 3,606 ms is that timeout. | |
| 312 | + | ||
| 313 | + | Now: | |
| 314 | + | ||
| 315 | + | - **One call.** `branch_drift` (services/repos/src/drift.rs) takes the | |
| 316 | + | default branch's head and every branch head, checks access once, opens | |
| 317 | + | the store once, and reads the default branch's history once per depth | |
| 318 | + | for all of them. Each answer is kept in repos' data-centre cache by the | |
| 319 | + | pair of hashes; only pairs that changed are walked. It also returns the | |
| 320 | + | default branch's head commit, which the site read with its own call. | |
| 321 | + | - **Shallower first.** Depths are (branch, default branch) 12/120, then | |
| 322 | + | 40/120, 40/1,000 and 1,000/1,000: most branches are a few commits | |
| 323 | + | ahead and meet at the first. | |
| 324 | + | - **Long histories spliced.** A log by hash of 100 commits or more is a | |
| 325 | + | 16-commit read plus the log kept from one of those commits (the | |
| 326 | + | first-parent chain from a commit never changes), so the default branch | |
| 327 | + | after a merge costs 16 commits instead of 120 or 1,000. | |
| 328 | + | - **Finished after the page.** The call runs in `waitUntil`, so repos | |
| 329 | + | keeps the answer even when the page stopped waiting for it. | |
| 330 | + | - **Crawlers wait 0.7 s** for the section (browsers 3.5 s, streamed); | |
| 331 | + | past that they get the link to Branches, as a slow browser does. | |
| 332 | + | - `tags` is kept until the refs move (it listed the refs from the store | |
| 333 | + | on every call), and the project is looked up once per request for the | |
| 334 | + | layout and the overview (`projectFor`, beside `repoFor`). | |
| 335 | + | - `tags`, `commit_checks`, `shortcuts` and the Files page's reads were | |
| 336 | + | missing from `READS`, so every signed-in overview counted as a write: | |
| 337 | + | it set `g1t_d1`, sent the next 30 seconds of reads to the primary and | |
| 338 | + | turned off the sidebar cache. | |
| 339 | + | ||
| 340 | + | | Overview, signed out, uncached | Before | After | | |
| 341 | + | | --- | --- | --- | | |
| 342 | + | | repos calls | 25 (6 when nothing had moved) | 6 whatever moved: `get`, `stars`, `branches`, `log`, `tags`, `branch_drift` | | |
| 343 | + | | projects calls | 3 (`get` twice) | 2 | | |
| 344 | + | | Crawler, after a push | 3,606 ms (the 3.5 s timeout) | at most about 0.8 s: the rest of the page, or 0.7 s for Active branches | | |
| 345 | + | | Crawler, nothing moved | 460 to 570 ms | not yet measured on production | | |
| 346 | + | | Browser, first byte | 115 to 160 ms (`total`), 200 to 245 ms measured from Colorado | unchanged: the page does not wait for any of this | | |
| 347 | + | ||
| 290 | 348 | ## Client navigation | |
| 291 | 349 | ||
| 292 | 350 | - `<Link prefetch="intent">` on the sidebar, project tabs, breadcrumbs, | |
| ⋯ | |||
| 350 | 408 | It prints p50 and p90 of the server's share (TLS handshake done to first | |
| 351 | 409 | byte), where the Worker ran, whether the answer set `g1t_d1` (it should | |
| 352 | 410 | not, for a page that only reads), and the slowest Server-Timing entries. | |
| 411 | + | ||
| 412 | + | Signed out, a public page is usually answered from the data centre's | |
| 413 | + | cache (`server-timing: cache;desc="hit, …"`), and `cache-control: | |
| 414 | + | no-cache` does not change that. To time a render, add a query string the | |
| 415 | + | page ignores: the cache is keyed by the whole URL, so | |
| 416 | + | `/flagon-io/g1t?nc=<random>` is always a miss. A crawler's user agent | |
| 417 | + | (`Googlebot/2.1`) waits for the whole page; a browser's gets the first | |
| 418 | + | byte and the streamed rest. | |
| 367 | 367 | call("git_access", { path, viewer, service }), | |
| 368 | 368 | branches: (path, viewer) => call("branches", { path, viewer }), | |
| 369 | 369 | lastCommits: (path, viewer, ref, treePath) => call("last_commits", { path, viewer, ref, treePath }), | |
| 370 | + | branchDrift: (path, viewer, base, heads) => call("branch_drift", { path, viewer, base, heads }), | |
| 370 | 371 | tags: (path, viewer) => call("tags", { path, viewer }), | |
| 371 | 372 | about: (path, viewer) => call("about", { path, viewer }), | |
| 372 | 373 | languages: (path, viewer) => call("languages", { path, viewer }), |
| 246 | 246 | /** Which commit last changed each entry of a directory at `ref` (the default branch when null). */ | |
| 247 | 247 | lastCommits(path: RepoPath, viewer: Viewer, ref: string | null, treePath: string): Promise<Result<LastCommits>>; | |
| 248 | 248 | ||
| 249 | + | /** | |
| 250 | + | * How far each branch head has moved from `base` (the default branch's | |
| 251 | + | * head commit), with each head's commit and `base`'s own, in one call. | |
| 252 | + | * Kept by the pair of hashes in the repos service. | |
| 253 | + | */ | |
| 254 | + | branchDrift(path: RepoPath, viewer: Viewer, base: string, heads: string[]): Promise<Result<BranchDrifts>>; | |
| 255 | + | ||
| 249 | 256 | /** The repository's tags, newest commit first, at most 100. */ | |
| 250 | 257 | tags(path: RepoPath, viewer: Viewer): Promise<Result<Tag[]>>; | |
| 251 | 258 | ||
| ⋯ | |||
| 372 | 379 | /** Each entry's last commit; `complete` is false when some were not reached. */ | |
| 373 | 380 | export type LastCommits = { entries: { name: string; commit: Commit }[]; complete: boolean }; | |
| 374 | 381 | ||
| 382 | + | /** Commits a branch has that the default branch does not, and the other way round. */ | |
| 383 | + | export type BranchDriftCount = { ahead: number; behind: number }; | |
| 384 | + | ||
| 385 | + | /** | |
| 386 | + | * `base`'s commit, and each head's commit and drift in the order asked; | |
| 387 | + | * `drift` is null when the two histories do not meet within what is read. | |
| 388 | + | */ | |
| 389 | + | export type BranchDrifts = { | |
| 390 | + | base: Commit | null; | |
| 391 | + | branches: { head: string; commit: Commit | null; drift: BranchDriftCount | null }[]; | |
| 392 | + | }; | |
| 393 | + | ||
| 375 | 394 | /** A tag, and its commit when it could be read. */ | |
| 376 | 395 | export type Tag = { name: string; commit: Commit | null }; | |
| 377 | 396 | ||
| 1 | + | //! How far branches have moved from the default branch, for Active branches | |
| 2 | + | //! on a project's overview and the Branches page (`branch_drift`). | |
| 3 | + | //! | |
| 4 | + | //! Each branch head's history and the default branch's are read to a depth, | |
| 5 | + | //! in turn deeper, until they meet; the default branch's history is read | |
| 6 | + | //! once per depth for every branch. Histories are read by commit hash, which | |
| 7 | + | //! the store keeps for good (store.rs), and each answer is kept by the pair | |
| 8 | + | //! of heads (lib.rs), so only heads that moved cost a walk. Before | |
| 9 | + | //! 2026-10-08 the site did this itself: up to twenty `log` calls per view, | |
| 10 | + | //! each with its own access check and store handle. | |
| 11 | + | ||
| 12 | + | use std::collections::{HashMap, HashSet}; | |
| 13 | + | ||
| 14 | + | use g1t_contracts::repos::{Commit, Drift}; | |
| 15 | + | use worker::Result; | |
| 16 | + | ||
| 17 | + | use crate::store::GitRepo; | |
| 18 | + | ||
| 19 | + | /// How deep each history is read, in turn: (branch, default branch). Most | |
| 20 | + | /// branches are a few commits ahead of where they left a default branch | |
| 21 | + | /// that has moved on a little; one left long ago needs the default | |
| 22 | + | /// branch's history further back; one far from both reads both deeply. | |
| 23 | + | /// Past the last, there is no answer. | |
| 24 | + | pub const DEPTHS: [(u32, u32); 4] = [(12, 120), (40, 120), (40, 1000), (1000, 1000)]; | |
| 25 | + | ||
| 26 | + | /// Branches read at once. | |
| 27 | + | const AT_ONCE: usize = 8; | |
| 28 | + | ||
| 29 | + | /// A branch head's commit and drift, and whether the answer may be kept: | |
| 30 | + | /// not when a read failed. | |
| 31 | + | #[derive(Clone, Debug)] | |
| 32 | + | pub struct Measured { | |
| 33 | + | pub commit: Option<Commit>, | |
| 34 | + | pub drift: Option<Drift>, | |
| 35 | + | pub settled: bool, | |
| 36 | + | } | |
| 37 | + | ||
| 38 | + | /// Commits `branch` has that `main` does not (ahead) and the other way | |
| 39 | + | /// round (behind), from what was read of the two histories (`commits`, in | |
| 40 | + | /// any order, repeats allowed). `None` when either head is missing, or when | |
| 41 | + | /// a commit only one side reaches has a parent that was not read: that | |
| 42 | + | /// parent's history could change either count. | |
| 43 | + | pub fn drift<'a>(branch: &str, main: &str, commits: impl IntoIterator<Item = &'a Commit>) -> Option<Drift> { | |
| 44 | + | let mut parents: HashMap<&str, &[String]> = HashMap::new(); | |
| 45 | + | for commit in commits { | |
| 46 | + | parents.insert(commit.hash.as_str(), commit.parents.as_slice()); | |
| 47 | + | } | |
| 48 | + | if !parents.contains_key(branch) || !parents.contains_key(main) { | |
| 49 | + | return None; | |
| 50 | + | } | |
| 51 | + | let from_branch = reach(branch, &parents); | |
| 52 | + | let from_main = reach(main, &parents); | |
| 53 | + | let (mut ahead, mut behind) = (0, 0); | |
| 54 | + | for (hash, above) in &parents { | |
| 55 | + | let on_branch = from_branch.contains(hash); | |
| 56 | + | if on_branch == from_main.contains(hash) { | |
| 57 | + | continue; | |
| 58 | + | } | |
| 59 | + | if above.iter().any(|parent| !parents.contains_key(parent.as_str())) { | |
| 60 | + | return None; | |
| 61 | + | } | |
| 62 | + | if on_branch { | |
| 63 | + | ahead += 1; | |
| 64 | + | } else { | |
| 65 | + | behind += 1; | |
| 66 | + | } | |
| 67 | + | } | |
| 68 | + | Some(Drift { ahead, behind }) | |
| 69 | + | } | |
| 70 | + | ||
| 71 | + | /// Every commit read that `head` descends from, itself included. | |
| 72 | + | fn reach<'a>(head: &'a str, parents: &HashMap<&'a str, &'a [String]>) -> HashSet<&'a str> { | |
| 73 | + | let mut seen = HashSet::from([head]); | |
| 74 | + | let mut next = vec![head]; | |
| 75 | + | while let Some(hash) = next.pop() { | |
| 76 | + | for parent in parents.get(hash).copied().unwrap_or_default() { | |
| 77 | + | if let Some((&known, _)) = parents.get_key_value(parent.as_str()) | |
| 78 | + | && seen.insert(known) | |
| 79 | + | { | |
| 80 | + | next.push(known); | |
| 81 | + | } | |
| 82 | + | } | |
| 83 | + | } | |
| 84 | + | seen | |
| 85 | + | } | |
| 86 | + | ||
| 87 | + | /// What was read of one branch's history so far. | |
| 88 | + | struct Reading { | |
| 89 | + | commits: Vec<Commit>, | |
| 90 | + | depth: u32, | |
| 91 | + | answer: Option<Measured>, | |
| 92 | + | } | |
| 93 | + | ||
| 94 | + | /// Each of `heads` measured against `base`, in the order given. | |
| 95 | + | pub async fn measure<R: GitRepo>(git: &R, base: &str, heads: &[String]) -> Vec<Measured> { | |
| 96 | + | let mut readings: Vec<Reading> = heads.iter().map(|_| Reading { commits: Vec::new(), depth: 0, answer: None }).collect(); | |
| 97 | + | // The default branch's history, read once per depth and not deeper | |
| 98 | + | // once a read reached its start. | |
| 99 | + | let mut main: Option<(u32, Vec<Commit>)> = None; | |
| 100 | + | let mut main_failed = false; | |
| 101 | + | for (branch_depth, main_depth) in DEPTHS { | |
| 102 | + | if readings.iter().all(|reading| reading.answer.is_some()) { | |
| 103 | + | break; | |
| 104 | + | } | |
| 105 | + | let deeper = match &main { | |
| 106 | + | Some((read_to, commits)) => *read_to < main_depth && commits.len() as u32 >= *read_to, | |
| 107 | + | None => true, | |
| 108 | + | }; | |
| 109 | + | if deeper && !main_failed { | |
| 110 | + | match git.log(base, main_depth).await { | |
| 111 | + | Ok(commits) if !commits.is_empty() => main = Some((main_depth, commits)), | |
| 112 | + | _ => main_failed = true, | |
| 113 | + | } | |
| 114 | + | } | |
| 115 | + | let Some((_, main_commits)) = &main else { | |
| 116 | + | for reading in readings.iter_mut().filter(|reading| reading.answer.is_none()) { | |
| 117 | + | reading.answer = Some(Measured { commit: None, drift: None, settled: false }); | |
| 118 | + | } | |
| 119 | + | break; | |
| 120 | + | }; | |
| 121 | + | let open: Vec<usize> = (0..heads.len()).filter(|&index| readings[index].answer.is_none()).collect(); | |
| 122 | + | for chunk in open.chunks(AT_ONCE) { | |
| 123 | + | let reads = futures_util::future::join_all(chunk.iter().map(|&index| { | |
| 124 | + | let reading = &readings[index]; | |
| 125 | + | // Read again only when deeper, and only when the last read | |
| 126 | + | // did not already reach the start. | |
| 127 | + | let again = reading.depth == 0 || (branch_depth > reading.depth && reading.commits.len() as u32 >= reading.depth); | |
| 128 | + | let head = heads[index].as_str(); | |
| 129 | + | async move { if again { Some(git.log(head, branch_depth).await) } else { None } } | |
| 130 | + | })) | |
| 131 | + | .await; | |
| 132 | + | for (&index, read) in chunk.iter().zip(reads) { | |
| 133 | + | let reading = &mut readings[index]; | |
| 134 | + | match read { | |
| 135 | + | Some(Ok(commits)) if !commits.is_empty() => { | |
| 136 | + | reading.commits = commits; | |
| 137 | + | reading.depth = branch_depth; | |
| 138 | + | } | |
| 139 | + | Some(_) => { | |
| 140 | + | reading.answer = Some(Measured { commit: None, drift: None, settled: false }); | |
| 141 | + | continue; | |
| 142 | + | } | |
| 143 | + | None => {} | |
| 144 | + | } | |
| 145 | + | let head = heads[index].as_str(); | |
| 146 | + | if let Some(counted) = drift(head, base, main_commits.iter().chain(reading.commits.iter())) { | |
| 147 | + | reading.answer = Some(Measured { commit: reading.commits.first().cloned(), drift: Some(counted), settled: true }); | |
| 148 | + | } | |
| 149 | + | } | |
| 150 | + | } | |
| 151 | + | if main_failed { | |
| 152 | + | break; | |
| 153 | + | } | |
| 154 | + | } | |
| 155 | + | readings | |
| 156 | + | .into_iter() | |
| 157 | + | .map(|reading| { | |
| 158 | + | reading.answer.unwrap_or_else(|| Measured { | |
| 159 | + | commit: reading.commits.first().cloned(), | |
| 160 | + | drift: None, | |
| 161 | + | // Read to the last depth without meeting: that is the answer | |
| 162 | + | // for this pair, and it will not change. | |
| 163 | + | settled: !main_failed && reading.depth > 0, | |
| 164 | + | }) | |
| 165 | + | }) | |
| 166 | + | .collect() | |
| 167 | + | } | |
| 168 | + | ||
| 169 | + | #[cfg(test)] | |
| 170 | + | mod tests { | |
| 171 | + | use std::cell::RefCell; | |
| 172 | + | use std::future::Future; | |
| 173 | + | use std::pin::pin; | |
| 174 | + | use std::task::{Context, Poll, Waker}; | |
| 175 | + | ||
| 176 | + | use g1t_contracts::repos::{Branch, GitAccess, Signature, TreeEntry}; | |
| 177 | + | ||
| 178 | + | use super::*; | |
| 179 | + | use crate::store::Scope; | |
| 180 | + | ||
| 181 | + | fn run<F: Future>(future: F) -> F::Output { | |
| 182 | + | match pin!(future).as_mut().poll(&mut Context::from_waker(Waker::noop())) { | |
| 183 | + | Poll::Ready(output) => output, | |
| 184 | + | Poll::Pending => panic!("the fake store never waits"), | |
| 185 | + | } | |
| 186 | + | } | |
| 187 | + | ||
| 188 | + | fn commit(hash: &str, parents: &[&str]) -> Commit { | |
| 189 | + | Commit { | |
| 190 | + | hash: hash.into(), | |
| 191 | + | tree_hash: format!("t{hash}"), | |
| 192 | + | message: format!("commit {hash}\n\nbody"), | |
| 193 | + | author: Signature { name: "a".into(), email: "a@example.com".into() }, | |
| 194 | + | parents: parents.iter().map(|&p| p.to_owned()).collect(), | |
| 195 | + | authored_at: String::new(), | |
| 196 | + | } | |
| 197 | + | } | |
| 198 | + | ||
| 199 | + | /// Commits by hash; `log` follows first parents. Counts each read. | |
| 200 | + | #[derive(Default)] | |
| 201 | + | struct Fake { | |
| 202 | + | commits: HashMap<String, Commit>, | |
| 203 | + | reads: RefCell<Vec<(String, u32)>>, | |
| 204 | + | fail: Option<String>, | |
| 205 | + | } | |
| 206 | + | ||
| 207 | + | impl Fake { | |
| 208 | + | fn with(commits: Vec<Commit>) -> Fake { | |
| 209 | + | Fake { commits: commits.into_iter().map(|c| (c.hash.clone(), c)).collect(), ..Fake::default() } | |
| 210 | + | } | |
| 211 | + | } | |
| 212 | + | ||
| 213 | + | impl GitRepo for Fake { | |
| 214 | + | async fn access(&self, _scope: Scope) -> Result<GitAccess> { | |
| 215 | + | unimplemented!() | |
| 216 | + | } | |
| 217 | + | async fn branches(&self) -> Result<Vec<Branch>> { | |
| 218 | + | Ok(Vec::new()) | |
| 219 | + | } | |
| 220 | + | async fn log(&self, git_ref: &str, limit: u32) -> Result<Vec<Commit>> { | |
| 221 | + | self.reads.borrow_mut().push((git_ref.to_owned(), limit)); | |
| 222 | + | if self.fail.as_deref() == Some(git_ref) { | |
| 223 | + | return Err(worker::Error::RustError("store busy".into())); | |
| 224 | + | } | |
| 225 | + | let mut out = Vec::new(); | |
| 226 | + | let mut at = self.commits.get(git_ref); | |
| 227 | + | while let Some(commit) = at { | |
| 228 | + | if out.len() as u32 >= limit { | |
| 229 | + | break; | |
| 230 | + | } | |
| 231 | + | out.push(commit.clone()); | |
| 232 | + | at = commit.parents.first().and_then(|parent| self.commits.get(parent)); | |
| 233 | + | } | |
| 234 | + | Ok(out) | |
| 235 | + | } | |
| 236 | + | async fn parents(&self, _commit_hash: &str) -> Result<Option<Vec<String>>> { | |
| 237 | + | Ok(None) | |
| 238 | + | } | |
| 239 | + | async fn read_tree(&self, _tree_hash: &str) -> Result<Option<Vec<TreeEntry>>> { | |
| 240 | + | Ok(None) | |
| 241 | + | } | |
| 242 | + | async fn read_blob(&self, _blob_hash: &str) -> Result<Option<Vec<u8>>> { | |
| 243 | + | Ok(None) | |
| 244 | + | } | |
| 245 | + | async fn read_file(&self, _git_ref: &str, _path: &str) -> Result<Option<Vec<u8>>> { | |
| 246 | + | Ok(None) | |
| 247 | + | } | |
| 248 | + | async fn fork(&self, _target_key: &str) -> Result<()> { | |
| 249 | + | Ok(()) | |
| 250 | + | } | |
| 251 | + | } | |
| 252 | + | ||
| 253 | + | /// m1 ← m2 ← m3 on main; b1 ← b2 branched from m2. | |
| 254 | + | fn forked() -> Vec<Commit> { | |
| 255 | + | vec![commit("m1", &[]), commit("m2", &["m1"]), commit("m3", &["m2"]), commit("b1", &["m2"]), commit("b2", &["b1"])] | |
| 256 | + | } | |
| 257 | + | ||
| 258 | + | /// A straight line of `n` commits named `{prefix}{i}`, the first on `from`. | |
| 259 | + | fn line(prefix: &str, from: Option<&str>, n: usize) -> Vec<Commit> { | |
| 260 | + | (1..=n) | |
| 261 | + | .map(|i| { | |
| 262 | + | let parent = if i == 1 { from.map(str::to_owned) } else { Some(format!("{prefix}{}", i - 1)) }; | |
| 263 | + | commit(&format!("{prefix}{i}"), &parent.iter().map(String::as_str).collect::<Vec<_>>()) | |
| 264 | + | }) | |
| 265 | + | .collect() | |
| 266 | + | } | |
| 267 | + | ||
| 268 | + | #[test] | |
| 269 | + | fn counts_both_sides_from_where_they_forked() { | |
| 270 | + | assert_eq!(drift("b2", "m3", &forked()), Some(Drift { ahead: 2, behind: 1 })); | |
| 271 | + | assert_eq!(drift("m3", "m3", &forked()), Some(Drift { ahead: 0, behind: 0 })); | |
| 272 | + | } | |
| 273 | + | ||
| 274 | + | #[test] | |
| 275 | + | fn a_merge_from_main_is_not_ahead() { | |
| 276 | + | // b3 merges m3 into the branch. | |
| 277 | + | let mut history = forked(); | |
| 278 | + | history.push(commit("b3", &["b2", "m3"])); | |
| 279 | + | assert_eq!(drift("b3", "m3", &history), Some(Drift { ahead: 3, behind: 0 })); | |
| 280 | + | } | |
| 281 | + | ||
| 282 | + | #[test] | |
| 283 | + | fn no_answer_when_the_histories_were_not_read_far_enough() { | |
| 284 | + | let history = vec![commit("m3", &["m2"]), commit("b2", &["b1"])]; | |
| 285 | + | assert_eq!(drift("b2", "m3", &history), None); | |
| 286 | + | assert_eq!(drift("b9", "m3", &forked()), None); | |
| 287 | + | } | |
| 288 | + | ||
| 289 | + | #[test] | |
| 290 | + | fn measures_every_head_with_one_read_of_main() { | |
| 291 | + | let git = Fake::with(forked()); | |
| 292 | + | let found = run(measure(&git, "m3", &["b2".to_owned(), "m2".to_owned(), "m3".to_owned()])); | |
| 293 | + | assert_eq!(found[0].drift, Some(Drift { ahead: 2, behind: 1 })); | |
| 294 | + | assert_eq!(found[0].commit.as_ref().map(|c| c.hash.as_str()), Some("b2")); | |
| 295 | + | assert_eq!(found[1].drift, Some(Drift { ahead: 0, behind: 1 })); | |
| 296 | + | assert_eq!(found[2].drift, Some(Drift { ahead: 0, behind: 0 })); | |
| 297 | + | assert!(found.iter().all(|m| m.settled)); | |
| 298 | + | let reads = git.reads.borrow(); | |
| 299 | + | assert_eq!(reads.iter().filter(|(hash, _)| hash == "m3").count(), 2, "main once, plus m3 as a head: {reads:?}"); | |
| 300 | + | assert!(reads.iter().all(|(_, depth)| *depth == 12 || *depth == 120)); | |
| 301 | + | } | |
| 302 | + | ||
| 303 | + | #[test] | |
| 304 | + | fn reads_deeper_only_for_a_branch_that_needs_it() { | |
| 305 | + | // main: 150 commits; "long" is 30 ahead of m100 (50 behind); "short" is 1 ahead of m149. | |
| 306 | + | let mut history = line("m", None, 150); | |
| 307 | + | history.extend(line("l", Some("m100"), 30)); | |
| 308 | + | history.push(commit("s1", &["m149"])); | |
| 309 | + | let git = Fake::with(history); | |
| 310 | + | let found = run(measure(&git, "m150", &["l30".to_owned(), "s1".to_owned()])); | |
| 311 | + | assert_eq!(found[0].drift, Some(Drift { ahead: 30, behind: 50 })); | |
| 312 | + | assert_eq!(found[1].drift, Some(Drift { ahead: 1, behind: 1 })); | |
| 313 | + | let reads = git.reads.borrow(); | |
| 314 | + | assert_eq!(reads.iter().filter(|(hash, _)| hash == "s1").count(), 1); | |
| 315 | + | assert_eq!(*reads.iter().filter(|(hash, _)| hash == "l30").map(|(_, depth)| depth).max().unwrap(), 40); | |
| 316 | + | assert!(!reads.iter().any(|(_, depth)| *depth == 1000), "{reads:?}"); | |
| 317 | + | } | |
| 318 | + | ||
| 319 | + | #[test] | |
| 320 | + | fn unrelated_histories_read_to_their_start_count_every_commit() { | |
| 321 | + | let mut history = line("m", None, 3); | |
| 322 | + | history.extend(line("x", None, 2)); | |
| 323 | + | let git = Fake::with(history); | |
| 324 | + | let found = run(measure(&git, "m3", &["x2".to_owned()])); | |
| 325 | + | assert_eq!(found[0].drift, Some(Drift { ahead: 2, behind: 3 })); | |
| 326 | + | assert_eq!(found[0].commit.as_ref().map(|c| c.hash.as_str()), Some("x2")); | |
| 327 | + | assert!(found[0].settled); | |
| 328 | + | // Both reached their start at the first depth: never read again. | |
| 329 | + | assert_eq!(git.reads.borrow().len(), 2); | |
| 330 | + | } | |
| 331 | + | ||
| 332 | + | #[test] | |
| 333 | + | fn past_the_last_depth_the_answer_is_settled_without_a_count() { | |
| 334 | + | // The branch is 1,200 commits long: no depth reaches where it left main. | |
| 335 | + | let mut history = line("m", None, 3); | |
| 336 | + | history.extend(line("x", Some("m1"), 1200)); | |
| 337 | + | let git = Fake::with(history); | |
| 338 | + | let found = run(measure(&git, "m3", &["x1200".to_owned()])); | |
| 339 | + | assert_eq!(found[0].drift, None); | |
| 340 | + | assert_eq!(found[0].commit.as_ref().map(|c| c.hash.as_str()), Some("x1200")); | |
| 341 | + | assert!(found[0].settled); | |
| 342 | + | } | |
| 343 | + | ||
| 344 | + | #[test] | |
| 345 | + | fn a_failed_read_is_not_kept() { | |
| 346 | + | let mut git = Fake::with(forked()); | |
| 347 | + | git.fail = Some("b2".into()); | |
| 348 | + | let found = run(measure(&git, "m3", &["b2".to_owned(), "b1".to_owned()])); | |
| 349 | + | assert!(!found[0].settled); | |
| 350 | + | assert_eq!(found[1].drift, Some(Drift { ahead: 1, behind: 1 })); | |
| 351 | + | assert!(found[1].settled); | |
| 352 | + | git.fail = Some("m3".into()); | |
| 353 | + | let found = run(measure(&git, "m3", &["b2".to_owned()])); | |
| 354 | + | assert!(!found[0].settled); | |
| 355 | + | } | |
| 356 | + | } |
| 13 | 13 | mod commit_file; | |
| 14 | 14 | mod contributors; | |
| 15 | 15 | mod diff; | |
| 16 | + | mod drift; | |
| 16 | 17 | mod fallback; | |
| 17 | 18 | mod forks; | |
| 18 | 19 | mod git_http; | |
| ⋯ | |||
| 74 | 75 | const MAX_ANCESTRY: u32 = 1000; | |
| 75 | 76 | /// The most tags a repository's Tags page reads and lists. | |
| 76 | 77 | const MAX_TAGS_READ: usize = 100; | |
| 78 | + | /// Branch heads measured in one `branch_drift` call. | |
| 79 | + | const MAX_DRIFT_HEADS: usize = 100; | |
| 77 | 80 | ||
| 78 | 81 | /// One path segment, percent-encoded for a cache key. | |
| 79 | 82 | fn urlencoding_segment(segment: &str) -> String { | |
| ⋯ | |||
| 941 | 944 | Ok(Outcome::Ok(found)) | |
| 942 | 945 | } | |
| 943 | 946 | ||
| 947 | + | /// How far each branch head has moved from the default branch's head, | |
| 948 | + | /// in one call (drift.rs). Each answer is kept in this colo's cache by | |
| 949 | + | /// repository and the pair of hashes, for good: neither history can | |
| 950 | + | /// change. A head that moved is the only one walked. | |
| 951 | + | async fn branch_drift(&self, a: BranchDriftArgs) -> Result<Outcome<BranchDrifts>> { | |
| 952 | + | let Some(repo) = self.readable(&a.path, &a.viewer).await? else { | |
| 953 | + | return Ok(not_found()); | |
| 954 | + | }; | |
| 955 | + | if !store::is_commit_hash(&a.base) { | |
| 956 | + | return Ok(Outcome::fail(FailureCode::Invalid, "The default branch's head is a full commit hash.")); | |
| 957 | + | } | |
| 958 | + | let heads: Vec<String> = a.heads.into_iter().take(MAX_DRIFT_HEADS).collect(); | |
| 959 | + | let git = self.read_git(&repo).await?; | |
| 960 | + | let key = |head: &str| format!("https://drift.g1t.internal/{}/{}/{head}", repo.id, a.base); | |
| 961 | + | let cache = worker::Cache::default(); | |
| 962 | + | let (base, kept) = futures_util::future::join( | |
| 963 | + | git.log(&a.base, 1), | |
| 964 | + | futures_util::future::join_all(heads.iter().map(|head| { | |
| 965 | + | let (cache, url) = (&cache, key(head)); | |
| 966 | + | async move { | |
| 967 | + | if !store::is_commit_hash(head) { | |
| 968 | + | return None; | |
| 969 | + | } | |
| 970 | + | let mut found = cache.get(url.as_str(), false).await.ok()??; | |
| 971 | + | found.json::<BranchDrift>().await.ok() | |
| 972 | + | } | |
| 973 | + | })), | |
| 974 | + | ) | |
| 975 | + | .await; | |
| 976 | + | let missing: Vec<String> = heads | |
| 977 | + | .iter() | |
| 978 | + | .zip(&kept) | |
| 979 | + | .filter(|(head, kept)| kept.is_none() && store::is_commit_hash(head)) | |
| 980 | + | .map(|(head, _)| head.clone()) | |
| 981 | + | .collect(); | |
| 982 | + | let measured = drift::measure(&git, &a.base, &missing).await; | |
| 983 | + | let mut fresh: HashMap<String, BranchDrift> = HashMap::new(); | |
| 984 | + | for (head, found) in missing.into_iter().zip(measured) { | |
| 985 | + | let answer = BranchDrift { head: head.clone(), commit: found.commit, drift: found.drift }; | |
| 986 | + | if found.settled | |
| 987 | + | && let Ok(mut response) = worker::Response::from_json(&answer) | |
| 988 | + | { | |
| 989 | + | let _ = response.headers_mut().set("cache-control", "public, max-age=31536000, immutable"); | |
| 990 | + | let _ = cache.put(key(&head).as_str(), response).await; | |
| 991 | + | } | |
| 992 | + | fresh.insert(head, answer); | |
| 993 | + | } | |
| 994 | + | let branches = heads | |
| 995 | + | .iter() | |
| 996 | + | .zip(kept) | |
| 997 | + | .map(|(head, kept)| { | |
| 998 | + | kept.or_else(|| fresh.get(head).cloned()) | |
| 999 | + | .unwrap_or_else(|| BranchDrift { head: head.clone(), commit: None, drift: None }) | |
| 1000 | + | }) | |
| 1001 | + | .collect(); | |
| 1002 | + | Ok(Outcome::Ok(BranchDrifts { base: base.ok().and_then(|log| log.into_iter().next()), branches })) | |
| 1003 | + | } | |
| 1004 | + | ||
| 944 | 1005 | /// The repository's tags, newest commit first, at most 100. | |
| 945 | 1006 | async fn tags(&self, a: g1t_contracts::repos::TagsArgs) -> Result<Outcome<Vec<g1t_contracts::repos::Tag>>> { | |
| 946 | 1007 | let Some(repo) = self.readable(&a.path, &a.viewer).await? else { | |
| 947 | 1008 | return Ok(not_found()); | |
| 948 | 1009 | }; | |
| 949 | − | let git = self.store.open(&store_key(&repo)).await?; | |
| 1010 | + | // Kept until the refs move, as the branch list is (store.rs): listing | |
| 1011 | + | // the refs is a round trip to the store on every call otherwise. | |
| 1012 | + | let version = refs_cache::usable(registry::refs_state(&repo.id), now_ms()).filter(|_| !self.store.on_fallback(&store_key(&repo))); | |
| 1013 | + | let kept_at = version.map(|version| format!("https://tags.g1t.internal/{}/{version}", repo.id)); | |
| 1014 | + | if let Some(url) = &kept_at | |
| 1015 | + | && let Ok(Some(mut kept)) = worker::Cache::default().get(url.as_str(), false).await | |
| 1016 | + | && let Ok(tags) = kept.json::<Vec<g1t_contracts::repos::Tag>>().await | |
| 1017 | + | { | |
| 1018 | + | return Ok(Outcome::Ok(tags)); | |
| 1019 | + | } | |
| 1020 | + | let (tags, complete) = self.read_tags(&repo).await?; | |
| 1021 | + | if complete | |
| 1022 | + | && let Some(url) = &kept_at | |
| 1023 | + | && let Ok(mut response) = worker::Response::from_json(&tags) | |
| 1024 | + | { | |
| 1025 | + | let _ = response.headers_mut().set("cache-control", "public, max-age=300"); | |
| 1026 | + | let _ = worker::Cache::default().put(url.as_str(), response).await; | |
| 1027 | + | } | |
| 1028 | + | Ok(Outcome::Ok(tags)) | |
| 1029 | + | } | |
| 1030 | + | ||
| 1031 | + | /// The tags, and whether every one's commit was read (only then kept). | |
| 1032 | + | async fn read_tags(&self, repo: &Repo) -> Result<(Vec<g1t_contracts::repos::Tag>, bool)> { | |
| 1033 | + | let git = self.store.open(&store_key(repo)).await?; | |
| 950 | 1034 | let access = git.access(Scope::Read).await?; | |
| 951 | 1035 | let named: Vec<(String, String)> = refs::heads_and_tags(refs::all(&access).await?) | |
| 952 | 1036 | .into_iter() | |
| 953 | 1037 | .filter_map(|(name, hash)| name.strip_prefix("refs/tags/").map(|tag| (tag.to_owned(), hash))) | |
| 954 | 1038 | .collect(); | |
| 955 | − | let read = self.read_git(&repo).await?; | |
| 1039 | + | let read = self.read_git(repo).await?; | |
| 956 | 1040 | let commits = futures_util::future::join_all(named.iter().take(MAX_TAGS_READ).map(|(_, hash)| read.log(hash, 1))).await; | |
| 1041 | + | let complete = commits.iter().all(Result::is_ok); | |
| 957 | 1042 | let mut tags: Vec<g1t_contracts::repos::Tag> = named | |
| 958 | 1043 | .into_iter() | |
| 959 | 1044 | .zip(commits.into_iter().map(|found| found.ok().and_then(|list| list.into_iter().next())).chain(std::iter::repeat(None))) | |
| ⋯ | |||
| 964 | 1049 | at(b).cmp(&at(a)).then_with(|| b.name.cmp(&a.name)) | |
| 965 | 1050 | }); | |
| 966 | 1051 | tags.truncate(MAX_TAGS_READ); | |
| 967 | − | Ok(Outcome::Ok(tags)) | |
| 1052 | + | Ok((tags, complete)) | |
| 968 | 1053 | } | |
| 969 | 1054 | ||
| 970 | 1055 | /// The repository's branches, default branch first. | |
| ⋯ | |||
| 2238 | 2323 | ||
| 2239 | 2324 | /// Read methods whose answer is an `Outcome`: when the git store is busy, | |
| 2240 | 2325 | /// the site is told so in words instead of failing the page. | |
| 2241 | − | const OUTCOME_READS: [&str; 6] = ["tree", "blob", "log", "branches", "blame", "compare"]; | |
| 2326 | + | const OUTCOME_READS: [&str; 7] = ["tree", "blob", "log", "branches", "blame", "compare", "branch_drift"]; | |
| 2242 | 2327 | ||
| 2243 | 2328 | #[event(fetch)] | |
| 2244 | 2329 | async fn fetch(mut request: Request, env: Env, ctx: Context) -> Result<Response> { | |
| ⋯ | |||
| 2336 | 2421 | "git_access" => reply(&repos.git_access(args(body)?).await?), | |
| 2337 | 2422 | "branches" => reply(&repos.branches(args(body)?).await?), | |
| 2338 | 2423 | "last_commits" => reply(&repos.last_commits(args(body)?).await?), | |
| 2424 | + | "branch_drift" => reply(&repos.branch_drift(args(body)?).await?), | |
| 2339 | 2425 | "tags" => reply(&repos.tags(args(body)?).await?), | |
| 2340 | 2426 | // The About: what the Files page shows beside the files (about.rs). | |
| 2341 | 2427 | // What is kept behind the head is worked out again after the answer. | |
| 733 | 733 | /// the store would not be. | |
| 734 | 734 | const ABSENT_MAX_AGE: &str = "public, max-age=600"; | |
| 735 | 735 | ||
| 736 | + | /// Histories by hash at least this long are put together from a short | |
| 737 | + | /// read and one kept before, where they can be (`spliced_log`). | |
| 738 | + | const SPLICE_FROM: u32 = 100; | |
| 739 | + | /// How many commits that short read takes. | |
| 740 | + | const SPLICE_PROBE: u32 = 16; | |
| 741 | + | ||
| 742 | + | /// `short[..at]` followed by `kept` (the history from `short[at]`), cut | |
| 743 | + | /// to `limit` commits. | |
| 744 | + | pub fn splice(short: &[Commit], at: usize, kept: Vec<Commit>, limit: u32) -> Vec<Commit> { | |
| 745 | + | let mut out: Vec<Commit> = short[..at.min(short.len())].to_vec(); | |
| 746 | + | out.extend(kept); | |
| 747 | + | out.truncate(limit as usize); | |
| 748 | + | out | |
| 749 | + | } | |
| 750 | + | ||
| 736 | 751 | /// Whether a ref is a full commit hash (SHA-1 or SHA-256), whose history | |
| 737 | 752 | /// can be kept for good. | |
| 738 | 753 | pub fn is_commit_hash(git_ref: &str) -> bool { | |
| ⋯ | |||
| 843 | 858 | found | |
| 844 | 859 | } | |
| 845 | 860 | ||
| 861 | + | /// A history from the store itself. | |
| 862 | + | async fn read_log(&self, git_ref: &str, limit: u32) -> Result<Vec<Commit>> { | |
| 863 | + | let options = js::to_js(&serde_json::json!({ "ref": git_ref, "limit": limit }))?; | |
| 864 | + | let raw: Vec<RawCommit> = js::from_js(&self.call("log", &[options], true).await?)?; | |
| 865 | + | Ok(raw | |
| 866 | + | .into_iter() | |
| 867 | + | .map(|commit| Commit { | |
| 868 | + | hash: commit.hash, | |
| 869 | + | tree_hash: commit.tree_hash, | |
| 870 | + | message: commit.message, | |
| 871 | + | author: commit.author, | |
| 872 | + | parents: commit.parents, | |
| 873 | + | authored_at: rfc3339(commit.authored_at * 1000), | |
| 874 | + | }) | |
| 875 | + | .collect()) | |
| 876 | + | } | |
| 877 | + | ||
| 878 | + | /// A long history by commit hash, from a short read and one kept | |
| 879 | + | /// before: when a branch moves a few commits, the history from its new | |
| 880 | + | /// head is those commits, then the history kept from its old one (the | |
| 881 | + | /// first-parent chain from a commit never changes). Only when none of | |
| 882 | + | /// the short read's commits has the same history kept is the whole of | |
| 883 | + | /// it read. A default branch that moved by a merge then costs a read | |
| 884 | + | /// of [`SPLICE_PROBE`] commits instead of a thousand. | |
| 885 | + | async fn spliced_log(&self, hash: &str, limit: u32) -> Result<Vec<Commit>> { | |
| 886 | + | let short = self.read_log(hash, SPLICE_PROBE).await?; | |
| 887 | + | if (short.len() as u32) < SPLICE_PROBE { | |
| 888 | + | // The whole history fits in the short read. | |
| 889 | + | return Ok(short); | |
| 890 | + | } | |
| 891 | + | let kept = futures_util::future::join_all(short.iter().enumerate().skip(1).map(|(at, commit)| async move { | |
| 892 | + | let Some(CacheKey::Forever(path)) = log_key(&commit.hash, limit, None) else { | |
| 893 | + | return None; | |
| 894 | + | }; | |
| 895 | + | let bytes = self.peek(&path).await?; | |
| 896 | + | serde_json::from_slice::<Vec<Commit>>(&bytes).ok().map(|kept| (at, kept)) | |
| 897 | + | })) | |
| 898 | + | .await; | |
| 899 | + | match kept.into_iter().flatten().next() { | |
| 900 | + | Some((at, kept)) => Ok(splice(&short, at, kept, limit)), | |
| 901 | + | None => self.read_log(hash, limit).await, | |
| 902 | + | } | |
| 903 | + | } | |
| 904 | + | ||
| 905 | + | /// A kept answer, if there is one, without counting a miss: the | |
| 906 | + | /// splice looks for many and expects most to be absent. | |
| 907 | + | async fn peek(&self, path: &str) -> Option<Vec<u8>> { | |
| 908 | + | let url = self.cache_url(path); | |
| 909 | + | if let Some(bytes) = MEMORY.with(|memory| memory.borrow().get(&url)) { | |
| 910 | + | meters::record("cache.memory_hit", &self.key, 0, bytes.len() as u64); | |
| 911 | + | return Some(bytes); | |
| 912 | + | } | |
| 913 | + | let mut response = worker::Cache::default().get(url.clone(), false).await.ok()??; | |
| 914 | + | let bytes = response.bytes().await.ok()?; | |
| 915 | + | meters::record("cache.edge_hit", &self.key, 0, bytes.len() as u64); | |
| 916 | + | MEMORY.with(|memory| memory.borrow_mut().put(url, &bytes)); | |
| 917 | + | Some(bytes) | |
| 918 | + | } | |
| 919 | + | ||
| 846 | 920 | async fn cached(&self, kind: &str, hash: &str) -> Option<Vec<u8>> { | |
| 847 | 921 | self.cached_at(&format!("{kind}/{hash}"), true).await | |
| 848 | 922 | } | |
| ⋯ | |||
| 1033 | 1107 | { | |
| 1034 | 1108 | return Ok(commits); | |
| 1035 | 1109 | } | |
| 1036 | − | let options = js::to_js(&serde_json::json!({ "ref": git_ref, "limit": limit }))?; | |
| 1037 | − | let raw: Vec<RawCommit> = js::from_js(&self.call("log", &[options], true).await?)?; | |
| 1038 | − | let commits: Vec<Commit> = raw | |
| 1039 | − | .into_iter() | |
| 1040 | − | .map(|commit| Commit { | |
| 1041 | − | hash: commit.hash, | |
| 1042 | − | tree_hash: commit.tree_hash, | |
| 1043 | − | message: commit.message, | |
| 1044 | − | author: commit.author, | |
| 1045 | − | parents: commit.parents, | |
| 1046 | − | authored_at: rfc3339(commit.authored_at * 1000), | |
| 1047 | − | }) | |
| 1048 | − | .collect(); | |
| 1110 | + | let commits = match &key { | |
| 1111 | + | Some(CacheKey::Forever(_)) if limit >= SPLICE_FROM => self.spliced_log(git_ref, limit).await?, | |
| 1112 | + | _ => self.read_log(git_ref, limit).await?, | |
| 1113 | + | }; | |
| 1049 | 1114 | // An unknown ref logs nothing; that is not kept, in case it arrives. | |
| 1050 | 1115 | if !commits.is_empty() | |
| 1051 | 1116 | && let Ok(bytes) = serde_json::to_vec(&commits) | |
| ⋯ | |||
| 1207 | 1272 | mod tests { | |
| 1208 | 1273 | use super::*; | |
| 1209 | 1274 | ||
| 1275 | + | fn chain(names: &[&str]) -> Vec<Commit> { | |
| 1276 | + | names | |
| 1277 | + | .iter() | |
| 1278 | + | .enumerate() | |
| 1279 | + | .map(|(at, name)| Commit { | |
| 1280 | + | hash: (*name).to_owned(), | |
| 1281 | + | tree_hash: String::new(), | |
| 1282 | + | message: String::new(), | |
| 1283 | + | author: g1t_contracts::repos::Signature { name: "a".into(), email: "a@example.com".into() }, | |
| 1284 | + | parents: names.get(at + 1).map(|parent| vec![(*parent).to_owned()]).unwrap_or_default(), | |
| 1285 | + | authored_at: String::new(), | |
| 1286 | + | }) | |
| 1287 | + | .collect() | |
| 1288 | + | } | |
| 1289 | + | ||
| 1290 | + | fn hashes(commits: &[Commit]) -> Vec<&str> { | |
| 1291 | + | commits.iter().map(|commit| commit.hash.as_str()).collect() | |
| 1292 | + | } | |
| 1293 | + | ||
| 1294 | + | #[test] | |
| 1295 | + | fn a_history_is_the_new_commits_then_the_one_kept_from_an_old_head() { | |
| 1296 | + | // The branch moved from c3 to c5; c3's history (limit 4) was kept. | |
| 1297 | + | let short = chain(&["c5", "c4", "c3", "c2"]); | |
| 1298 | + | let kept = chain(&["c3", "c2", "c1", "c0"]); | |
| 1299 | + | assert_eq!(hashes(&splice(&short, 2, kept.clone(), 4)), ["c5", "c4", "c3", "c2"]); | |
| 1300 | + | assert_eq!(hashes(&splice(&short, 2, kept.clone(), 6)), ["c5", "c4", "c3", "c2", "c1", "c0"]); | |
| 1301 | + | // A kept history that reached the first commit ends there. | |
| 1302 | + | assert_eq!(hashes(&splice(&short, 2, chain(&["c3", "c2"]), 10)), ["c5", "c4", "c3", "c2"]); | |
| 1303 | + | } | |
| 1304 | + | ||
| 1305 | + | #[test] | |
| 1306 | + | fn only_long_histories_by_hash_are_spliced() { | |
| 1307 | + | assert!(SPLICE_PROBE < SPLICE_FROM); | |
| 1308 | + | let hash = "a".repeat(40); | |
| 1309 | + | assert!(matches!(log_key(&hash, SPLICE_FROM, None), Some(CacheKey::Forever(_)))); | |
| 1310 | + | assert!(matches!(log_key("main", SPLICE_FROM, Some(1)), Some(CacheKey::Versioned(_)))); | |
| 1311 | + | } | |
| 1312 | + | ||
| 1210 | 1313 | fn access(token: &str) -> GitAccess { | |
| 1211 | 1314 | GitAccess { | |
| 1212 | 1315 | remote: "https://store.example/acme--rocket.git".to_owned(), | |