Overview: Active branches in one branch_drift call to repos, kept by head pair
The project overview's Active branches cost up to twenty repos log calls per view (a branch per depth, the default branch, its head again), each with its own access check and store handle. Every push or merge started the walks over in every data centre, and walks past the page's 3.5 s wait were dropped with the request, so they were never kept. A crawler's uncached render measured 3,606 ms with 25 repos calls, 17.9 s inside. - repos `branch_drift` (drift.rs): one access check and store handle, the default branch's history read once per depth for every branch, depths 12/120 first, each answer kept in the colo cache by the pair of hashes; also returns the default branch's head commit. - store.rs: a log by hash of 100+ commits is a 16-commit read spliced onto the log kept from one of those commits (a merge to main costs 16 commits, not 120 or 1,000). - tags kept under the refs version, as the branch list is. - Site: the drift call runs in waitUntil so its answer is kept even when the page stops waiting; crawlers wait 0.7 s for it; the project is looked up once for layout and overview; tags, commit_checks, shortcuts and the Files page's reads join READS (signed-in overviews were counted as writes). - docs/PERFORMANCE.md: before/after, and how to force a cache miss. Deploy repos before the site: until then branch_drift is unknown and Active branches show without commits or counts.
| 14 | 14 | } from "@g1t/contracts"; | |
| 15 | 15 | ||
| 16 | 16 | import type { ViewerAccess } from "./access"; | |
| 17 | − | import { repos, work } from "./services.server"; | |
| 17 | + | import { projects, repos, work } from "./services.server"; | |
| 18 | 18 | import { getViewer } from "./session.server"; | |
| 19 | 19 | ||
| 20 | 20 | type Context = Parameters<typeof getViewer>[0]; | |
| ⋯ | |||
| 68 | 68 | return found; | |
| 69 | 69 | } | |
| 70 | 70 | ||
| 71 | + | /** | |
| 72 | + | * One lookup of a project per request: the project's layout and its | |
| 73 | + | * overview load at the same time and both need it. | |
| 74 | + | */ | |
| 75 | + | const projectLookups = new WeakMap<object, Map<string, ReturnType<typeof projects.get>>>(); | |
| 76 | + | ||
| 77 | + | export function projectFor(context: Context, params: RepoParams): ReturnType<typeof projects.get> { | |
| 78 | + | const key = `${params.owner}/${params.repo}`.toLowerCase(); | |
| 79 | + | let seen = projectLookups.get(context); | |
| 80 | + | if (!seen) projectLookups.set(context, (seen = new Map())); | |
| 81 | + | let found = seen.get(key); | |
| 82 | + | if (!found) { | |
| 83 | + | found = projects.get(params.owner ?? "", params.repo ?? "", getViewer(context)); | |
| 84 | + | seen.set(key, found); | |
| 85 | + | } | |
| 86 | + | return found; | |
| 87 | + | } | |
| 88 | + | ||
| 71 | 89 | /** The viewer's role on a repository and what it lets them do. */ | |
| 72 | 90 | export function accessFor(viewer: Viewer, repo: Repo): ViewerAccess { | |
| 73 | 91 | return { | |
| 4 | 4 | * it with its checks, and its preview. The overview shows the newest few; | |
| 5 | 5 | * the Branches page shows them all. | |
| 6 | 6 | */ | |
| 7 | − | import type { Branch, Commit, Pull, RepoPath, Viewer } from "@g1t/contracts"; | |
| 7 | + | import type { Branch, BranchDrifts, Pull, RepoPath, Viewer } from "@g1t/contracts"; | |
| 8 | 8 | ||
| 9 | 9 | import type { ActiveBranch } from "../components/branches"; | |
| 10 | − | import { bounded, drift } from "./branches"; | |
| 11 | − | import { immutable } from "./immutable.server"; | |
| 10 | + | import { activeBranches, branchesToRead, summary } from "./branches"; | |
| 12 | 11 | import { repos } from "./services.server"; | |
| 13 | − | ||
| 14 | − | /** | |
| 15 | − | * How deep each history is read, in turn, to find where a branch and the | |
| 16 | − | * default branch meet: most branches meet it within the first; a branch | |
| 17 | − | * left long ago needs the default branch's history further back; one far | |
| 18 | − | * from both reads both deeply. Past the last, the counts are not shown. | |
| 19 | − | */ | |
| 20 | − | const DEPTHS: ReadonlyArray<{ branch: number; main: number }> = [ | |
| 21 | − | { branch: 40, main: 120 }, | |
| 22 | − | { branch: 40, main: 1000 }, | |
| 23 | − | { branch: 1000, main: 1000 }, | |
| 24 | − | ]; | |
| 25 | − | /** Branches counted at once. */ | |
| 26 | − | const COUNTING = 8; | |
| 27 | 12 | ||
| 28 | 13 | type Preview = { branch?: string | null; number?: number | null; url: string }; | |
| 29 | − | /** What never changes for a branch head and a default branch head. */ | |
| 30 | − | type Measured = { commit: ActiveBranch["commit"]; drift: ActiveBranch["drift"] }; | |
| 31 | 14 | ||
| 32 | − | const summary = (commit: Commit | undefined): ActiveBranch["commit"] => | |
| 33 | − | commit ? { hash: commit.hash, message: commit.message.split("\n")[0] ?? "", author: commit.author.name, at: commit.authoredAt } : null; | |
| 34 | − | ||
| 35 | 15 | /** | |
| 36 | 16 | * The branches other than the default, at most `read` of them read (those | |
| 37 | − | * with an open pull request first), newest commit first. `main` is the | |
| 38 | − | * default branch's head commit with its first line. | |
| 17 | + | * with an open pull request first), newest commit first, and the default | |
| 18 | + | * branch's head commit with its first line. | |
| 19 | + | * | |
| 20 | + | * One call to repos (`branch_drift`, services/repos/src/drift.rs) measures | |
| 21 | + | * every branch. It keeps each answer by the pair of head commits, so only | |
| 22 | + | * branches that moved, or every branch once the default branch moved, cost | |
| 23 | + | * a walk, and that walk reads the default branch's history once for all of | |
| 24 | + | * them. Before 2026-10-08 this was a `log` call per branch per depth from | |
| 25 | + | * here, up to twenty on an overview. | |
| 39 | 26 | */ | |
| 40 | 27 | export async function readBranches( | |
| 41 | 28 | path: RepoPath, | |
| ⋯ | |||
| 43 | 30 | input: { defaultBranch: string; branches: Branch[]; pulls: Pull[]; previews: Preview[] }, | |
| 44 | 31 | read: number, | |
| 45 | 32 | ): Promise<{ main: string; total: number; shown: ActiveBranch[]; head: ActiveBranch["commit"] }> { | |
| 46 | − | const soft = <T,>(promise: Promise<T>): Promise<T | null> => promise.catch(() => null); | |
| 47 | − | const main = input.defaultBranch; | |
| 48 | − | const pullOn = new Map(input.pulls.filter((pull) => pull.branch).map((pull) => [pull.branch as string, pull])); | |
| 49 | − | const others = input.branches.filter((branch) => branch.name !== main); | |
| 50 | − | const reading = [...others.filter((b) => pullOn.has(b.name)), ...others.filter((b) => !pullOn.has(b.name))].slice(0, read); | |
| 51 | − | // By commit hash, not name: history from a commit never changes, so | |
| 52 | − | // repos keeps it (services/repos/src/store.rs) and only new heads cost a walk. | |
| 53 | − | const mainHead = input.branches.find((branch) => branch.name === main)?.hash ?? null; | |
| 54 | − | const log = async (hash: string, depth: number): Promise<Commit[] | null> => { | |
| 55 | − | const found = await soft(repos.log(path, viewer, hash, depth)); | |
| 56 | − | return found?.ok && found.value.length > 0 ? found.value : null; | |
| 57 | − | }; | |
| 58 | − | // The default branch's history, read once per depth for every branch, | |
| 59 | − | // and not read deeper when a shallower read already reached its start. | |
| 60 | − | const mainLogs = new Map<number, Promise<Commit[] | null>>(); | |
| 61 | − | const mainLog = (depth: number): Promise<Commit[] | null> => { | |
| 62 | − | if (!mainHead) return Promise.resolve(null); | |
| 63 | − | let kept = mainLogs.get(depth); | |
| 64 | − | if (!kept) { | |
| 65 | − | const shallower = Math.max(0, ...[...mainLogs.keys()].filter((read) => read < depth)); | |
| 66 | − | const before = mainLogs.get(shallower); | |
| 67 | − | kept = before | |
| 68 | − | ? before.then((read) => (read && read.length < shallower ? read : log(mainHead, depth))) | |
| 69 | − | : log(mainHead, depth); | |
| 70 | − | mainLogs.set(depth, kept); | |
| 71 | − | } | |
| 72 | − | return kept; | |
| 73 | − | }; | |
| 74 | − | // Reads deeper only while the two histories have not met. Null when a | |
| 75 | − | // read failed, so a failure is not kept as the answer. | |
| 76 | − | const measure = async (head: string, main: string): Promise<Measured | null> => { | |
| 77 | − | let branch: Commit[] | null = null; | |
| 78 | − | let readTo = 0; | |
| 79 | − | for (const depth of DEPTHS) { | |
| 80 | − | // Not read again when no deeper, or when it already reached the start. | |
| 81 | − | const again: boolean = branch == null || (depth.branch > readTo && branch.length >= readTo); | |
| 82 | − | const [read, mainRead]: [Commit[] | null, Commit[] | null] = await Promise.all([ | |
| 83 | − | again ? log(head, depth.branch) : branch, | |
| 84 | − | mainLog(depth.main), | |
| 85 | − | ]); | |
| 86 | − | if (!read || !mainRead) return null; | |
| 87 | − | if (again) readTo = depth.branch; | |
| 88 | − | branch = read; | |
| 89 | − | const counted = drift(head, main, [...mainRead, ...read]); | |
| 90 | − | if (counted) return { commit: summary(read[0]), drift: counted }; | |
| 91 | − | } | |
| 92 | − | return { commit: summary(branch?.[0]), drift: null }; | |
| 93 | − | }; | |
| 94 | − | // Kept by the pair of heads: neither history can change, so neither can | |
| 95 | − | // the answer. Without one, the head commit alone. | |
| 96 | − | const measureOrHead = async (branch: Branch): Promise<Measured> => { | |
| 97 | − | const kept = | |
| 98 | − | branch.hash && mainHead | |
| 99 | − | ? await immutable(`branch-drift:${path.namespace}/${path.name}:${mainHead}:${branch.hash}`, () => measure(branch.hash, mainHead)) | |
| 100 | − | : null; | |
| 101 | − | if (kept) return kept; | |
| 102 | − | const head = await log(branch.hash || branch.name, 1); | |
| 103 | − | return { commit: summary(head?.[0]), drift: null }; | |
| 104 | − | }; | |
| 105 | − | const [mainTop, measured] = await Promise.all([ | |
| 106 | − | mainHead ? log(mainHead, 1) : Promise.resolve(null), | |
| 107 | − | bounded(reading, COUNTING, measureOrHead), | |
| 108 | − | ]); | |
| 109 | − | const shown = reading | |
| 110 | − | .map((branch, index): ActiveBranch => { | |
| 111 | − | const pull = pullOn.get(branch.name); | |
| 112 | − | return { | |
| 113 | − | name: branch.name, | |
| 114 | − | commit: measured[index]?.commit ?? null, | |
| 115 | − | drift: measured[index]?.drift ?? null, | |
| 116 | − | pull: pull ? { number: pull.number, title: pull.title, checkStatus: pull.checkStatus, draft: pull.status === "draft" } : null, | |
| 117 | − | preview: input.previews.find((app) => app.branch === branch.name || (pull != null && app.number === pull.number))?.url ?? null, | |
| 118 | − | }; | |
| 119 | − | }) | |
| 120 | − | .sort((a, b) => Date.parse(b.commit?.at ?? "0") - Date.parse(a.commit?.at ?? "0")); | |
| 121 | − | return { main, total: others.length, shown, head: summary(mainTop?.[0]) }; | |
| 33 | + | const { reading, total, mainHead } = branchesToRead(input, read); | |
| 34 | + | const measured: BranchDrifts | null = mainHead | |
| 35 | + | ? await repos | |
| 36 | + | .branchDrift(path, viewer, mainHead, reading.map((branch) => branch.hash).filter(Boolean)) | |
| 37 | + | .then((found) => (found.ok ? found.value : null)) | |
| 38 | + | .catch(() => null) | |
| 39 | + | : null; | |
| 40 | + | return { main: input.defaultBranch, total, shown: activeBranches(reading, measured, input), head: summary(measured?.base) }; | |
| 122 | 41 | } | |
| 1 | 1 | import assert from "node:assert/strict"; | |
| 2 | 2 | import { test } from "node:test"; | |
| 3 | 3 | ||
| 4 | − | import { bounded, drift, type Link } from "./branches.ts"; | |
| 4 | + | import type { BranchDrifts, Commit } from "@g1t/contracts"; | |
| 5 | 5 | ||
| 6 | − | /** A history from `[hash, ...parents]` rows. */ | |
| 7 | − | const graph = (...rows: string[][]): Link[] => rows.map(([hash, ...parents]) => ({ hash: hash as string, parents })); | |
| 6 | + | import { activeBranches, branchesToRead } from "./branches.ts"; | |
| 8 | 7 | ||
| 9 | − | // main: m1 <- m2 <- m3; the branch left at m2 and added b1 <- b2. | |
| 10 | − | const forked = graph(["m3", "m2"], ["m2", "m1"], ["m1"], ["b2", "b1"], ["b1", "m2"]); | |
| 11 | − | ||
| 12 | − | test("a branch two ahead of where main was, with main one further on", () => { | |
| 13 | − | assert.deepEqual(drift("b2", "m3", forked), { ahead: 2, behind: 1 }); | |
| 14 | − | }); | |
| 15 | − | ||
| 16 | − | test("a branch at main's head is level", () => { | |
| 17 | − | assert.deepEqual(drift("m3", "m3", forked), { ahead: 0, behind: 0 }); | |
| 18 | − | }); | |
| 19 | − | ||
| 20 | − | test("a branch merged long ago is nothing ahead and all of main since behind", () => { | |
| 21 | − | const history = graph(["m5", "m4"], ["m4", "m3"], ["m3", "m2"], ["m2", "m1"]); | |
| 22 | − | // Neither history was read to its start, but they meet in what was. | |
| 23 | − | assert.deepEqual(drift("m2", "m5", history), { ahead: 0, behind: 3 }); | |
| 8 | + | const commit = (hash: string, at: string, message = `${hash}\n\nbody`): Commit => ({ | |
| 9 | + | hash, | |
| 10 | + | treeHash: `t${hash}`, | |
| 11 | + | message, | |
| 12 | + | author: { name: "Ada", email: "ada@example.com" }, | |
| 13 | + | parents: [], | |
| 14 | + | authoredAt: at, | |
| 24 | 15 | }); | |
| 25 | 16 | ||
| 26 | − | test("a merge into main counts the merged side once", () => { | |
| 27 | − | // main merged side branch s1 <- s2 at m3; the branch is still at m1. | |
| 28 | − | const history = graph(["m3", "m2", "s2"], ["s2", "s1"], ["s1", "m1"], ["m2", "m1"], ["m1"], ["b1", "m1"]); | |
| 29 | − | assert.deepEqual(drift("b1", "m3", history), { ahead: 1, behind: 4 }); | |
| 30 | − | }); | |
| 17 | + | const branches = [ | |
| 18 | + | { name: "main", hash: "m3" }, | |
| 19 | + | { name: "old", hash: "o1" }, | |
| 20 | + | { name: "fix", hash: "f2" }, | |
| 21 | + | { name: "idea", hash: "i1" }, | |
| 22 | + | ]; | |
| 23 | + | const pulls = [{ branch: "fix", number: 7, title: "Fix it", checkStatus: "passed" as const, status: "open" as const }]; | |
| 31 | 24 | ||
| 32 | − | test("histories that never meet count everything on each, once read to the start", () => { | |
| 33 | − | const history = graph(["b2", "b1"], ["b1"], ["m2", "m1"], ["m1"]); | |
| 34 | − | assert.deepEqual(drift("b2", "m2", history), { ahead: 2, behind: 2 }); | |
| 25 | + | test("branches with an open pull request are read first, the default branch never", () => { | |
| 26 | + | const read = branchesToRead({ defaultBranch: "main", branches, pulls }, 2); | |
| 27 | + | assert.deepEqual(read.reading.map((b) => b.name), ["fix", "old"]); | |
| 28 | + | assert.equal(read.total, 3); | |
| 29 | + | assert.equal(read.mainHead, "m3"); | |
| 35 | 30 | }); | |
| 36 | 31 | ||
| 37 | − | test("no answer when what was read stops before the two meet", () => { | |
| 38 | − | // Main's history was read only to m2, whose parent the branch may share. | |
| 39 | − | const history = graph(["m3", "m2"], ["m2", "m1"], ["b2", "b1"], ["b1", "m0"]); | |
| 40 | − | assert.equal(drift("b2", "m3", history), null); | |
| 32 | + | test("no default branch head, nothing to measure against", () => { | |
| 33 | + | assert.equal(branchesToRead({ defaultBranch: "trunk", branches, pulls: [] }, 10).mainHead, null); | |
| 41 | 34 | }); | |
| 42 | 35 | ||
| 43 | − | test("no answer without either head", () => { | |
| 44 | − | assert.equal(drift("b9", "m3", forked), null); | |
| 45 | − | assert.equal(drift("b2", "m9", forked), null); | |
| 36 | + | test("each branch gets its measured commit and drift, newest first, by head hash", () => { | |
| 37 | + | const measured: BranchDrifts = { | |
| 38 | + | base: commit("m3", "2026-10-08T00:00:00Z"), | |
| 39 | + | branches: [ | |
| 40 | + | { head: "f2", commit: commit("f2", "2026-10-07T00:00:00Z"), drift: { ahead: 2, behind: 1 } }, | |
| 41 | + | { head: "o1", commit: commit("o1", "2026-01-01T00:00:00Z"), drift: null }, | |
| 42 | + | { head: "i1", commit: commit("i1", "2026-10-08T00:00:00Z", "one line"), drift: { ahead: 1, behind: 0 } }, | |
| 43 | + | ], | |
| 44 | + | }; | |
| 45 | + | const { reading } = branchesToRead({ defaultBranch: "main", branches, pulls }, 10); | |
| 46 | + | const shown = activeBranches(reading, measured, { pulls, previews: [{ number: 7, url: "https://fix.g1t.page" }] }); | |
| 47 | + | assert.deepEqual(shown.map((b) => b.name), ["idea", "fix", "old"]); | |
| 48 | + | assert.deepEqual(shown[1], { | |
| 49 | + | name: "fix", | |
| 50 | + | commit: { hash: "f2", message: "f2", author: "Ada", at: "2026-10-07T00:00:00Z" }, | |
| 51 | + | drift: { ahead: 2, behind: 1 }, | |
| 52 | + | pull: { number: 7, title: "Fix it", checkStatus: "passed", draft: false }, | |
| 53 | + | preview: "https://fix.g1t.page", | |
| 54 | + | }); | |
| 55 | + | assert.equal(shown[2]?.drift, null); | |
| 46 | 56 | }); | |
| 47 | 57 | ||
| 48 | − | test("bounded loads everything in order, never more at once than asked", async () => { | |
| 49 | − | let running = 0; | |
| 50 | − | let most = 0; | |
| 51 | − | const out = await bounded([5, 1, 4, 2, 3], 2, async (wait) => { | |
| 52 | − | running++; | |
| 53 | − | most = Math.max(most, running); | |
| 54 | − | await new Promise((done) => setTimeout(done, wait)); | |
| 55 | − | running--; | |
| 56 | − | return wait * 10; | |
| 57 | − | }); | |
| 58 | − | assert.deepEqual(out, [50, 10, 40, 20, 30]); | |
| 59 | − | assert.equal(most, 2); | |
| 58 | + | test("when repos could not answer, the branches still show, without commits or counts", () => { | |
| 59 | + | const { reading } = branchesToRead({ defaultBranch: "main", branches, pulls: [] }, 10); | |
| 60 | + | const shown = activeBranches(reading, null, { pulls: [], previews: [] }); | |
| 61 | + | assert.equal(shown.length, 3); | |
| 62 | + | assert.ok(shown.every((b) => b.commit == null && b.drift == null && b.pull == null)); | |
| 60 | 63 | }); |
| 1 | 1 | /** | |
| 2 | + | * Active branches as the overview and the Branches page show them, from | |
| 3 | + | * the repos service's `branch_drift` answer. The counting itself is done | |
| 4 | + | * there (services/repos/src/drift.rs), kept by the pair of head commits. | |
| 5 | + | */ | |
| 6 | + | import type { Branch, BranchDrifts, Commit, Pull } from "@g1t/contracts"; | |
| 7 | + | ||
| 8 | + | import type { ActiveBranch } from "../components/branches"; | |
| 9 | + | ||
| 10 | + | /** | |
| 2 | 11 | * How far a branch has moved from the default branch: commits it has that | |
| 3 | 12 | * the default branch does not (ahead), and commits the default branch has | |
| 4 | 13 | * that it does not (behind), the way `git rev-list --left-right --count` | |
| 5 | − | * says it. Worked out from what was read of the two histories, merges | |
| 6 | − | * included; when what was read stops short of where they meet, there is no | |
| 7 | − | * answer rather than a guess. | |
| 14 | + | * says it. | |
| 8 | 15 | */ | |
| 9 | 16 | export type Drift = { ahead: number; behind: number }; | |
| 10 | 17 | ||
| 11 | − | /** A commit as far as counting needs it. */ | |
| 12 | − | export type Link = { hash: string; parents: string[] }; | |
| 18 | + | type Preview = { branch?: string | null; number?: number | null; url: string }; | |
| 13 | 19 | ||
| 20 | + | /** A commit as a branch row shows it: its first line. */ | |
| 21 | + | export const summary = (commit: Commit | null | undefined): ActiveBranch["commit"] => | |
| 22 | + | commit ? { hash: commit.hash, message: commit.message.split("\n")[0] ?? "", author: commit.author.name, at: commit.authoredAt } : null; | |
| 23 | + | ||
| 14 | 24 | /** | |
| 15 | − | * `commits` is everything read of either history, in any order, repeats | |
| 16 | − | * allowed. Null when either head is missing from it, or when a commit only | |
| 17 | − | * one side reaches has a parent that was not read: that parent's history | |
| 18 | − | * could change either count. | |
| 25 | + | * The branches other than the default that are read (at most `read`, those | |
| 26 | + | * with an open pull request first), how many there are, and the default | |
| 27 | + | * branch's head commit. | |
| 19 | 28 | */ | |
| 20 | − | export function drift(branch: string, main: string, commits: Iterable<Link>): Drift | null { | |
| 21 | − | const parents = new Map<string, string[]>(); | |
| 22 | − | for (const commit of commits) parents.set(commit.hash, commit.parents); | |
| 23 | − | if (!parents.has(branch) || !parents.has(main)) return null; | |
| 24 | − | const fromBranch = reach(branch, parents); | |
| 25 | − | const fromMain = reach(main, parents); | |
| 26 | − | let ahead = 0; | |
| 27 | − | let behind = 0; | |
| 28 | − | for (const [hash, above] of parents) { | |
| 29 | − | const onBranch = fromBranch.has(hash); | |
| 30 | − | const onMain = fromMain.has(hash); | |
| 31 | − | if (onBranch === onMain) continue; | |
| 32 | − | if (above.some((parent) => !parents.has(parent))) return null; | |
| 33 | − | if (onBranch) ahead++; | |
| 34 | − | else behind++; | |
| 35 | − | } | |
| 36 | − | return { ahead, behind }; | |
| 37 | − | } | |
| 38 | − | ||
| 39 | − | /** Every commit read that `head` descends from, itself included. */ | |
| 40 | − | function reach(head: string, parents: Map<string, string[]>): Set<string> { | |
| 41 | − | const seen = new Set([head]); | |
| 42 | − | const next = [head]; | |
| 43 | − | for (let hash = next.pop(); hash != null; hash = next.pop()) { | |
| 44 | − | for (const parent of parents.get(hash) ?? []) { | |
| 45 | − | if (parents.has(parent) && !seen.has(parent)) { | |
| 46 | − | seen.add(parent); | |
| 47 | − | next.push(parent); | |
| 48 | − | } | |
| 49 | − | } | |
| 50 | − | } | |
| 51 | − | return seen; | |
| 29 | + | export function branchesToRead( | |
| 30 | + | input: { defaultBranch: string; branches: Branch[]; pulls: Pick<Pull, "branch">[] }, | |
| 31 | + | read: number, | |
| 32 | + | ): { reading: Branch[]; total: number; mainHead: string | null } { | |
| 33 | + | const pullOn = new Set(input.pulls.flatMap((pull) => (pull.branch ? [pull.branch] : []))); | |
| 34 | + | const others = input.branches.filter((branch) => branch.name !== input.defaultBranch); | |
| 35 | + | const reading = [...others.filter((b) => pullOn.has(b.name)), ...others.filter((b) => !pullOn.has(b.name))].slice(0, read); | |
| 36 | + | const mainHead = input.branches.find((branch) => branch.name === input.defaultBranch)?.hash || null; | |
| 37 | + | return { reading, total: others.length, mainHead }; | |
| 52 | 38 | } | |
| 53 | 39 | ||
| 54 | − | /** `load` over each of `items`, at most `limit` at a time, answers in order. */ | |
| 55 | − | export async function bounded<T, R>(items: readonly T[], limit: number, load: (item: T) => Promise<R>): Promise<R[]> { | |
| 56 | − | const out = new Array<R>(items.length); | |
| 57 | − | let taken = 0; | |
| 58 | − | const worker = async () => { | |
| 59 | − | while (taken < items.length) { | |
| 60 | − | const index = taken++; | |
| 61 | − | out[index] = await load(items[index] as T); | |
| 62 | − | } | |
| 63 | − | }; | |
| 64 | − | await Promise.all(Array.from({ length: Math.min(limit, items.length) }, worker)); | |
| 65 | − | return out; | |
| 40 | + | /** | |
| 41 | + | * Each branch read, newest commit first, with what repos measured (null | |
| 42 | + | * when that could not be had: no commits, no counts), its pull request and | |
| 43 | + | * its preview. | |
| 44 | + | */ | |
| 45 | + | export function activeBranches( | |
| 46 | + | reading: Branch[], | |
| 47 | + | measured: BranchDrifts | null, | |
| 48 | + | input: { pulls: Pick<Pull, "branch" | "number" | "title" | "checkStatus" | "status">[]; previews: Preview[] }, | |
| 49 | + | ): ActiveBranch[] { | |
| 50 | + | const pullOn = new Map(input.pulls.flatMap((pull) => (pull.branch ? [[pull.branch, pull] as const] : []))); | |
| 51 | + | const byHead = new Map((measured?.branches ?? []).map((found) => [found.head, found])); | |
| 52 | + | return reading | |
| 53 | + | .map((branch): ActiveBranch => { | |
| 54 | + | const pull = pullOn.get(branch.name); | |
| 55 | + | const found = byHead.get(branch.hash); | |
| 56 | + | return { | |
| 57 | + | name: branch.name, | |
| 58 | + | commit: summary(found?.commit), | |
| 59 | + | drift: found?.drift ?? null, | |
| 60 | + | pull: pull ? { number: pull.number, title: pull.title, checkStatus: pull.checkStatus, draft: pull.status === "draft" } : null, | |
| 61 | + | preview: input.previews.find((app) => app.branch === branch.name || (pull != null && app.number === pull.number))?.url ?? null, | |
| 62 | + | }; | |
| 63 | + | }) | |
| 64 | + | .sort((a, b) => Date.parse(b.commit?.at ?? "0") - Date.parse(a.commit?.at ?? "0")); | |
| 66 | 65 | } |
| 64 | 64 | }); | |
| 65 | 65 | ||
| 66 | 66 | test("only known reads are taken not to write", () => { | |
| 67 | − | for (const method of ["get_pull", "list_pulls", "counts", "user_for_session", "explore", "usage", "get", "list", "queue", "pulls_for_repos", "stars", "about", "public_links"]) { | |
| 67 | + | for (const method of ["get_pull", "list_pulls", "counts", "user_for_session", "explore", "usage", "get", "list", "queue", "pulls_for_repos", "stars", "about", "public_links", "branch_drift", "tags", "commit_checks", "shortcuts", "last_commits", "languages"]) { | |
| 68 | 68 | assert.equal(mayWrite(method), false, method); | |
| 69 | 69 | } | |
| 70 | 70 | for (const method of ["merge_pull", "verify_email", "github_finish", "sign_in", "something_new"]) { |
| 100 | 100 | * page called `get` (repos, projects), so every one set the cookie, was | |
| 101 | 101 | * never kept in the public cache, and sent the next 30 s of the person's | |
| 102 | 102 | * reads to the primary. `stars`, `about` and `public_links` (2026-10-08) | |
| 103 | − | * did the same to every project page and Explore for 13 hours. | |
| 103 | + | * did the same to every project page and Explore for 13 hours. `tags`, | |
| 104 | + | * `commit_checks` and `shortcuts` (every overview), and the Files page's | |
| 105 | + | * `last_commits`, `languages`, `contributors` and `license`, still made a | |
| 106 | + | * signed-in view count as a write until 2026-10-08: the cookie, 30 s of | |
| 107 | + | * primary reads, and no sidebar cache after every overview. | |
| 104 | 108 | */ | |
| 105 | 109 | const READS = new Set( | |
| 106 | 110 | ( | |
| ⋯ | |||
| 113 | 117 | "routes run run_context run_cost runner_groups runner_settings runners runs scorecards search search_memories " + | |
| 114 | 118 | "settings statement statement_entries status status_by_id suggest tree usage usage_meters user_by_username " + | |
| 115 | 119 | "user_for_session usernames waiting_workspaces workflows workspace workspace_invites github_enabled " + | |
| 116 | − | "stars about public_links" | |
| 120 | + | "stars about public_links branch_drift tags last_commits languages contributors license releases release " + | |
| 121 | + | "stargazers starred commit_checks shortcuts" | |
| 117 | 122 | ).split(" "), | |
| 118 | 123 | ); | |
| 119 | 124 | ||
| 20 | 20 | import { clearWelcome, welcomes } from "../../lib/invites"; | |
| 21 | 21 | import { notFound } from "../../lib/not-found.server"; | |
| 22 | 22 | import { redirectIfRenamed, redirectIfTransferred } from "../../lib/renamed.server"; | |
| 23 | − | import { accessFor, countsFor, repoFor } from "../../lib/access.server"; | |
| 23 | + | import { accessFor, countsFor, projectFor, repoFor } from "../../lib/access.server"; | |
| 24 | 24 | import { inbox, projects, repos } from "../../lib/services.server"; | |
| 25 | 25 | import { getViewer, roleIn, unwrap } from "../../lib/session.server"; | |
| 26 | 26 | ||
| ⋯ | |||
| 36 | 36 | const [repo, counts, found, watching, shortcuts, stars] = await Promise.all([ | |
| 37 | 37 | repoFor(context, params), | |
| 38 | 38 | countsFor(context, params), | |
| 39 | − | projects.get(params.owner, params.repo, viewer), | |
| 39 | + | projectFor(context, params), | |
| 40 | 40 | // How the person watches it, for the header's Watch menu, as soon as | |
| 41 | 41 | // the repository is known. | |
| 42 | 42 | viewer | |
| 22 | 22 | Rocket, | |
| 23 | 23 | RotateCw, | |
| 24 | 24 | } from "lucide-react"; | |
| 25 | + | import { waitUntil } from "cloudflare:workers"; | |
| 26 | + | import { isbot } from "isbot"; | |
| 25 | 27 | import { type ReactNode, Suspense } from "react"; | |
| 26 | 28 | import { Await, Form, Link } from "react-router"; | |
| 27 | 29 | ||
| ⋯ | |||
| 92 | 94 | import { actions, agents, deployments, events as eventLog, identity, packages, projects, repos, work } from "../../lib/services.server"; | |
| 93 | 95 | import { madeByG1t } from "../../lib/opened-by"; | |
| 94 | 96 | import { assertSameOrigin, getViewer, requireUser } from "../../lib/session.server"; | |
| 95 | − | import { accessTo, countsFor, refusal, repoFor } from "../../lib/access.server"; | |
| 97 | + | import { accessTo, countsFor, projectFor, refusal, repoFor } from "../../lib/access.server"; | |
| 96 | 98 | import { shotVersion } from "./production-screenshot"; | |
| 97 | 99 | import { DeploymentsPanel } from "../../components/deployments-panel"; | |
| 98 | 100 | import { environmentUrl, productionEnvironment } from "../../lib/deployments"; | |
| ⋯ | |||
| 111 | 113 | const COMMITS_SHOWN = 5; | |
| 112 | 114 | /** How long Active branches may take before the section links to Branches instead. */ | |
| 113 | 115 | const BRANCHES_WAIT_MS = 3_500; | |
| 116 | + | /** | |
| 117 | + | * The same for a crawler, which gets the page only once everything in it | |
| 118 | + | * has settled (entry.server.tsx): past this, it gets the link to Branches. | |
| 119 | + | */ | |
| 120 | + | const BRANCHES_WAIT_CRAWLER_MS = 700; | |
| 114 | 121 | ||
| 115 | 122 | /** | |
| 116 | 123 | * The overview streams: the layout's header and tabs (one repository | |
| ⋯ | |||
| 119 | 126 | * same response as it settles. Crawlers wait for all of it | |
| 120 | 127 | * (entry.server.tsx). Before, the first byte waited on the slowest of them. | |
| 121 | 128 | */ | |
| 122 | − | export function loader({ params, context }: Route.LoaderArgs) { | |
| 123 | − | return { overview: overviewData({ params, context }) }; | |
| 129 | + | export function loader({ params, context, request }: Route.LoaderArgs) { | |
| 130 | + | const userAgent = request.headers.get("user-agent"); | |
| 131 | + | return { overview: overviewData({ params, context }, Boolean(userAgent && isbot(userAgent))) }; | |
| 124 | 132 | } | |
| 125 | 133 | ||
| 126 | − | async function overviewData({ params, context }: Pick<Route.LoaderArgs, "params" | "context">) { | |
| 134 | + | async function overviewData({ params, context }: Pick<Route.LoaderArgs, "params" | "context">, crawler: boolean) { | |
| 127 | 135 | const viewer = getViewer(context); | |
| 128 | 136 | const path = { namespace: params.owner, name: params.repo }; | |
| 129 | 137 | const ref = { workspace: params.owner, slug: params.repo }; | |
| ⋯ | |||
| 140 | 148 | const forMembers = <T,>(start: () => Promise<T>): Promise<T | null> => | |
| 141 | 149 | memberP.then((member) => (member ? soft(start()) : null)); | |
| 142 | 150 | const repoP = soft(repoFor(context, params)); | |
| 143 | − | const projectP = soft(projects.get(params.owner, params.repo, viewer)); | |
| 151 | + | const projectP = soft(projectFor(context, params)); | |
| 144 | 152 | // A library or a tool shows its packages where an app shows production, | |
| 145 | 153 | // and every project g1t does not deploy counts a workflow in its checklist. | |
| 146 | 154 | // Deployments from anywhere (g1t.page, g1t Actions, the API), by environment, for anyone who can read it. | |
| ⋯ | |||
| 182 | 190 | return { main, total: read.total, shown: read.shown.slice(0, BRANCHES_SHOWN) }; | |
| 183 | 191 | }, | |
| 184 | 192 | ); | |
| 185 | − | // Active branches read several logs each: streamed, so the rest shows | |
| 186 | − | // first. Bounded well inside the response's stream timeout | |
| 187 | − | // (entry.server.tsx): a promise still pending when the stream ends never | |
| 188 | − | // settles in the browser, and its skeleton would stay. | |
| 193 | + | // Active branches walk history when a branch or the default branch has | |
| 194 | + | // moved: streamed, so the rest shows first. Bounded well inside the | |
| 195 | + | // response's stream timeout (entry.server.tsx): a promise still pending | |
| 196 | + | // when the stream ends never settles in the browser, and its skeleton | |
| 197 | + | // would stay. The walk finishes after the page if it must (waitUntil), | |
| 198 | + | // so repos keeps the answer and the next view has it. Crawlers, which | |
| 199 | + | // wait for the whole page, wait for it only briefly. | |
| 200 | + | waitUntil(branchesP.then(() => undefined, () => undefined)); | |
| 189 | 201 | const branches = Promise.race([ | |
| 190 | 202 | branchesP.catch(() => null), | |
| 191 | − | new Promise<"slow">((resolve) => setTimeout(() => resolve("slow"), BRANCHES_WAIT_MS)), | |
| 203 | + | new Promise<"slow">((resolve) => setTimeout(() => resolve("slow"), crawler ? BRANCHES_WAIT_CRAWLER_MS : BRANCHES_WAIT_MS)), | |
| 192 | 204 | ]); | |
| 193 | 205 | const [{ insider: member, can }, project, settings, list, open, closed, log, counts, deps, runs, queue, issues, memories, recent, mine, domains, root, packageList, workflows, tagList] = await Promise.all([ | |
| 194 | 206 | accessP, | |
| 617 | 617 | pub complete: bool, | |
| 618 | 618 | } | |
| 619 | 619 | ||
| 620 | + | /// `branch_drift`: how far each of `heads` (branch head commits) has moved | |
| 621 | + | /// from `base` (the default branch's head commit), and each one's head | |
| 622 | + | /// commit, in one call. Every answer is kept by the pair of hashes: neither | |
| 623 | + | /// history can change, so neither can it. Returns `Outcome<BranchDrifts>`. | |
| 624 | + | #[derive(Debug, Serialize, Deserialize)] | |
| 625 | + | #[serde(rename_all = "camelCase")] | |
| 626 | + | pub struct BranchDriftArgs { | |
| 627 | + | pub path: RepoPath, | |
| 628 | + | pub viewer: Viewer, | |
| 629 | + | pub base: String, | |
| 630 | + | pub heads: Vec<String>, | |
| 631 | + | } | |
| 632 | + | ||
| 633 | + | /// Commits a branch has that the default branch does not (`ahead`), and | |
| 634 | + | /// the other way round (`behind`), as `git rev-list --left-right --count`. | |
| 635 | + | #[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize, Deserialize)] | |
| 636 | + | pub struct Drift { | |
| 637 | + | pub ahead: u32, | |
| 638 | + | pub behind: u32, | |
| 639 | + | } | |
| 640 | + | ||
| 641 | + | /// One branch head's commit and drift. `drift` is absent when the two | |
| 642 | + | /// histories do not meet within what is read (or could not be read). | |
| 643 | + | #[derive(Clone, Debug, Serialize, Deserialize)] | |
| 644 | + | pub struct BranchDrift { | |
| 645 | + | pub head: String, | |
| 646 | + | pub commit: Option<Commit>, | |
| 647 | + | pub drift: Option<Drift>, | |
| 648 | + | } | |
| 649 | + | ||
| 650 | + | /// `base`'s own commit, and each head's answer in the order asked. | |
| 651 | + | #[derive(Clone, Debug, Serialize, Deserialize)] | |
| 652 | + | pub struct BranchDrifts { | |
| 653 | + | pub base: Option<Commit>, | |
| 654 | + | pub branches: Vec<BranchDrift>, | |
| 655 | + | } | |
| 656 | + | ||
| 620 | 657 | /// `tags`: the repository's tags, newest commit first, each with the | |
| 621 | 658 | /// commit it names. Returns `Outcome<Vec<Tag>>`. | |
| 622 | 659 | #[derive(Debug, Serialize, Deserialize)] |
| 188 | 188 | | Public pages for people signed out | the data centre's cache (`workers/app.ts`, `servePublic`) | fresh 30 s, then served once more while a new copy is made, up to 5 min | GET, no `g1t_session` cookie, an allowlisted path (home, pricing, explore, policies, a project's pages), status 200 or 404, no `Set-Cookie`, nothing private. Reserved first segments and workspace pages (`-`) are never kept. A project's kept page is served only after repos' `visibility` says the repository is still there and public (one indexed read, alongside the cache lookup); a repository made private or deleted is never served from any data centre's copy, and the copy is dropped. The answer says `server-timing: cache;desc="hit, Ns old"`. | | |
| 189 | 189 | | Sidebar data (projects, spend, limit, entitlements) | per isolate (`lib/cache.server.ts`) | 15 s, per person and workspace | skipped during a write and for 30 s after the person's last one; failures not kept; only settled answers kept | | |
| 190 | 190 | | Registration mode | per isolate | 60 s | | | |
| 191 | − | | A commit's log by hash | repos' data-centre cache | for good | history from a commit never changes; Active branches asks by hash | | |
| 191 | + | | A commit's log by hash | repos' data-centre cache | for good | history from a commit never changes. One of 100 commits or more is put together from a 16-commit read and the history kept from any of those commits, when there is one (`store.rs` `spliced_log`): a default branch that moved by a merge costs 16 commits, not 120 or 1,000 | | |
| 192 | + | | A branch's drift from the default branch (Active branches) | repos' data-centre cache (`branch_drift`) | for good | by repository and the pair of head commits; a failed read is not kept | | |
| 193 | + | | A repository's tags | repos' data-centre cache | until the refs move, 5 min at most | as the branch list; not kept when a tag's commit could not be read | | |
| 192 | 194 | | Git objects, trees, refs | repos' caches | see services/repos | | | |
| 193 | 195 | | A branch's log, the branch list, a file by branch and path | repos' data-centre cache | until the repository's refs change (`refs_version`), 5 min at most | only while no handed-out push credential is live; by commit hash for good (docs/ARTIFACTS.md R9) | | |
| 194 | 196 | | A target branch's history, for mergeability | the repos isolate | 60 s, per target head | 100 pull requests checked after a push walk it once (R10) | | |
| ⋯ | |||
| 235 | 237 | | --- | --- | --- | | |
| 236 | 238 | | Any in-app navigation | root (sidebar: 6 calls, then up to 20 `get_by_id` for shared repositories) and the project layout re-ran when moving between pages of a project | root re-runs only when the workspace or project changes, or after a form; the project layout likewise; shared repositories are one `readable` call, in the same round; the sidebar's workspace data is cached for 15 s; open counts are read once per request for both | | |
| 237 | 239 | | Pull request ("Review and respond") | access, then 8 calls, then checks' runs / comparison / session, then up to 5 more deployments lookups for stacked previews | one round of 9 (access-dependent ones start as soon as the repository lookup returns), then the comparison on Changes; the workflow jobs and stacked previews stream in | | |
| 238 | − | | Project overview | access, then 17 calls, one of which (Active branches) read the default branch's last 120 commits and up to 10 branches' last 40 | one round; Active branches streams in with a skeleton, reading logs by commit hash so a branch that has not moved costs nothing | | |
| 240 | + | | Project overview | access, then 17 calls, one of which (Active branches) read the default branch's last 120 commits and up to 10 branches' last 40 | one round; Active branches streams in with a skeleton, in one `branch_drift` call (below) | | |
| 239 | 241 | | Mission control | per project: open pulls, closed pulls and events (3 × up to 10), then `get_by_id` per unknown repository | one `pulls_for_repos` call for every project (one access check, one query), events per project alongside, one `readable` for the rest | | |
| 240 | 242 | | Issue, issues | access, then the rest | one round | | |
| 241 | 243 | ||
| ⋯ | |||
| 287 | 289 | waits for `allReady` for bots), and signed out it is kept in the public | |
| 288 | 290 | cache like any other project page. | |
| 289 | 291 | ||
| 292 | + | ### Active branches (2026-10-08) | |
| 293 | + | ||
| 294 | + | Measured on production, `/flagon-io/g1t` uncached, as a crawler, soon | |
| 295 | + | after pushes: **3,606 ms** to the first byte, `rpc` 3,605 ms over 38 | |
| 296 | + | service calls, **repos 25 calls, 17,953 ms inside**. Nearly all of it was | |
| 297 | + | Active branches: | |
| 298 | + | ||
| 299 | + | - The site measured each branch itself: a `log` call per branch per | |
| 300 | + | depth (40, then 1,000), plus the default branch's (120, then 1,000), plus | |
| 301 | + | one more for the default branch's head, each call with its own access | |
| 302 | + | check and store handle. The answer was kept by the pair of heads in the | |
| 303 | + | site's cache, so every push to a branch, and every merge to the default | |
| 304 | + | branch (which changes every pair), started the walks again, in every | |
| 305 | + | data centre. | |
| 306 | + | - A branch far from the default branch read 1,000 commits of both, and | |
| 307 | + | the default branch's 1,000 again after each merge. | |
| 308 | + | - The page waits 3.5 s for the section, then shows a link to Branches. | |
| 309 | + | The walks were not in `waitUntil`, so when the page stopped waiting | |
| 310 | + | they were dropped with it: an answer that took longer than 3.5 s was | |
| 311 | + | never kept, and the next view started over. 3,606 ms is that timeout. | |
| 312 | + | ||
| 313 | + | Now: | |
| 314 | + | ||
| 315 | + | - **One call.** `branch_drift` (services/repos/src/drift.rs) takes the | |
| 316 | + | default branch's head and every branch head, checks access once, opens | |
| 317 | + | the store once, and reads the default branch's history once per depth | |
| 318 | + | for all of them. Each answer is kept in repos' data-centre cache by the | |
| 319 | + | pair of hashes; only pairs that changed are walked. It also returns the | |
| 320 | + | default branch's head commit, which the site read with its own call. | |
| 321 | + | - **Shallower first.** Depths are (branch, default branch) 12/120, then | |
| 322 | + | 40/120, 40/1,000 and 1,000/1,000: most branches are a few commits | |
| 323 | + | ahead and meet at the first. | |
| 324 | + | - **Long histories spliced.** A log by hash of 100 commits or more is a | |
| 325 | + | 16-commit read plus the log kept from one of those commits (the | |
| 326 | + | first-parent chain from a commit never changes), so the default branch | |
| 327 | + | after a merge costs 16 commits instead of 120 or 1,000. | |
| 328 | + | - **Finished after the page.** The call runs in `waitUntil`, so repos | |
| 329 | + | keeps the answer even when the page stopped waiting for it. | |
| 330 | + | - **Crawlers wait 0.7 s** for the section (browsers 3.5 s, streamed); | |
| 331 | + | past that they get the link to Branches, as a slow browser does. | |
| 332 | + | - `tags` is kept until the refs move (it listed the refs from the store | |
| 333 | + | on every call), and the project is looked up once per request for the | |
| 334 | + | layout and the overview (`projectFor`, beside `repoFor`). | |
| 335 | + | - `tags`, `commit_checks`, `shortcuts` and the Files page's reads were | |
| 336 | + | missing from `READS`, so every signed-in overview counted as a write: | |
| 337 | + | it set `g1t_d1`, sent the next 30 seconds of reads to the primary and | |
| 338 | + | turned off the sidebar cache. | |
| 339 | + | ||
| 340 | + | | Overview, signed out, uncached | Before | After | | |
| 341 | + | | --- | --- | --- | | |
| 342 | + | | repos calls | 25 (6 when nothing had moved) | 6 whatever moved: `get`, `stars`, `branches`, `log`, `tags`, `branch_drift` | | |
| 343 | + | | projects calls | 3 (`get` twice) | 2 | | |
| 344 | + | | Crawler, after a push | 3,606 ms (the 3.5 s timeout) | at most about 0.8 s: the rest of the page, or 0.7 s for Active branches | | |
| 345 | + | | Crawler, nothing moved | 460 to 570 ms | not yet measured on production | | |
| 346 | + | | Browser, first byte | 115 to 160 ms (`total`), 200 to 245 ms measured from Colorado | unchanged: the page does not wait for any of this | | |
| 347 | + | ||
| 290 | 348 | ## Client navigation | |
| 291 | 349 | ||
| 292 | 350 | - `<Link prefetch="intent">` on the sidebar, project tabs, breadcrumbs, | |
| ⋯ | |||
| 341 | 399 | It prints p50 and p90 of the server's share (TLS handshake done to first | |
| 342 | 400 | byte), where the Worker ran, whether the answer set `g1t_d1` (it should | |
| 343 | 401 | not, for a page that only reads), and the slowest Server-Timing entries. | |
| 402 | + | ||
| 403 | + | Signed out, a public page is usually answered from the data centre's | |
| 404 | + | cache (`server-timing: cache;desc="hit, …"`), and `cache-control: | |
| 405 | + | no-cache` does not change that. To time a render, add a query string the | |
| 406 | + | page ignores: the cache is keyed by the whole URL, so | |
| 407 | + | `/flagon-io/g1t?nc=<random>` is always a miss. A crawler's user agent | |
| 408 | + | (`Googlebot/2.1`) waits for the whole page; a browser's gets the first | |
| 409 | + | byte and the streamed rest. | |
| 367 | 367 | call("git_access", { path, viewer, service }), | |
| 368 | 368 | branches: (path, viewer) => call("branches", { path, viewer }), | |
| 369 | 369 | lastCommits: (path, viewer, ref, treePath) => call("last_commits", { path, viewer, ref, treePath }), | |
| 370 | + | branchDrift: (path, viewer, base, heads) => call("branch_drift", { path, viewer, base, heads }), | |
| 370 | 371 | tags: (path, viewer) => call("tags", { path, viewer }), | |
| 371 | 372 | about: (path, viewer) => call("about", { path, viewer }), | |
| 372 | 373 | languages: (path, viewer) => call("languages", { path, viewer }), |
| 246 | 246 | /** Which commit last changed each entry of a directory at `ref` (the default branch when null). */ | |
| 247 | 247 | lastCommits(path: RepoPath, viewer: Viewer, ref: string | null, treePath: string): Promise<Result<LastCommits>>; | |
| 248 | 248 | ||
| 249 | + | /** | |
| 250 | + | * How far each branch head has moved from `base` (the default branch's | |
| 251 | + | * head commit), with each head's commit and `base`'s own, in one call. | |
| 252 | + | * Kept by the pair of hashes in the repos service. | |
| 253 | + | */ | |
| 254 | + | branchDrift(path: RepoPath, viewer: Viewer, base: string, heads: string[]): Promise<Result<BranchDrifts>>; | |
| 255 | + | ||
| 249 | 256 | /** The repository's tags, newest commit first, at most 100. */ | |
| 250 | 257 | tags(path: RepoPath, viewer: Viewer): Promise<Result<Tag[]>>; | |
| 251 | 258 | ||
| ⋯ | |||
| 372 | 379 | /** Each entry's last commit; `complete` is false when some were not reached. */ | |
| 373 | 380 | export type LastCommits = { entries: { name: string; commit: Commit }[]; complete: boolean }; | |
| 374 | 381 | ||
| 382 | + | /** Commits a branch has that the default branch does not, and the other way round. */ | |
| 383 | + | export type BranchDriftCount = { ahead: number; behind: number }; | |
| 384 | + | ||
| 385 | + | /** | |
| 386 | + | * `base`'s commit, and each head's commit and drift in the order asked; | |
| 387 | + | * `drift` is null when the two histories do not meet within what is read. | |
| 388 | + | */ | |
| 389 | + | export type BranchDrifts = { | |
| 390 | + | base: Commit | null; | |
| 391 | + | branches: { head: string; commit: Commit | null; drift: BranchDriftCount | null }[]; | |
| 392 | + | }; | |
| 393 | + | ||
| 375 | 394 | /** A tag, and its commit when it could be read. */ | |
| 376 | 395 | export type Tag = { name: string; commit: Commit | null }; | |
| 377 | 396 | ||
| 1 | + | //! How far branches have moved from the default branch, for Active branches | |
| 2 | + | //! on a project's overview and the Branches page (`branch_drift`). | |
| 3 | + | //! | |
| 4 | + | //! Each branch head's history and the default branch's are read to a depth, | |
| 5 | + | //! in turn deeper, until they meet; the default branch's history is read | |
| 6 | + | //! once per depth for every branch. Histories are read by commit hash, which | |
| 7 | + | //! the store keeps for good (store.rs), and each answer is kept by the pair | |
| 8 | + | //! of heads (lib.rs), so only heads that moved cost a walk. Before | |
| 9 | + | //! 2026-10-08 the site did this itself: up to twenty `log` calls per view, | |
| 10 | + | //! each with its own access check and store handle. | |
| 11 | + | ||
| 12 | + | use std::collections::{HashMap, HashSet}; | |
| 13 | + | ||
| 14 | + | use g1t_contracts::repos::{Commit, Drift}; | |
| 15 | + | use worker::Result; | |
| 16 | + | ||
| 17 | + | use crate::store::GitRepo; | |
| 18 | + | ||
| 19 | + | /// How deep each history is read, in turn: (branch, default branch). Most | |
| 20 | + | /// branches are a few commits ahead of where they left a default branch | |
| 21 | + | /// that has moved on a little; one left long ago needs the default | |
| 22 | + | /// branch's history further back; one far from both reads both deeply. | |
| 23 | + | /// Past the last, there is no answer. | |
| 24 | + | pub const DEPTHS: [(u32, u32); 4] = [(12, 120), (40, 120), (40, 1000), (1000, 1000)]; | |
| 25 | + | ||
| 26 | + | /// Branches read at once. | |
| 27 | + | const AT_ONCE: usize = 8; | |
| 28 | + | ||
| 29 | + | /// A branch head's commit and drift, and whether the answer may be kept: | |
| 30 | + | /// not when a read failed. | |
| 31 | + | #[derive(Clone, Debug)] | |
| 32 | + | pub struct Measured { | |
| 33 | + | pub commit: Option<Commit>, | |
| 34 | + | pub drift: Option<Drift>, | |
| 35 | + | pub settled: bool, | |
| 36 | + | } | |
| 37 | + | ||
| 38 | + | /// Commits `branch` has that `main` does not (ahead) and the other way | |
| 39 | + | /// round (behind), from what was read of the two histories (`commits`, in | |
| 40 | + | /// any order, repeats allowed). `None` when either head is missing, or when | |
| 41 | + | /// a commit only one side reaches has a parent that was not read: that | |
| 42 | + | /// parent's history could change either count. | |
| 43 | + | pub fn drift<'a>(branch: &str, main: &str, commits: impl IntoIterator<Item = &'a Commit>) -> Option<Drift> { | |
| 44 | + | let mut parents: HashMap<&str, &[String]> = HashMap::new(); | |
| 45 | + | for commit in commits { | |
| 46 | + | parents.insert(commit.hash.as_str(), commit.parents.as_slice()); | |
| 47 | + | } | |
| 48 | + | if !parents.contains_key(branch) || !parents.contains_key(main) { | |
| 49 | + | return None; | |
| 50 | + | } | |
| 51 | + | let from_branch = reach(branch, &parents); | |
| 52 | + | let from_main = reach(main, &parents); | |
| 53 | + | let (mut ahead, mut behind) = (0, 0); | |
| 54 | + | for (hash, above) in &parents { | |
| 55 | + | let on_branch = from_branch.contains(hash); | |
| 56 | + | if on_branch == from_main.contains(hash) { | |
| 57 | + | continue; | |
| 58 | + | } | |
| 59 | + | if above.iter().any(|parent| !parents.contains_key(parent.as_str())) { | |
| 60 | + | return None; | |
| 61 | + | } | |
| 62 | + | if on_branch { | |
| 63 | + | ahead += 1; | |
| 64 | + | } else { | |
| 65 | + | behind += 1; | |
| 66 | + | } | |
| 67 | + | } | |
| 68 | + | Some(Drift { ahead, behind }) | |
| 69 | + | } | |
| 70 | + | ||
| 71 | + | /// Every commit read that `head` descends from, itself included. | |
| 72 | + | fn reach<'a>(head: &'a str, parents: &HashMap<&'a str, &'a [String]>) -> HashSet<&'a str> { | |
| 73 | + | let mut seen = HashSet::from([head]); | |
| 74 | + | let mut next = vec![head]; | |
| 75 | + | while let Some(hash) = next.pop() { | |
| 76 | + | for parent in parents.get(hash).copied().unwrap_or_default() { | |
| 77 | + | if let Some((&known, _)) = parents.get_key_value(parent.as_str()) | |
| 78 | + | && seen.insert(known) | |
| 79 | + | { | |
| 80 | + | next.push(known); | |
| 81 | + | } | |
| 82 | + | } | |
| 83 | + | } | |
| 84 | + | seen | |
| 85 | + | } | |
| 86 | + | ||
| 87 | + | /// What was read of one branch's history so far. | |
| 88 | + | struct Reading { | |
| 89 | + | commits: Vec<Commit>, | |
| 90 | + | depth: u32, | |
| 91 | + | answer: Option<Measured>, | |
| 92 | + | } | |
| 93 | + | ||
| 94 | + | /// Each of `heads` measured against `base`, in the order given. | |
| 95 | + | pub async fn measure<R: GitRepo>(git: &R, base: &str, heads: &[String]) -> Vec<Measured> { | |
| 96 | + | let mut readings: Vec<Reading> = heads.iter().map(|_| Reading { commits: Vec::new(), depth: 0, answer: None }).collect(); | |
| 97 | + | // The default branch's history, read once per depth and not deeper | |
| 98 | + | // once a read reached its start. | |
| 99 | + | let mut main: Option<(u32, Vec<Commit>)> = None; | |
| 100 | + | let mut main_failed = false; | |
| 101 | + | for (branch_depth, main_depth) in DEPTHS { | |
| 102 | + | if readings.iter().all(|reading| reading.answer.is_some()) { | |
| 103 | + | break; | |
| 104 | + | } | |
| 105 | + | let deeper = match &main { | |
| 106 | + | Some((read_to, commits)) => *read_to < main_depth && commits.len() as u32 >= *read_to, | |
| 107 | + | None => true, | |
| 108 | + | }; | |
| 109 | + | if deeper && !main_failed { | |
| 110 | + | match git.log(base, main_depth).await { | |
| 111 | + | Ok(commits) if !commits.is_empty() => main = Some((main_depth, commits)), | |
| 112 | + | _ => main_failed = true, | |
| 113 | + | } | |
| 114 | + | } | |
| 115 | + | let Some((_, main_commits)) = &main else { | |
| 116 | + | for reading in readings.iter_mut().filter(|reading| reading.answer.is_none()) { | |
| 117 | + | reading.answer = Some(Measured { commit: None, drift: None, settled: false }); | |
| 118 | + | } | |
| 119 | + | break; | |
| 120 | + | }; | |
| 121 | + | let open: Vec<usize> = (0..heads.len()).filter(|&index| readings[index].answer.is_none()).collect(); | |
| 122 | + | for chunk in open.chunks(AT_ONCE) { | |
| 123 | + | let reads = futures_util::future::join_all(chunk.iter().map(|&index| { | |
| 124 | + | let reading = &readings[index]; | |
| 125 | + | // Read again only when deeper, and only when the last read | |
| 126 | + | // did not already reach the start. | |
| 127 | + | let again = reading.depth == 0 || (branch_depth > reading.depth && reading.commits.len() as u32 >= reading.depth); | |
| 128 | + | let head = heads[index].as_str(); | |
| 129 | + | async move { if again { Some(git.log(head, branch_depth).await) } else { None } } | |
| 130 | + | })) | |
| 131 | + | .await; | |
| 132 | + | for (&index, read) in chunk.iter().zip(reads) { | |
| 133 | + | let reading = &mut readings[index]; | |
| 134 | + | match read { | |
| 135 | + | Some(Ok(commits)) if !commits.is_empty() => { | |
| 136 | + | reading.commits = commits; | |
| 137 | + | reading.depth = branch_depth; | |
| 138 | + | } | |
| 139 | + | Some(_) => { | |
| 140 | + | reading.answer = Some(Measured { commit: None, drift: None, settled: false }); | |
| 141 | + | continue; | |
| 142 | + | } | |
| 143 | + | None => {} | |
| 144 | + | } | |
| 145 | + | let head = heads[index].as_str(); | |
| 146 | + | if let Some(counted) = drift(head, base, main_commits.iter().chain(reading.commits.iter())) { | |
| 147 | + | reading.answer = Some(Measured { commit: reading.commits.first().cloned(), drift: Some(counted), settled: true }); | |
| 148 | + | } | |
| 149 | + | } | |
| 150 | + | } | |
| 151 | + | if main_failed { | |
| 152 | + | break; | |
| 153 | + | } | |
| 154 | + | } | |
| 155 | + | readings | |
| 156 | + | .into_iter() | |
| 157 | + | .map(|reading| { | |
| 158 | + | reading.answer.unwrap_or_else(|| Measured { | |
| 159 | + | commit: reading.commits.first().cloned(), | |
| 160 | + | drift: None, | |
| 161 | + | // Read to the last depth without meeting: that is the answer | |
| 162 | + | // for this pair, and it will not change. | |
| 163 | + | settled: !main_failed && reading.depth > 0, | |
| 164 | + | }) | |
| 165 | + | }) | |
| 166 | + | .collect() | |
| 167 | + | } | |
| 168 | + | ||
| 169 | + | #[cfg(test)] | |
| 170 | + | mod tests { | |
| 171 | + | use std::cell::RefCell; | |
| 172 | + | use std::future::Future; | |
| 173 | + | use std::pin::pin; | |
| 174 | + | use std::task::{Context, Poll, Waker}; | |
| 175 | + | ||
| 176 | + | use g1t_contracts::repos::{Branch, GitAccess, Signature, TreeEntry}; | |
| 177 | + | ||
| 178 | + | use super::*; | |
| 179 | + | use crate::store::Scope; | |
| 180 | + | ||
| 181 | + | fn run<F: Future>(future: F) -> F::Output { | |
| 182 | + | match pin!(future).as_mut().poll(&mut Context::from_waker(Waker::noop())) { | |
| 183 | + | Poll::Ready(output) => output, | |
| 184 | + | Poll::Pending => panic!("the fake store never waits"), | |
| 185 | + | } | |
| 186 | + | } | |
| 187 | + | ||
| 188 | + | fn commit(hash: &str, parents: &[&str]) -> Commit { | |
| 189 | + | Commit { | |
| 190 | + | hash: hash.into(), | |
| 191 | + | tree_hash: format!("t{hash}"), | |
| 192 | + | message: format!("commit {hash}\n\nbody"), | |
| 193 | + | author: Signature { name: "a".into(), email: "a@example.com".into() }, | |
| 194 | + | parents: parents.iter().map(|&p| p.to_owned()).collect(), | |
| 195 | + | authored_at: String::new(), | |
| 196 | + | } | |
| 197 | + | } | |
| 198 | + | ||
| 199 | + | /// Commits by hash; `log` follows first parents. Counts each read. | |
| 200 | + | #[derive(Default)] | |
| 201 | + | struct Fake { | |
| 202 | + | commits: HashMap<String, Commit>, | |
| 203 | + | reads: RefCell<Vec<(String, u32)>>, | |
| 204 | + | fail: Option<String>, | |
| 205 | + | } | |
| 206 | + | ||
| 207 | + | impl Fake { | |
| 208 | + | fn with(commits: Vec<Commit>) -> Fake { | |
| 209 | + | Fake { commits: commits.into_iter().map(|c| (c.hash.clone(), c)).collect(), ..Fake::default() } | |
| 210 | + | } | |
| 211 | + | } | |
| 212 | + | ||
| 213 | + | impl GitRepo for Fake { | |
| 214 | + | async fn access(&self, _scope: Scope) -> Result<GitAccess> { | |
| 215 | + | unimplemented!() | |
| 216 | + | } | |
| 217 | + | async fn branches(&self) -> Result<Vec<Branch>> { | |
| 218 | + | Ok(Vec::new()) | |
| 219 | + | } | |
| 220 | + | async fn log(&self, git_ref: &str, limit: u32) -> Result<Vec<Commit>> { | |
| 221 | + | self.reads.borrow_mut().push((git_ref.to_owned(), limit)); | |
| 222 | + | if self.fail.as_deref() == Some(git_ref) { | |
| 223 | + | return Err(worker::Error::RustError("store busy".into())); | |
| 224 | + | } | |
| 225 | + | let mut out = Vec::new(); | |
| 226 | + | let mut at = self.commits.get(git_ref); | |
| 227 | + | while let Some(commit) = at { | |
| 228 | + | if out.len() as u32 >= limit { | |
| 229 | + | break; | |
| 230 | + | } | |
| 231 | + | out.push(commit.clone()); | |
| 232 | + | at = commit.parents.first().and_then(|parent| self.commits.get(parent)); | |
| 233 | + | } | |
| 234 | + | Ok(out) | |
| 235 | + | } | |
| 236 | + | async fn parents(&self, _commit_hash: &str) -> Result<Option<Vec<String>>> { | |
| 237 | + | Ok(None) | |
| 238 | + | } | |
| 239 | + | async fn read_tree(&self, _tree_hash: &str) -> Result<Option<Vec<TreeEntry>>> { | |
| 240 | + | Ok(None) | |
| 241 | + | } | |
| 242 | + | async fn read_blob(&self, _blob_hash: &str) -> Result<Option<Vec<u8>>> { | |
| 243 | + | Ok(None) | |
| 244 | + | } | |
| 245 | + | async fn read_file(&self, _git_ref: &str, _path: &str) -> Result<Option<Vec<u8>>> { | |
| 246 | + | Ok(None) | |
| 247 | + | } | |
| 248 | + | async fn fork(&self, _target_key: &str) -> Result<()> { | |
| 249 | + | Ok(()) | |
| 250 | + | } | |
| 251 | + | } | |
| 252 | + | ||
| 253 | + | /// m1 ← m2 ← m3 on main; b1 ← b2 branched from m2. | |
| 254 | + | fn forked() -> Vec<Commit> { | |
| 255 | + | vec![commit("m1", &[]), commit("m2", &["m1"]), commit("m3", &["m2"]), commit("b1", &["m2"]), commit("b2", &["b1"])] | |
| 256 | + | } | |
| 257 | + | ||
| 258 | + | /// A straight line of `n` commits named `{prefix}{i}`, the first on `from`. | |
| 259 | + | fn line(prefix: &str, from: Option<&str>, n: usize) -> Vec<Commit> { | |
| 260 | + | (1..=n) | |
| 261 | + | .map(|i| { | |
| 262 | + | let parent = if i == 1 { from.map(str::to_owned) } else { Some(format!("{prefix}{}", i - 1)) }; | |
| 263 | + | commit(&format!("{prefix}{i}"), &parent.iter().map(String::as_str).collect::<Vec<_>>()) | |
| 264 | + | }) | |
| 265 | + | .collect() | |
| 266 | + | } | |
| 267 | + | ||
| 268 | + | #[test] | |
| 269 | + | fn counts_both_sides_from_where_they_forked() { | |
| 270 | + | assert_eq!(drift("b2", "m3", &forked()), Some(Drift { ahead: 2, behind: 1 })); | |
| 271 | + | assert_eq!(drift("m3", "m3", &forked()), Some(Drift { ahead: 0, behind: 0 })); | |
| 272 | + | } | |
| 273 | + | ||
| 274 | + | #[test] | |
| 275 | + | fn a_merge_from_main_is_not_ahead() { | |
| 276 | + | // b3 merges m3 into the branch. | |
| 277 | + | let mut history = forked(); | |
| 278 | + | history.push(commit("b3", &["b2", "m3"])); | |
| 279 | + | assert_eq!(drift("b3", "m3", &history), Some(Drift { ahead: 3, behind: 0 })); | |
| 280 | + | } | |
| 281 | + | ||
| 282 | + | #[test] | |
| 283 | + | fn no_answer_when_the_histories_were_not_read_far_enough() { | |
| 284 | + | let history = vec![commit("m3", &["m2"]), commit("b2", &["b1"])]; | |
| 285 | + | assert_eq!(drift("b2", "m3", &history), None); | |
| 286 | + | assert_eq!(drift("b9", "m3", &forked()), None); | |
| 287 | + | } | |
| 288 | + | ||
| 289 | + | #[test] | |
| 290 | + | fn measures_every_head_with_one_read_of_main() { | |
| 291 | + | let git = Fake::with(forked()); | |
| 292 | + | let found = run(measure(&git, "m3", &["b2".to_owned(), "m2".to_owned(), "m3".to_owned()])); | |
| 293 | + | assert_eq!(found[0].drift, Some(Drift { ahead: 2, behind: 1 })); | |
| 294 | + | assert_eq!(found[0].commit.as_ref().map(|c| c.hash.as_str()), Some("b2")); | |
| 295 | + | assert_eq!(found[1].drift, Some(Drift { ahead: 0, behind: 1 })); | |
| 296 | + | assert_eq!(found[2].drift, Some(Drift { ahead: 0, behind: 0 })); | |
| 297 | + | assert!(found.iter().all(|m| m.settled)); | |
| 298 | + | let reads = git.reads.borrow(); | |
| 299 | + | assert_eq!(reads.iter().filter(|(hash, _)| hash == "m3").count(), 2, "main once, plus m3 as a head: {reads:?}"); | |
| 300 | + | assert!(reads.iter().all(|(_, depth)| *depth == 12 || *depth == 120)); | |
| 301 | + | } | |
| 302 | + | ||
| 303 | + | #[test] | |
| 304 | + | fn reads_deeper_only_for_a_branch_that_needs_it() { | |
| 305 | + | // main: 150 commits; "long" is 30 ahead of m100 (50 behind); "short" is 1 ahead of m149. | |
| 306 | + | let mut history = line("m", None, 150); | |
| 307 | + | history.extend(line("l", Some("m100"), 30)); | |
| 308 | + | history.push(commit("s1", &["m149"])); | |
| 309 | + | let git = Fake::with(history); | |
| 310 | + | let found = run(measure(&git, "m150", &["l30".to_owned(), "s1".to_owned()])); | |
| 311 | + | assert_eq!(found[0].drift, Some(Drift { ahead: 30, behind: 50 })); | |
| 312 | + | assert_eq!(found[1].drift, Some(Drift { ahead: 1, behind: 1 })); | |
| 313 | + | let reads = git.reads.borrow(); | |
| 314 | + | assert_eq!(reads.iter().filter(|(hash, _)| hash == "s1").count(), 1); | |
| 315 | + | assert_eq!(*reads.iter().filter(|(hash, _)| hash == "l30").map(|(_, depth)| depth).max().unwrap(), 40); | |
| 316 | + | assert!(!reads.iter().any(|(_, depth)| *depth == 1000), "{reads:?}"); | |
| 317 | + | } | |
| 318 | + | ||
| 319 | + | #[test] | |
| 320 | + | fn unrelated_histories_read_to_their_start_count_every_commit() { | |
| 321 | + | let mut history = line("m", None, 3); | |
| 322 | + | history.extend(line("x", None, 2)); | |
| 323 | + | let git = Fake::with(history); | |
| 324 | + | let found = run(measure(&git, "m3", &["x2".to_owned()])); | |
| 325 | + | assert_eq!(found[0].drift, Some(Drift { ahead: 2, behind: 3 })); | |
| 326 | + | assert_eq!(found[0].commit.as_ref().map(|c| c.hash.as_str()), Some("x2")); | |
| 327 | + | assert!(found[0].settled); | |
| 328 | + | // Both reached their start at the first depth: never read again. | |
| 329 | + | assert_eq!(git.reads.borrow().len(), 2); | |
| 330 | + | } | |
| 331 | + | ||
| 332 | + | #[test] | |
| 333 | + | fn past_the_last_depth_the_answer_is_settled_without_a_count() { | |
| 334 | + | // The branch is 1,200 commits long: no depth reaches where it left main. | |
| 335 | + | let mut history = line("m", None, 3); | |
| 336 | + | history.extend(line("x", Some("m1"), 1200)); | |
| 337 | + | let git = Fake::with(history); | |
| 338 | + | let found = run(measure(&git, "m3", &["x1200".to_owned()])); | |
| 339 | + | assert_eq!(found[0].drift, None); | |
| 340 | + | assert_eq!(found[0].commit.as_ref().map(|c| c.hash.as_str()), Some("x1200")); | |
| 341 | + | assert!(found[0].settled); | |
| 342 | + | } | |
| 343 | + | ||
| 344 | + | #[test] | |
| 345 | + | fn a_failed_read_is_not_kept() { | |
| 346 | + | let mut git = Fake::with(forked()); | |
| 347 | + | git.fail = Some("b2".into()); | |
| 348 | + | let found = run(measure(&git, "m3", &["b2".to_owned(), "b1".to_owned()])); | |
| 349 | + | assert!(!found[0].settled); | |
| 350 | + | assert_eq!(found[1].drift, Some(Drift { ahead: 1, behind: 1 })); | |
| 351 | + | assert!(found[1].settled); | |
| 352 | + | git.fail = Some("m3".into()); | |
| 353 | + | let found = run(measure(&git, "m3", &["b2".to_owned()])); | |
| 354 | + | assert!(!found[0].settled); | |
| 355 | + | } | |
| 356 | + | } |
| 13 | 13 | mod commit_file; | |
| 14 | 14 | mod contributors; | |
| 15 | 15 | mod diff; | |
| 16 | + | mod drift; | |
| 16 | 17 | mod fallback; | |
| 17 | 18 | mod forks; | |
| 18 | 19 | mod git_http; | |
| ⋯ | |||
| 74 | 75 | const MAX_ANCESTRY: u32 = 1000; | |
| 75 | 76 | /// The most tags a repository's Tags page reads and lists. | |
| 76 | 77 | const MAX_TAGS_READ: usize = 100; | |
| 78 | + | /// Branch heads measured in one `branch_drift` call. | |
| 79 | + | const MAX_DRIFT_HEADS: usize = 100; | |
| 77 | 80 | ||
| 78 | 81 | /// One path segment, percent-encoded for a cache key. | |
| 79 | 82 | fn urlencoding_segment(segment: &str) -> String { | |
| ⋯ | |||
| 941 | 944 | Ok(Outcome::Ok(found)) | |
| 942 | 945 | } | |
| 943 | 946 | ||
| 947 | + | /// How far each branch head has moved from the default branch's head, | |
| 948 | + | /// in one call (drift.rs). Each answer is kept in this colo's cache by | |
| 949 | + | /// repository and the pair of hashes, for good: neither history can | |
| 950 | + | /// change. A head that moved is the only one walked. | |
| 951 | + | async fn branch_drift(&self, a: BranchDriftArgs) -> Result<Outcome<BranchDrifts>> { | |
| 952 | + | let Some(repo) = self.readable(&a.path, &a.viewer).await? else { | |
| 953 | + | return Ok(not_found()); | |
| 954 | + | }; | |
| 955 | + | if !store::is_commit_hash(&a.base) { | |
| 956 | + | return Ok(Outcome::fail(FailureCode::Invalid, "The default branch's head is a full commit hash.")); | |
| 957 | + | } | |
| 958 | + | let heads: Vec<String> = a.heads.into_iter().take(MAX_DRIFT_HEADS).collect(); | |
| 959 | + | let git = self.read_git(&repo).await?; | |
| 960 | + | let key = |head: &str| format!("https://drift.g1t.internal/{}/{}/{head}", repo.id, a.base); | |
| 961 | + | let cache = worker::Cache::default(); | |
| 962 | + | let (base, kept) = futures_util::future::join( | |
| 963 | + | git.log(&a.base, 1), | |
| 964 | + | futures_util::future::join_all(heads.iter().map(|head| { | |
| 965 | + | let (cache, url) = (&cache, key(head)); | |
| 966 | + | async move { | |
| 967 | + | if !store::is_commit_hash(head) { | |
| 968 | + | return None; | |
| 969 | + | } | |
| 970 | + | let mut found = cache.get(url.as_str(), false).await.ok()??; | |
| 971 | + | found.json::<BranchDrift>().await.ok() | |
| 972 | + | } | |
| 973 | + | })), | |
| 974 | + | ) | |
| 975 | + | .await; | |
| 976 | + | let missing: Vec<String> = heads | |
| 977 | + | .iter() | |
| 978 | + | .zip(&kept) | |
| 979 | + | .filter(|(head, kept)| kept.is_none() && store::is_commit_hash(head)) | |
| 980 | + | .map(|(head, _)| head.clone()) | |
| 981 | + | .collect(); | |
| 982 | + | let measured = drift::measure(&git, &a.base, &missing).await; | |
| 983 | + | let mut fresh: HashMap<String, BranchDrift> = HashMap::new(); | |
| 984 | + | for (head, found) in missing.into_iter().zip(measured) { | |
| 985 | + | let answer = BranchDrift { head: head.clone(), commit: found.commit, drift: found.drift }; | |
| 986 | + | if found.settled | |
| 987 | + | && let Ok(mut response) = worker::Response::from_json(&answer) | |
| 988 | + | { | |
| 989 | + | let _ = response.headers_mut().set("cache-control", "public, max-age=31536000, immutable"); | |
| 990 | + | let _ = cache.put(key(&head).as_str(), response).await; | |
| 991 | + | } | |
| 992 | + | fresh.insert(head, answer); | |
| 993 | + | } | |
| 994 | + | let branches = heads | |
| 995 | + | .iter() | |
| 996 | + | .zip(kept) | |
| 997 | + | .map(|(head, kept)| { | |
| 998 | + | kept.or_else(|| fresh.get(head).cloned()) | |
| 999 | + | .unwrap_or_else(|| BranchDrift { head: head.clone(), commit: None, drift: None }) | |
| 1000 | + | }) | |
| 1001 | + | .collect(); | |
| 1002 | + | Ok(Outcome::Ok(BranchDrifts { base: base.ok().and_then(|log| log.into_iter().next()), branches })) | |
| 1003 | + | } | |
| 1004 | + | ||
| 944 | 1005 | /// The repository's tags, newest commit first, at most 100. | |
| 945 | 1006 | async fn tags(&self, a: g1t_contracts::repos::TagsArgs) -> Result<Outcome<Vec<g1t_contracts::repos::Tag>>> { | |
| 946 | 1007 | let Some(repo) = self.readable(&a.path, &a.viewer).await? else { | |
| 947 | 1008 | return Ok(not_found()); | |
| 948 | 1009 | }; | |
| 949 | − | let git = self.store.open(&store_key(&repo)).await?; | |
| 1010 | + | // Kept until the refs move, as the branch list is (store.rs): listing | |
| 1011 | + | // the refs is a round trip to the store on every call otherwise. | |
| 1012 | + | let version = refs_cache::usable(registry::refs_state(&repo.id), now_ms()).filter(|_| !self.store.on_fallback(&store_key(&repo))); | |
| 1013 | + | let kept_at = version.map(|version| format!("https://tags.g1t.internal/{}/{version}", repo.id)); | |
| 1014 | + | if let Some(url) = &kept_at | |
| 1015 | + | && let Ok(Some(mut kept)) = worker::Cache::default().get(url.as_str(), false).await | |
| 1016 | + | && let Ok(tags) = kept.json::<Vec<g1t_contracts::repos::Tag>>().await | |
| 1017 | + | { | |
| 1018 | + | return Ok(Outcome::Ok(tags)); | |
| 1019 | + | } | |
| 1020 | + | let (tags, complete) = self.read_tags(&repo).await?; | |
| 1021 | + | if complete | |
| 1022 | + | && let Some(url) = &kept_at | |
| 1023 | + | && let Ok(mut response) = worker::Response::from_json(&tags) | |
| 1024 | + | { | |
| 1025 | + | let _ = response.headers_mut().set("cache-control", "public, max-age=300"); | |
| 1026 | + | let _ = worker::Cache::default().put(url.as_str(), response).await; | |
| 1027 | + | } | |
| 1028 | + | Ok(Outcome::Ok(tags)) | |
| 1029 | + | } | |
| 1030 | + | ||
| 1031 | + | /// The tags, and whether every one's commit was read (only then kept). | |
| 1032 | + | async fn read_tags(&self, repo: &Repo) -> Result<(Vec<g1t_contracts::repos::Tag>, bool)> { | |
| 1033 | + | let git = self.store.open(&store_key(repo)).await?; | |
| 950 | 1034 | let access = git.access(Scope::Read).await?; | |
| 951 | 1035 | let named: Vec<(String, String)> = refs::heads_and_tags(refs::all(&access).await?) | |
| 952 | 1036 | .into_iter() | |
| 953 | 1037 | .filter_map(|(name, hash)| name.strip_prefix("refs/tags/").map(|tag| (tag.to_owned(), hash))) | |
| 954 | 1038 | .collect(); | |
| 955 | − | let read = self.read_git(&repo).await?; | |
| 1039 | + | let read = self.read_git(repo).await?; | |
| 956 | 1040 | let commits = futures_util::future::join_all(named.iter().take(MAX_TAGS_READ).map(|(_, hash)| read.log(hash, 1))).await; | |
| 1041 | + | let complete = commits.iter().all(Result::is_ok); | |
| 957 | 1042 | let mut tags: Vec<g1t_contracts::repos::Tag> = named | |
| 958 | 1043 | .into_iter() | |
| 959 | 1044 | .zip(commits.into_iter().map(|found| found.ok().and_then(|list| list.into_iter().next())).chain(std::iter::repeat(None))) | |
| ⋯ | |||
| 964 | 1049 | at(b).cmp(&at(a)).then_with(|| b.name.cmp(&a.name)) | |
| 965 | 1050 | }); | |
| 966 | 1051 | tags.truncate(MAX_TAGS_READ); | |
| 967 | − | Ok(Outcome::Ok(tags)) | |
| 1052 | + | Ok((tags, complete)) | |
| 968 | 1053 | } | |
| 969 | 1054 | ||
| 970 | 1055 | /// The repository's branches, default branch first. | |
| ⋯ | |||
| 2229 | 2314 | ||
| 2230 | 2315 | /// Read methods whose answer is an `Outcome`: when the git store is busy, | |
| 2231 | 2316 | /// the site is told so in words instead of failing the page. | |
| 2232 | − | const OUTCOME_READS: [&str; 6] = ["tree", "blob", "log", "branches", "blame", "compare"]; | |
| 2317 | + | const OUTCOME_READS: [&str; 7] = ["tree", "blob", "log", "branches", "blame", "compare", "branch_drift"]; | |
| 2233 | 2318 | ||
| 2234 | 2319 | #[event(fetch)] | |
| 2235 | 2320 | async fn fetch(mut request: Request, env: Env, ctx: Context) -> Result<Response> { | |
| ⋯ | |||
| 2327 | 2412 | "git_access" => reply(&repos.git_access(args(body)?).await?), | |
| 2328 | 2413 | "branches" => reply(&repos.branches(args(body)?).await?), | |
| 2329 | 2414 | "last_commits" => reply(&repos.last_commits(args(body)?).await?), | |
| 2415 | + | "branch_drift" => reply(&repos.branch_drift(args(body)?).await?), | |
| 2330 | 2416 | "tags" => reply(&repos.tags(args(body)?).await?), | |
| 2331 | 2417 | // The About: what the Files page shows beside the files (about.rs). | |
| 2332 | 2418 | // What is kept behind the head is worked out again after the answer. | |
| 733 | 733 | /// the store would not be. | |
| 734 | 734 | const ABSENT_MAX_AGE: &str = "public, max-age=600"; | |
| 735 | 735 | ||
| 736 | + | /// Histories by hash at least this long are put together from a short | |
| 737 | + | /// read and one kept before, where they can be (`spliced_log`). | |
| 738 | + | const SPLICE_FROM: u32 = 100; | |
| 739 | + | /// How many commits that short read takes. | |
| 740 | + | const SPLICE_PROBE: u32 = 16; | |
| 741 | + | ||
| 742 | + | /// `short[..at]` followed by `kept` (the history from `short[at]`), cut | |
| 743 | + | /// to `limit` commits. | |
| 744 | + | pub fn splice(short: &[Commit], at: usize, kept: Vec<Commit>, limit: u32) -> Vec<Commit> { | |
| 745 | + | let mut out: Vec<Commit> = short[..at.min(short.len())].to_vec(); | |
| 746 | + | out.extend(kept); | |
| 747 | + | out.truncate(limit as usize); | |
| 748 | + | out | |
| 749 | + | } | |
| 750 | + | ||
| 736 | 751 | /// Whether a ref is a full commit hash (SHA-1 or SHA-256), whose history | |
| 737 | 752 | /// can be kept for good. | |
| 738 | 753 | pub fn is_commit_hash(git_ref: &str) -> bool { | |
| ⋯ | |||
| 843 | 858 | found | |
| 844 | 859 | } | |
| 845 | 860 | ||
| 861 | + | /// A history from the store itself. | |
| 862 | + | async fn read_log(&self, git_ref: &str, limit: u32) -> Result<Vec<Commit>> { | |
| 863 | + | let options = js::to_js(&serde_json::json!({ "ref": git_ref, "limit": limit }))?; | |
| 864 | + | let raw: Vec<RawCommit> = js::from_js(&self.call("log", &[options], true).await?)?; | |
| 865 | + | Ok(raw | |
| 866 | + | .into_iter() | |
| 867 | + | .map(|commit| Commit { | |
| 868 | + | hash: commit.hash, | |
| 869 | + | tree_hash: commit.tree_hash, | |
| 870 | + | message: commit.message, | |
| 871 | + | author: commit.author, | |
| 872 | + | parents: commit.parents, | |
| 873 | + | authored_at: rfc3339(commit.authored_at * 1000), | |
| 874 | + | }) | |
| 875 | + | .collect()) | |
| 876 | + | } | |
| 877 | + | ||
| 878 | + | /// A long history by commit hash, from a short read and one kept | |
| 879 | + | /// before: when a branch moves a few commits, the history from its new | |
| 880 | + | /// head is those commits, then the history kept from its old one (the | |
| 881 | + | /// first-parent chain from a commit never changes). Only when none of | |
| 882 | + | /// the short read's commits has the same history kept is the whole of | |
| 883 | + | /// it read. A default branch that moved by a merge then costs a read | |
| 884 | + | /// of [`SPLICE_PROBE`] commits instead of a thousand. | |
| 885 | + | async fn spliced_log(&self, hash: &str, limit: u32) -> Result<Vec<Commit>> { | |
| 886 | + | let short = self.read_log(hash, SPLICE_PROBE).await?; | |
| 887 | + | if (short.len() as u32) < SPLICE_PROBE { | |
| 888 | + | // The whole history fits in the short read. | |
| 889 | + | return Ok(short); | |
| 890 | + | } | |
| 891 | + | let kept = futures_util::future::join_all(short.iter().enumerate().skip(1).map(|(at, commit)| async move { | |
| 892 | + | let Some(CacheKey::Forever(path)) = log_key(&commit.hash, limit, None) else { | |
| 893 | + | return None; | |
| 894 | + | }; | |
| 895 | + | let bytes = self.peek(&path).await?; | |
| 896 | + | serde_json::from_slice::<Vec<Commit>>(&bytes).ok().map(|kept| (at, kept)) | |
| 897 | + | })) | |
| 898 | + | .await; | |
| 899 | + | match kept.into_iter().flatten().next() { | |
| 900 | + | Some((at, kept)) => Ok(splice(&short, at, kept, limit)), | |
| 901 | + | None => self.read_log(hash, limit).await, | |
| 902 | + | } | |
| 903 | + | } | |
| 904 | + | ||
| 905 | + | /// A kept answer, if there is one, without counting a miss: the | |
| 906 | + | /// splice looks for many and expects most to be absent. | |
| 907 | + | async fn peek(&self, path: &str) -> Option<Vec<u8>> { | |
| 908 | + | let url = self.cache_url(path); | |
| 909 | + | if let Some(bytes) = MEMORY.with(|memory| memory.borrow().get(&url)) { | |
| 910 | + | meters::record("cache.memory_hit", &self.key, 0, bytes.len() as u64); | |
| 911 | + | return Some(bytes); | |
| 912 | + | } | |
| 913 | + | let mut response = worker::Cache::default().get(url.clone(), false).await.ok()??; | |
| 914 | + | let bytes = response.bytes().await.ok()?; | |
| 915 | + | meters::record("cache.edge_hit", &self.key, 0, bytes.len() as u64); | |
| 916 | + | MEMORY.with(|memory| memory.borrow_mut().put(url, &bytes)); | |
| 917 | + | Some(bytes) | |
| 918 | + | } | |
| 919 | + | ||
| 846 | 920 | async fn cached(&self, kind: &str, hash: &str) -> Option<Vec<u8>> { | |
| 847 | 921 | self.cached_at(&format!("{kind}/{hash}"), true).await | |
| 848 | 922 | } | |
| ⋯ | |||
| 1033 | 1107 | { | |
| 1034 | 1108 | return Ok(commits); | |
| 1035 | 1109 | } | |
| 1036 | − | let options = js::to_js(&serde_json::json!({ "ref": git_ref, "limit": limit }))?; | |
| 1037 | − | let raw: Vec<RawCommit> = js::from_js(&self.call("log", &[options], true).await?)?; | |
| 1038 | − | let commits: Vec<Commit> = raw | |
| 1039 | − | .into_iter() | |
| 1040 | − | .map(|commit| Commit { | |
| 1041 | − | hash: commit.hash, | |
| 1042 | − | tree_hash: commit.tree_hash, | |
| 1043 | − | message: commit.message, | |
| 1044 | − | author: commit.author, | |
| 1045 | − | parents: commit.parents, | |
| 1046 | − | authored_at: rfc3339(commit.authored_at * 1000), | |
| 1047 | − | }) | |
| 1048 | − | .collect(); | |
| 1110 | + | let commits = match &key { | |
| 1111 | + | Some(CacheKey::Forever(_)) if limit >= SPLICE_FROM => self.spliced_log(git_ref, limit).await?, | |
| 1112 | + | _ => self.read_log(git_ref, limit).await?, | |
| 1113 | + | }; | |
| 1049 | 1114 | // An unknown ref logs nothing; that is not kept, in case it arrives. | |
| 1050 | 1115 | if !commits.is_empty() | |
| 1051 | 1116 | && let Ok(bytes) = serde_json::to_vec(&commits) | |
| ⋯ | |||
| 1207 | 1272 | mod tests { | |
| 1208 | 1273 | use super::*; | |
| 1209 | 1274 | ||
| 1275 | + | fn chain(names: &[&str]) -> Vec<Commit> { | |
| 1276 | + | names | |
| 1277 | + | .iter() | |
| 1278 | + | .enumerate() | |
| 1279 | + | .map(|(at, name)| Commit { | |
| 1280 | + | hash: (*name).to_owned(), | |
| 1281 | + | tree_hash: String::new(), | |
| 1282 | + | message: String::new(), | |
| 1283 | + | author: g1t_contracts::repos::Signature { name: "a".into(), email: "a@example.com".into() }, | |
| 1284 | + | parents: names.get(at + 1).map(|parent| vec![(*parent).to_owned()]).unwrap_or_default(), | |
| 1285 | + | authored_at: String::new(), | |
| 1286 | + | }) | |
| 1287 | + | .collect() | |
| 1288 | + | } | |
| 1289 | + | ||
| 1290 | + | fn hashes(commits: &[Commit]) -> Vec<&str> { | |
| 1291 | + | commits.iter().map(|commit| commit.hash.as_str()).collect() | |
| 1292 | + | } | |
| 1293 | + | ||
| 1294 | + | #[test] | |
| 1295 | + | fn a_history_is_the_new_commits_then_the_one_kept_from_an_old_head() { | |
| 1296 | + | // The branch moved from c3 to c5; c3's history (limit 4) was kept. | |
| 1297 | + | let short = chain(&["c5", "c4", "c3", "c2"]); | |
| 1298 | + | let kept = chain(&["c3", "c2", "c1", "c0"]); | |
| 1299 | + | assert_eq!(hashes(&splice(&short, 2, kept.clone(), 4)), ["c5", "c4", "c3", "c2"]); | |
| 1300 | + | assert_eq!(hashes(&splice(&short, 2, kept.clone(), 6)), ["c5", "c4", "c3", "c2", "c1", "c0"]); | |
| 1301 | + | // A kept history that reached the first commit ends there. | |
| 1302 | + | assert_eq!(hashes(&splice(&short, 2, chain(&["c3", "c2"]), 10)), ["c5", "c4", "c3", "c2"]); | |
| 1303 | + | } | |
| 1304 | + | ||
| 1305 | + | #[test] | |
| 1306 | + | fn only_long_histories_by_hash_are_spliced() { | |
| 1307 | + | assert!(SPLICE_PROBE < SPLICE_FROM); | |
| 1308 | + | let hash = "a".repeat(40); | |
| 1309 | + | assert!(matches!(log_key(&hash, SPLICE_FROM, None), Some(CacheKey::Forever(_)))); | |
| 1310 | + | assert!(matches!(log_key("main", SPLICE_FROM, Some(1)), Some(CacheKey::Versioned(_)))); | |
| 1311 | + | } | |
| 1312 | + | ||
| 1210 | 1313 | fn access(token: &str) -> GitAccess { | |
| 1211 | 1314 | GitAccess { | |
| 1212 | 1315 | remote: "https://store.example/acme--rocket.git".to_owned(), | |