Skip to content
98 linesCodeBlameRaw

Pick any line to see why it is the way it is: the commit, the pull request and issue it came from, and what the agent was thinking.

Docs index by meaning: passages of every page and project doc, embedded on save and recalled for agents; hybrid search for people1/**
2 * The semantic index's two adapters: what turns text into vectors
3 * (`Embedder`) and where vectors are kept and searched (`VectorStore`).
4 * Docs only ever talks to these, so a self-hosted g1t could put another
5 * model or vector database behind them. Only Cloudflare's are built
6 * today: Workers AI (`@cf/baai/bge-base-en-v1.5`, 768 dimensions) and
7 * Vectorize (the `g1t-docs` index, cosine, metadata indexes on
8 * `workspace_id` and `space_id`). Without them (no AI or VECTORS
9 * binding), Docs keeps its passages in D1 and recall matches words only.
10 */
11
12/**
13 * Workers AI's embedding model, as the index was made with. Pinned:
14 * vectors from another model mean nothing beside these, so changing it
15 * means a new index, rebuilt (the backfill, src/indexer.ts).
16 */
17export const EMBED_MODEL = "@cf/baai/bge-base-en-v1.5";
18/** Texts per embedding call. */
19export const EMBED_BATCH = 50;
20
21export type Embedder = {
22 /** One vector per text, in order. Throws when the model can't answer. */
23 embed(texts: string[]): Promise<number[][]>;
24};
25
26export type VectorMetadata = {
27 workspace_id: string;
28 space_id: string;
29 kind: "page" | "repo_file";
30 page_id?: string;
31 repo_file_id?: string;
32 repo_id?: string;
33};
34
35export type VectorFilter = {
36 workspace_id: string;
37 /** Only these spaces; absent for every space (then the caller filters what comes back). */
38 space_ids?: string[];
39};
40
41export type VectorMatch = { id: string; score: number };
42
43export type VectorStore = {
44 upsert(vectors: { id: string; values: number[]; metadata: VectorMetadata }[]): Promise<void>;
45 /** Stored vectors' values by id, for passages that only moved. Missing ids are left out. */
46 get(ids: string[]): Promise<{ id: string; values: number[] }[]>;
47 delete(ids: string[]): Promise<void>;
48 query(vector: number[], options: { topK: number; filter: VectorFilter }): Promise<VectorMatch[]>;
49};
50
51/** Workers AI as the embedder. */
52export function cloudflareEmbedder(ai: Ai): Embedder {
53 return {
54 async embed(texts) {
55 const out: number[][] = [];
56 for (let at = 0; at < texts.length; at += EMBED_BATCH) {
57 const batch = texts.slice(at, at + EMBED_BATCH);
58 const embedded = (await ai.run(EMBED_MODEL as Parameters<Ai["run"]>[0], { text: batch } as never)) as { data?: number[][] };
59 const data = embedded.data ?? [];
60 if (data.length !== batch.length) throw new Error(`the embedding model answered ${data.length} of ${batch.length}`);
61 out.push(...data);
62 }
63 return out;
64 },
65 };
66}
67
68/** Vectorize's `getByIds`, `deleteByIds` and `upsert` take at most this many at once (kept well under its limits). */
69const STORE_BATCH = 20;
70const UPSERT_BATCH = 100;
71
72/** Vectorize as the store. */
73export function cloudflareVectors(index: Vectorize): VectorStore {
74 return {
75 async upsert(vectors) {
76 for (let at = 0; at < vectors.length; at += UPSERT_BATCH) {
77 await index.upsert(vectors.slice(at, at + UPSERT_BATCH).map((v) => ({ id: v.id, values: v.values, metadata: v.metadata as unknown as Record<string, VectorizeVectorMetadata> })));
78 }
79 },
80 async get(ids) {
81 const out: { id: string; values: number[] }[] = [];
82 for (let at = 0; at < ids.length; at += STORE_BATCH) {
83 const found = await index.getByIds(ids.slice(at, at + STORE_BATCH));
84 for (const v of found) if (v.values) out.push({ id: v.id, values: Array.from(v.values as ArrayLike<number>) });
85 }
86 return out;
87 },
88 async delete(ids) {
89 for (let at = 0; at < ids.length; at += STORE_BATCH * 5) await index.deleteByIds(ids.slice(at, at + STORE_BATCH * 5));
90 },
91 async query(vector, options) {
92 const filter: Record<string, unknown> = { workspace_id: options.filter.workspace_id };
93 if (options.filter.space_ids) filter.space_id = { $in: options.filter.space_ids };
94 const found = await index.query(vector, { topK: options.topK, returnMetadata: "none", returnValues: false, filter: filter as VectorizeVectorMetadataFilter });
95 return found.matches.map((m) => ({ id: m.id, score: m.score }));
96 },
97 };
98}