Skip to content
109 linesCodeBlameRaw
1/**
2 * The semantic index's two adapters: what turns text into vectors
3 * (`Embedder`) and where vectors are kept and searched (`VectorStore`).
4 * Docs only ever talks to these, so a self-hosted g1t could put another
5 * model or vector database behind them. Only Cloudflare's are built
6 * today: Workers AI (`@cf/baai/bge-base-en-v1.5`, 768 dimensions) and
7 * Vectorize (the `g1t-docs` index, cosine, metadata indexes on
8 * `workspace_id` and `space_id`). Without them (no AI or VECTORS
9 * binding), Docs keeps its passages in D1 and recall matches words only.
10 *
11 * Folios (Artifacts mode) have an index of their own, `g1t-folios`
12 * (binding FOLIO_VECTORS), filtered by `workspace_id`, `scope` and `kind`:
13 * the same two adapters, another index.
14 */
15
16/**
17 * Workers AI's embedding model, as the index was made with. Pinned:
18 * vectors from another model mean nothing beside these, so changing it
19 * means a new index, rebuilt (the backfill, src/indexer.ts).
20 */
21export const EMBED_MODEL = "@cf/baai/bge-base-en-v1.5";
22/** Texts per embedding call. */
23export const EMBED_BATCH = 50;
24
25export type Embedder = {
26 /** One vector per text, in order. Throws when the model can't answer. */
27 embed(texts: string[]): Promise<number[][]>;
28};
29
30export type VectorMetadata = {
31 workspace_id: string;
32 /** Docs' pages and projects' docs (index `g1t-docs`). */
33 space_id?: string;
34 /** Folios (index `g1t-folios`): `space:<id>` or `folio:<access root>` (src/access.ts `folioScope`). */
35 scope?: string;
36 kind: "page" | "repo_file" | "doc" | "slides" | "design" | "dashboard";
37 page_id?: string;
38 repo_file_id?: string;
39 repo_id?: string;
40 folio_id?: string;
41};
42
43export type VectorFilter = {
44 workspace_id: string;
45 /** Only these spaces; absent for every space (then the caller filters what comes back). */
46 space_ids?: string[];
47 /** Folios: only these scopes; absent for every scope (then the caller filters what comes back). */
48 scopes?: string[];
49};
50
51export type VectorMatch = { id: string; score: number };
52
53export type VectorStore = {
54 upsert(vectors: { id: string; values: number[]; metadata: VectorMetadata }[]): Promise<void>;
55 /** Stored vectors' values by id, for passages that only moved. Missing ids are left out. */
56 get(ids: string[]): Promise<{ id: string; values: number[] }[]>;
57 delete(ids: string[]): Promise<void>;
58 query(vector: number[], options: { topK: number; filter: VectorFilter }): Promise<VectorMatch[]>;
59};
60
61/** Workers AI as the embedder. */
62export function cloudflareEmbedder(ai: Ai): Embedder {
63 return {
64 async embed(texts) {
65 const out: number[][] = [];
66 for (let at = 0; at < texts.length; at += EMBED_BATCH) {
67 const batch = texts.slice(at, at + EMBED_BATCH);
68 const embedded = (await ai.run(EMBED_MODEL as Parameters<Ai["run"]>[0], { text: batch } as never)) as { data?: number[][] };
69 const data = embedded.data ?? [];
70 if (data.length !== batch.length) throw new Error(`the embedding model answered ${data.length} of ${batch.length}`);
71 out.push(...data);
72 }
73 return out;
74 },
75 };
76}
77
78/** Vectorize's `getByIds`, `deleteByIds` and `upsert` take at most this many at once (kept well under its limits). */
79const STORE_BATCH = 20;
80const UPSERT_BATCH = 100;
81
82/** Vectorize as the store. */
83export function cloudflareVectors(index: Vectorize): VectorStore {
84 return {
85 async upsert(vectors) {
86 for (let at = 0; at < vectors.length; at += UPSERT_BATCH) {
87 await index.upsert(vectors.slice(at, at + UPSERT_BATCH).map((v) => ({ id: v.id, values: v.values, metadata: v.metadata as unknown as Record<string, VectorizeVectorMetadata> })));
88 }
89 },
90 async get(ids) {
91 const out: { id: string; values: number[] }[] = [];
92 for (let at = 0; at < ids.length; at += STORE_BATCH) {
93 const found = await index.getByIds(ids.slice(at, at + STORE_BATCH));
94 for (const v of found) if (v.values) out.push({ id: v.id, values: Array.from(v.values as ArrayLike<number>) });
95 }
96 return out;
97 },
98 async delete(ids) {
99 for (let at = 0; at < ids.length; at += STORE_BATCH * 5) await index.deleteByIds(ids.slice(at, at + STORE_BATCH * 5));
100 },
101 async query(vector, options) {
102 const filter: Record<string, unknown> = { workspace_id: options.filter.workspace_id };
103 if (options.filter.space_ids) filter.space_id = { $in: options.filter.space_ids };
104 if (options.filter.scopes) filter.scope = { $in: options.filter.scopes };
105 const found = await index.query(vector, { topK: options.topK, returnMetadata: "none", returnValues: false, filter: filter as VectorizeVectorMetadataFilter });
106 return found.matches.map((m) => ({ id: m.id, score: m.score }));
107 },
108 };
109}