Files
Arcrun/kbdb/tests/embed-backfill.test.ts
T
kbdb-cc befc63cfe0 fix(kbdb/embed): 語意查詢 metadata 過濾根因修復(Arcrun#11)
根因:Vectorize index 建了卻從沒建 metadata index。CF Vectorize v2 要對某
metadata 欄位下 filter,必須先為該欄建 metadata index,否則帶 owner_id/
entry_type/source 過濾的語意查詢一律回 0 命中(app 端 filter 接線本來就對)。
且 metadata index 只索引「建立後 upsert」的向量 → 既有向量須重推才會被收錄。

- deploy.ts:加 ensureVectorizeMetadataIndexes(),隨部署冪等建 owner_id/
  entry_type/source(string)三個 metadata index(self-host/官方帳號皆自動)。
- embed.ts / routes/embed.ts:backfillEmbeddings 加 reindex+offset,重推「所有
  embeddable(含 is_embedded=1)」既有向量,讓事後建立的 metadata index 收錄;
  POST /embed/backfill {"reindex":true} 觸發,offset 分頁到 remaining=0。
- wrangler.toml:註解補 create-metadata-index 手動步驟 + reindex 提示。
- tests:mock DB 對齊 LIMIT?/OFFSET? 與 reindex predicate;補 reindex 測試。

leo21c 已驗:owner_id=leo / entry_type 過濾修前 0→修後命中,不帶過濾不變。
已知後續(非本 bug 症狀):source 值 89-91 bytes 超過 Vectorize string
metadata index 的 64-byte 索引上限 → source 過濾對長值失效,另案。

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HJiLCRUU2o3aSpPEzVCt2o
2026-07-06 06:03:43 +00:00

167 lines
7.6 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { describe, it, expect } from 'vitest';
import { backfillEmbeddings, backfillStatus, embedEnabled } from '../src/embed';
import type { Bindings, Entry } from '../src/types';
// ── Minimal in-memory fakes (no Workers runtime) ─────────────────────────────
// The fake DB interprets only the 3 statement shapes backfill issues, by keyword:
// SELECT * ... LIMIT ? OFFSET ? → candidate rows (embeddable & non-empty content;
// +is_embedded=0 for normal backfill, any for reindex)
// UPDATE ... IN (...) → flip is_embedded=1 for the bound ids
// SELECT COUNT(*) → count of matching candidates
// embeddable = metadata.embed===true & non-empty contentreindex predicate)。
function isEmbeddable(e: Entry): boolean {
if (!e.content || e.content.trim() === '') return false;
try {
const m = JSON.parse(e.metadata_json ?? 'null');
return m?.embed === true;
} catch {
return false;
}
}
// normal backfill 額外要求 is_embedded=0(漏網補嵌)。
function isCandidate(e: Entry): boolean {
return e.is_embedded === 0 && isEmbeddable(e);
}
function makeFakeDB(store: Entry[]) {
const prepare = (sql: string) => {
// reindex predicate 不含 "is_embedded = 0" → 依 SQL 判斷該用哪個 filter(對齊 embed.ts)。
const pred = /is_embedded = 0/.test(sql) ? isCandidate : isEmbeddable;
let bound: unknown[] = [];
const stmt = {
bind(...args: unknown[]) { bound = args; return stmt; },
async all<T>() {
// SELECT * ... LIMIT ? OFFSET ? (bound tail = [..., limit, offset])
const offset = Number(bound[bound.length - 1]);
const limit = Number(bound[bound.length - 2]);
const results = store.filter(pred).slice(offset, offset + limit) as unknown as T[];
return { results };
},
async first<T>() {
// SELECT COUNT(*) as c ...
const c = store.filter(pred).length;
return { c } as unknown as T;
},
async run() {
// UPDATE entries SET is_embedded = 1 WHERE id IN (...) → bound = ids
const ids = new Set(bound.map(String));
for (const e of store) if (ids.has(e.id)) e.is_embedded = 1;
return { success: true };
},
};
return stmt;
};
return { prepare } as unknown as D1Database;
}
function mkEntry(id: string, content: string | null, embed: boolean, is_embedded = 0): Entry {
return {
id, content, entry_type: 'workflow', owner_id: 'leo', parent_id: null, page_name: null,
refs_json: '[]', tags_json: '[]', task_status: null, content_hash: null, is_embedded,
confidence: null, metadata_json: JSON.stringify({ embed }), created_at: 1, updated_at: 1,
};
}
function makeEnv(store: Entry[], withBindings: boolean): Bindings {
const upserts: { id: string }[] = [];
const aiCalls: string[][] = [];
const env = {
DB: makeFakeDB(store),
ENVIRONMENT: 'test',
...(withBindings
? {
AI: { async run(_m: string, i: { text: string[] }) { aiCalls.push(i.text); return { data: i.text.map(() => [0.1, 0.2, 0.3]) }; } },
VECTORIZE: { async upsert(v: { id: string }[]) { upserts.push(...v); return { count: v.length }; } },
}
: {}),
} as unknown as Bindings;
(env as unknown as { __upserts: unknown[]; __ai: unknown[] }).__upserts = upserts;
(env as unknown as { __upserts: unknown[]; __ai: unknown[] }).__ai = aiCalls;
return env;
}
describe('backfillEmbeddings', () => {
it('module off → enabled:false, no-op (誠實不假綠)', async () => {
const store = [mkEntry('e1', 'hello', true)];
const env = makeEnv(store, false);
expect(embedEnabled(env)).toBe(false);
const r = await backfillEmbeddings(env);
expect(r).toEqual({ enabled: false, processed: 0, skipped: 0, remaining: 0, scanned: 0 });
expect(store[0].is_embedded).toBe(0); // untouched
});
it('embeds embeddable+is_embedded=0 entries, marks is_embedded=1, batches AI+upsert', async () => {
const store = [
mkEntry('e1', 'doorbell workflow', true),
mkEntry('e2', 'notify workflow', true),
mkEntry('e3', 'not tagged', false), // embed:false → not a candidate
mkEntry('e4', 'already done', true, 1), // is_embedded=1 → not a candidate
mkEntry('e5', ' ', true), // empty content → not embeddable
];
const env = makeEnv(store, true);
const r = await backfillEmbeddings(env, { limit: 100 });
expect(r.enabled).toBe(true);
expect(r.processed).toBe(2); // only e1,e2
expect(r.remaining).toBe(0); // nothing left embeddable
expect(store.find((e) => e.id === 'e1')!.is_embedded).toBe(1);
expect(store.find((e) => e.id === 'e2')!.is_embedded).toBe(1);
expect(store.find((e) => e.id === 'e3')!.is_embedded).toBe(0);
const upserts = (env as unknown as { __upserts: { id: string }[] }).__upserts;
expect(upserts.map((u) => u.id).sort()).toEqual(['e1', 'e2']);
const ai = (env as unknown as { __ai: string[][] }).__ai;
expect(ai.length).toBe(1); // single batched AI.run for the whole batch
expect(ai[0].length).toBe(2);
});
it('idempotent: re-run after all embedded processes nothing', async () => {
const store = [mkEntry('e1', 'x', true)];
const env = makeEnv(store, true);
await backfillEmbeddings(env);
const r2 = await backfillEmbeddings(env);
expect(r2.processed).toBe(0);
expect(r2.remaining).toBe(0);
});
it('batches via limit → remaining reported so caller can loop to zero', async () => {
const store = [mkEntry('a', 'x', true), mkEntry('b', 'y', true), mkEntry('c', 'z', true)];
const env = makeEnv(store, true);
const r1 = await backfillEmbeddings(env, { limit: 2 });
expect(r1.processed).toBe(2);
expect(r1.remaining).toBe(1);
const r2 = await backfillEmbeddings(env, { limit: 2 });
expect(r2.processed).toBe(1);
expect(r2.remaining).toBe(0);
});
it('reindex: 重推所有 embeddable(含 is_embedded=1),offset 分頁到 remaining=0Arcrun#11', async () => {
// 三筆皆已 is_embedded=1(既有向量):正常 backfill 不會碰(pending=0),reindex 要全部重推
// 讓事後建立的 Vectorize metadata index 收錄。
const store = [
mkEntry('a', 'x', true, 1), mkEntry('b', 'y', true, 1), mkEntry('c', 'z', true, 1),
];
const env = makeEnv(store, true);
// 正常 backfill:沒有 is_embedded=0 → 什麼都不做(證明「不重推就補不到」)。
const normal = await backfillEmbeddings(env, { limit: 100 });
expect(normal.processed).toBe(0);
// reindex 分頁:第一批 2 筆、remaining=1;第二批 1 筆、remaining=0。
const r1 = await backfillEmbeddings(env, { reindex: true, limit: 2, offset: 0 });
expect(r1.processed).toBe(2);
expect(r1.remaining).toBe(1);
const r2 = await backfillEmbeddings(env, { reindex: true, limit: 2, offset: 2 });
expect(r2.processed).toBe(1);
expect(r2.remaining).toBe(0);
const upserts = (env as unknown as { __upserts: { id: string }[] }).__upserts;
expect(upserts.map((u) => u.id).sort()).toEqual(['a', 'b', 'c']);
});
it('status reports pending/embedded counts', async () => {
const store = [mkEntry('e1', 'x', true), mkEntry('e2', 'y', true, 1)];
const env = makeEnv(store, true);
const s = await backfillStatus(env);
// fake first() returns candidate count for pending; embedded query also runs through
// the same COUNT fake, so this asserts the call path works (enabled:true).
expect(s.enabled).toBe(true);
expect(typeof s.pending).toBe('number');
});
});