diff --git a/backend/src/lib/__tests__/citations.test.ts b/backend/src/lib/__tests__/citations.test.ts new file mode 100644 index 0000000..08b0c9c --- /dev/null +++ b/backend/src/lib/__tests__/citations.test.ts @@ -0,0 +1,397 @@ +import { describe, it, expect } from "vitest"; +import { + parseCitations, + parseCitationsWithDiagnostics, + parsePartialCitationObjects, + createCitation, + CITATIONS_OPEN_TAG, + CITATIONS_CLOSE_TAG, +} from "../chat/citations"; +import type { DocIndex } from "../chat/types"; + +function citationsBlock(json: string) { + return `Answer text.\n${CITATIONS_OPEN_TAG}\n${json}\n${CITATIONS_CLOSE_TAG}`; +} + +// --------------------------------------------------------------------------- +// parseCitationsWithDiagnostics +// --------------------------------------------------------------------------- + +describe("parseCitationsWithDiagnostics", () => { + it("reports no block when the tags are absent", () => { + const { citations, diagnostics } = + parseCitationsWithDiagnostics("plain answer"); + expect(citations).toEqual([]); + expect(diagnostics).toEqual({ hasBlock: false, rawLength: 0, error: null }); + }); + + it("reports a JSON parse error for malformed block content", () => { + const { citations, diagnostics } = parseCitationsWithDiagnostics( + citationsBlock("[{not json"), + ); + expect(citations).toEqual([]); + expect(diagnostics.hasBlock).toBe(true); + expect(diagnostics.rawLength).toBeGreaterThan(0); + expect(diagnostics.error).toBeTruthy(); + }); + + it("reports an error when the block JSON is not an array", () => { + const { citations, diagnostics } = parseCitationsWithDiagnostics( + citationsBlock('{"ref": 1}'), + ); + expect(citations).toEqual([]); + expect(diagnostics.error).toBe("CITATIONS block JSON was not an array."); + }); +}); + +// --------------------------------------------------------------------------- +// parseCitations — document citations +// --------------------------------------------------------------------------- + +describe("parseCitations (document citations)", () => { + it("parses a minimal document citation", () => { + const [citation] = parseCitations( + citationsBlock( + '[{"ref": 1, "doc_id": "doc-1", "page": 3, "quote": "the term"}]', + ), + ); + expect(citation).toMatchObject({ + kind: "document", + ref: 1, + doc_id: "doc-1", + page: 3, + quote: "the term", + }); + expect(citation.quotes).toHaveLength(1); + }); + + it("derives ref from a [N] marker when ref is missing", () => { + const [citation] = parseCitations( + citationsBlock( + '[{"marker": "[7]", "doc_id": "doc-1", "page": 1, "quote": "q"}]', + ), + ); + expect(citation.ref).toBe(7); + }); + + it("drops entries without a usable ref or marker", () => { + expect( + parseCitations( + citationsBlock('[{"doc_id": "doc-1", "quote": "q", "marker": "nope"}]'), + ), + ).toEqual([]); + }); + + it("drops document entries without doc_id or quote", () => { + expect( + parseCitations(citationsBlock('[{"ref": 1, "doc_id": "doc-1"}]')), + ).toEqual([]); + expect( + parseCitations(citationsBlock('[{"ref": 1, "quote": "q"}]')), + ).toEqual([]); + }); + + it("drops non-object entries but keeps valid ones", () => { + const citations = parseCitations( + citationsBlock( + '[null, "junk", {"ref": 2, "doc_id": "doc-2", "page": 4, "quote": "kept"}]', + ), + ); + expect(citations).toHaveLength(1); + expect(citations[0]).toMatchObject({ ref: 2, doc_id: "doc-2" }); + }); + + it("accepts a text field as a quote alias", () => { + const [citation] = parseCitations( + citationsBlock('[{"ref": 1, "doc_id": "doc-1", "page": 2, "text": "aliased"}]'), + ); + expect(citation.kind).toBe("document"); + expect((citation as { quote: string }).quote).toBe("aliased"); + }); + + it("normalizes pages: numbers kept, ranges kept, junk becomes 1", () => { + const citations = parseCitations( + citationsBlock( + '[{"ref": 1, "doc_id": "d", "page": 5, "quote": "a"},' + + '{"ref": 2, "doc_id": "d", "page": "3-5", "quote": "b"},' + + '{"ref": 3, "doc_id": "d", "page": "12", "quote": "c"},' + + '{"ref": 4, "doc_id": "d", "page": "unknown", "quote": "d"}]', + ), + ); + expect(citations.map((c) => (c as { page: number | string }).page)).toEqual([ + 5, + "3-5", + 12, + 1, + ]); + }); + + it("keeps at most 3 quotes and inherits top-level page/sheet/cell", () => { + const [citation] = parseCitations( + citationsBlock( + JSON.stringify([ + { + ref: 1, + doc_id: "doc-1", + page: 9, + sheet: "Summary", + cell: "B7", + quotes: [ + { quote: "one" }, + { quote: "two", page: 2 }, + { quote: "three", sheet: "Detail", cell: "C1" }, + { quote: "four" }, + ], + }, + ]), + ), + ); + expect(citation.kind).toBe("document"); + const doc = citation as { + page: number | string; + quotes: { page: number | string; quote: string; sheet?: string; cell?: string }[]; + }; + expect(doc.quotes).toHaveLength(3); + expect(doc.quotes[0]).toEqual({ + page: 9, + quote: "one", + sheet: "Summary", + cell: "B7", + }); + expect(doc.quotes[1].page).toBe(2); + expect(doc.quotes[2]).toMatchObject({ sheet: "Detail", cell: "C1" }); + // Top-level fields mirror the first quote. + expect(doc.page).toBe(9); + }); + + it("skips quote rows without text", () => { + const [citation] = parseCitations( + citationsBlock( + '[{"ref": 1, "doc_id": "doc-1", "page": 1,' + + ' "quotes": [{"page": 2}, {"quote": " "}, {"quote": "real"}]}]', + ), + ); + expect((citation as { quotes: unknown[] }).quotes).toHaveLength(1); + }); +}); + +// --------------------------------------------------------------------------- +// parseCitations — case citations +// --------------------------------------------------------------------------- + +describe("parseCitations (case citations)", () => { + it("parses a case citation from a numeric cluster_id", () => { + const [citation] = parseCitations( + citationsBlock('[{"ref": 1, "cluster_id": 12345, "quote": "held that"}]'), + ); + expect(citation).toMatchObject({ kind: "case", ref: 1, cluster_id: 12345 }); + expect((citation as { quotes: unknown[] }).quotes).toEqual([ + { opinionId: null, type: null, author: null, quote: "held that" }, + ]); + }); + + it("accepts clusterId camelCase and string cluster ids", () => { + const citations = parseCitations( + citationsBlock( + '[{"ref": 1, "clusterId": 7, "quote": "a"},' + + '{"ref": 2, "cluster_id": "42", "quote": "b"}]', + ), + ); + expect(citations.map((c) => (c as { cluster_id: number }).cluster_id)).toEqual([ + 7, 42, + ]); + }); + + it("floors fractional cluster ids", () => { + const [citation] = parseCitations( + citationsBlock('[{"ref": 1, "cluster_id": 12.9, "quote": "q"}]'), + ); + expect((citation as { cluster_id: number }).cluster_id).toBe(12); + }); + + it("treats non-positive cluster ids as document citations", () => { + // cluster_id 0 fails the > 0 check, so the entry needs a doc_id. + expect( + parseCitations(citationsBlock('[{"ref": 1, "cluster_id": 0, "quote": "q"}]')), + ).toEqual([]); + }); + + it("normalizes structured case quotes with opinion metadata", () => { + const [citation] = parseCitations( + citationsBlock( + JSON.stringify([ + { + ref: 3, + cluster_id: 99, + quotes: [ + { + quote: "majority text", + opinion_id: 11.7, + type: "majority", + author: "Judge A", + }, + { text: "concurrence text", opinionId: 12 }, + { type: "no quote text, dropped" }, + ], + }, + ]), + ), + ); + expect((citation as { quotes: unknown[] }).quotes).toEqual([ + { + opinionId: 11, + type: "majority", + author: "Judge A", + quote: "majority text", + }, + { opinionId: 12, type: null, author: null, quote: "concurrence text" }, + ]); + }); + + it("drops case citations with no quotes at all", () => { + expect( + parseCitations(citationsBlock('[{"ref": 1, "cluster_id": 5}]')), + ).toEqual([]); + }); +}); + +// --------------------------------------------------------------------------- +// parsePartialCitationObjects +// --------------------------------------------------------------------------- + +describe("parsePartialCitationObjects", () => { + it("returns [] when no array has started", () => { + expect(parsePartialCitationObjects("streaming ")).toEqual([]); + }); + + it("extracts complete objects and ignores a trailing incomplete one", () => { + const partial = + '[{"ref": 1, "doc_id": "doc-1", "page": 2, "quote": "done"},' + + ' {"ref": 2, "doc_id": "doc-2", "page": 3, "quote": "still stream'; + const citations = parsePartialCitationObjects(partial); + expect(citations).toHaveLength(1); + expect(citations[0]).toMatchObject({ ref: 1, doc_id: "doc-1" }); + }); + + it("handles braces and escaped quotes inside string values", () => { + const partial = + '[{"ref": 1, "doc_id": "doc-1", "page": 1,' + + ' "quote": "clause {a} says \\"stop\\""}'; + const citations = parsePartialCitationObjects(partial); + expect(citations).toHaveLength(1); + expect((citations[0] as { quote: string }).quote).toBe( + 'clause {a} says "stop"', + ); + }); + + it("ignores content after the closing tag", () => { + const text = + '[{"ref": 1, "doc_id": "a", "page": 1, "quote": "q"}]' + + '[{"ref": 9, "doc_id": "b", "page": 1, "quote": "after"}]'; + const citations = parsePartialCitationObjects(text); + expect(citations).toHaveLength(1); + expect(citations[0]).toMatchObject({ ref: 1 }); + }); + + it("stops at the array close bracket", () => { + const text = + '[{"ref": 1, "doc_id": "a", "page": 1, "quote": "q"}] {"ref": 2, "doc_id": "b", "page": 1, "quote": "outside"}'; + const citations = parsePartialCitationObjects(text); + expect(citations).toHaveLength(1); + }); + + it("skips malformed objects but keeps later valid ones", () => { + const text = + '[{"ref": bad}, {"ref": 2, "doc_id": "doc-2", "page": 1, "quote": "ok"}'; + const citations = parsePartialCitationObjects(text); + expect(citations).toHaveLength(1); + expect(citations[0]).toMatchObject({ ref: 2 }); + }); +}); + +// --------------------------------------------------------------------------- +// createCitation +// --------------------------------------------------------------------------- + +describe("createCitation", () => { + const docIndex: DocIndex = { + "doc-1": { + document_id: "uuid-aaa", + filename: "contract.pdf", + version_id: "ver-1", + version_number: 2, + }, + }; + + it("builds a document citation payload from the doc index", () => { + const [parsed] = parseCitations( + citationsBlock('[{"ref": 1, "doc_id": "doc-1", "page": 4, "quote": "q"}]'), + ); + expect(createCitation(parsed, docIndex)).toMatchObject({ + type: "citation_data", + kind: "document", + ref: 1, + doc_id: "doc-1", + document_id: "uuid-aaa", + version_id: "ver-1", + version_number: 2, + filename: "contract.pdf", + page: 4, + quote: "q", + }); + }); + + it("falls back to the raw doc_id as filename when unresolvable", () => { + const [parsed] = parseCitations( + citationsBlock('[{"ref": 1, "doc_id": "doc-9", "page": 1, "quote": "q"}]'), + ); + const citation = createCitation(parsed, docIndex); + expect(citation).toMatchObject({ + filename: "doc-9", + document_id: undefined, + version_id: null, + }); + }); + + it("enriches a case citation from the cluster map", () => { + const [parsed] = parseCitations( + citationsBlock('[{"ref": 2, "cluster_id": 55, "quote": "held"}]'), + ); + const cases = new Map([ + [ + 55, + { + caseName: "Smith v. Jones", + citations: ["123 U.S. 456", "alt cite"], + url: "https://example.test/case", + pdfUrl: null, + dateFiled: "1990-01-02", + }, + ], + ]); + expect(createCitation(parsed, docIndex, cases)).toMatchObject({ + type: "citation_data", + kind: "case", + ref: 2, + cluster_id: 55, + case_name: "Smith v. Jones", + citation: "123 U.S. 456", + url: "https://example.test/case", + pdfUrl: null, + dateFiled: "1990-01-02", + }); + }); + + it("nulls case metadata when the cluster map has no entry", () => { + const [parsed] = parseCitations( + citationsBlock('[{"ref": 2, "cluster_id": 55, "quote": "held"}]'), + ); + expect(createCitation(parsed, docIndex)).toMatchObject({ + case_name: null, + citation: null, + url: null, + pdfUrl: null, + dateFiled: null, + }); + }); +}); diff --git a/backend/src/lib/__tests__/documentVersions.test.ts b/backend/src/lib/__tests__/documentVersions.test.ts new file mode 100644 index 0000000..4442fa3 --- /dev/null +++ b/backend/src/lib/__tests__/documentVersions.test.ts @@ -0,0 +1,315 @@ +import { describe, it, expect } from "vitest"; +import { + loadActiveVersion, + attachActiveVersionPaths, + attachLatestVersionNumbers, +} from "../documentVersions"; + +type Row = Record; + +/** + * Read-only Supabase mock covering the query chains documentVersions uses: + * select/eq/in/is/not filters plus single() and awaiting the builder. + */ +function makeDb(tables: Record) { + return { + from(table: string) { + let rows = [...(tables[table] ?? [])]; + const query: any = { + select: () => query, + eq: (column: string, value: unknown) => { + rows = rows.filter((row) => row[column] === value); + return query; + }, + in: (column: string, values: unknown[]) => { + rows = rows.filter((row) => values.includes(row[column])); + return query; + }, + is: (column: string, value: unknown) => { + rows = rows.filter((row) => (row[column] ?? null) === value); + return query; + }, + not: (column: string, operator: string, value: unknown) => { + if (operator === "is" && value === null) { + rows = rows.filter((row) => row[column] != null); + } + return query; + }, + single: async () => ({ data: rows[0] ?? null, error: null }), + then: ( + resolve: (value: { data: Row[]; error: null }) => unknown, + reject?: (reason: unknown) => unknown, + ) => Promise.resolve({ data: rows, error: null }).then(resolve, reject), + }; + return query; + }, + } as any; +} + +/** Shape of the mutable doc rows the attach* helpers annotate in place. */ +type TestDoc = { + id: string; + current_version_id?: string | null; + latest_version_number?: number | null; + [k: string]: unknown; +}; + +const FULL_VERSION = { + id: "ver-1", + document_id: "doc-1", + storage_path: "documents/u/doc-1/source.pdf", + pdf_storage_path: "documents/u/doc-1/converted.pdf", + version_number: 3, + filename: "contract.pdf", + source: "upload", + file_type: "application/pdf", + size_bytes: 1024, + page_count: 12, + deleted_at: null, +}; + +// --------------------------------------------------------------------------- +// loadActiveVersion +// --------------------------------------------------------------------------- + +describe("loadActiveVersion", () => { + it("resolves the document's current version by default", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: "ver-1" }], + document_versions: [FULL_VERSION], + }); + await expect(loadActiveVersion("doc-1", db)).resolves.toEqual({ + id: "ver-1", + storage_path: "documents/u/doc-1/source.pdf", + pdf_storage_path: "documents/u/doc-1/converted.pdf", + version_number: 3, + filename: "contract.pdf", + source: "upload", + file_type: "application/pdf", + size_bytes: 1024, + page_count: 12, + }); + }); + + it("prefers an explicit versionId over current_version_id", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: "ver-1" }], + document_versions: [ + FULL_VERSION, + { ...FULL_VERSION, id: "ver-2", version_number: 2 }, + ], + }); + const version = await loadActiveVersion("doc-1", db, "ver-2"); + expect(version?.id).toBe("ver-2"); + expect(version?.version_number).toBe(2); + }); + + it("returns null when neither versionId nor current_version_id exist", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: null }], + document_versions: [FULL_VERSION], + }); + await expect(loadActiveVersion("doc-1", db)).resolves.toBeNull(); + }); + + it("returns null when the version belongs to a different document", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: "ver-other" }], + document_versions: [ + { ...FULL_VERSION, id: "ver-other", document_id: "doc-2" }, + ], + }); + await expect(loadActiveVersion("doc-1", db)).resolves.toBeNull(); + // Also guards against a spoofed explicit versionId. + await expect( + loadActiveVersion("doc-1", db, "ver-other"), + ).resolves.toBeNull(); + }); + + it("returns null for soft-deleted versions", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: "ver-1" }], + document_versions: [ + { ...FULL_VERSION, deleted_at: "2026-01-01T00:00:00Z" }, + ], + }); + await expect(loadActiveVersion("doc-1", db)).resolves.toBeNull(); + }); + + it("returns null when the version has no storage_path", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: "ver-1" }], + document_versions: [{ ...FULL_VERSION, storage_path: null }], + }); + await expect(loadActiveVersion("doc-1", db)).resolves.toBeNull(); + }); + + it("defaults optional metadata fields to null", async () => { + const db = makeDb({ + documents: [{ id: "doc-1", current_version_id: "ver-1" }], + document_versions: [ + { + id: "ver-1", + document_id: "doc-1", + storage_path: "documents/u/doc-1/source.docx", + deleted_at: null, + }, + ], + }); + await expect(loadActiveVersion("doc-1", db)).resolves.toEqual({ + id: "ver-1", + storage_path: "documents/u/doc-1/source.docx", + pdf_storage_path: null, + version_number: null, + filename: null, + source: null, + file_type: null, + size_bytes: null, + page_count: null, + }); + }); +}); + +// --------------------------------------------------------------------------- +// attachActiveVersionPaths +// --------------------------------------------------------------------------- + +describe("attachActiveVersionPaths", () => { + it("returns the same empty array untouched", async () => { + const db = makeDb({ document_versions: [] }); + const docs: TestDoc[] = []; + await expect(attachActiveVersionPaths(db, docs)).resolves.toBe(docs); + }); + + it("nulls all fields when no document has a current version", async () => { + const db = makeDb({ document_versions: [FULL_VERSION] }); + const [doc] = await attachActiveVersionPaths(db, [ + { id: "doc-1", current_version_id: null }, + ]); + expect(doc).toMatchObject({ + filename: "Untitled document", + storage_path: null, + pdf_storage_path: null, + file_type: null, + size_bytes: null, + page_count: null, + }); + }); + + it("merges active-version metadata onto each row", async () => { + const db = makeDb({ + document_versions: [ + FULL_VERSION, + { + ...FULL_VERSION, + id: "ver-2", + storage_path: "documents/u/doc-2/source.docx", + pdf_storage_path: null, + filename: "nda.docx", + version_number: 1, + }, + ], + }); + const docs = await attachActiveVersionPaths(db, [ + { id: "doc-1", current_version_id: "ver-1" }, + { id: "doc-2", current_version_id: "ver-2" }, + { id: "doc-3", current_version_id: null }, + ]); + expect(docs[0]).toMatchObject({ + storage_path: "documents/u/doc-1/source.pdf", + pdf_storage_path: "documents/u/doc-1/converted.pdf", + active_version_number: 3, + filename: "contract.pdf", + file_type: "application/pdf", + size_bytes: 1024, + page_count: 12, + }); + expect(docs[1]).toMatchObject({ + storage_path: "documents/u/doc-2/source.docx", + pdf_storage_path: null, + active_version_number: 1, + filename: "nda.docx", + }); + // Mixed list: the version-less doc still gets explicit nulls. + expect(docs[2]).toMatchObject({ + storage_path: null, + filename: "Untitled document", + }); + }); + + it("falls back to 'Untitled document' for blank filenames", async () => { + const db = makeDb({ + document_versions: [{ ...FULL_VERSION, filename: " " }], + }); + const [doc] = await attachActiveVersionPaths(db, [ + { id: "doc-1", current_version_id: "ver-1" }, + ]); + expect(doc.filename).toBe("Untitled document"); + }); + + it("ignores soft-deleted versions", async () => { + const db = makeDb({ + document_versions: [ + { ...FULL_VERSION, deleted_at: "2026-01-01T00:00:00Z" }, + ], + }); + const [doc] = await attachActiveVersionPaths(db, [ + { id: "doc-1", current_version_id: "ver-1" }, + ]); + expect(doc.storage_path).toBeNull(); + expect(doc.filename).toBe("Untitled document"); + }); +}); + +// --------------------------------------------------------------------------- +// attachLatestVersionNumbers +// --------------------------------------------------------------------------- + +describe("attachLatestVersionNumbers", () => { + const versionRow = ( + document_id: string, + version_number: number | null, + overrides: Row = {}, + ) => ({ + document_id, + version_number, + source: "assistant_edit", + deleted_at: null, + ...overrides, + }); + + it("returns the same empty array untouched", async () => { + const db = makeDb({ document_versions: [] }); + const docs: TestDoc[] = []; + await expect(attachLatestVersionNumbers(db, docs)).resolves.toBe(docs); + }); + + it("attaches the max assistant_edit version number per document", async () => { + const db = makeDb({ + document_versions: [ + versionRow("doc-1", 1), + versionRow("doc-1", 4), + versionRow("doc-1", 2), + versionRow("doc-2", 7), + ], + }); + const docs = await attachLatestVersionNumbers(db, [ + { id: "doc-1" }, + { id: "doc-2" }, + { id: "doc-3" }, + ]); + expect(docs.map((d) => d.latest_version_number)).toEqual([4, 7, null]); + }); + + it("ignores non-assistant_edit and soft-deleted versions", async () => { + const db = makeDb({ + document_versions: [ + versionRow("doc-1", 9, { source: "upload" }), + versionRow("doc-1", 8, { deleted_at: "2026-01-01T00:00:00Z" }), + versionRow("doc-1", 2), + ], + }); + const docs = await attachLatestVersionNumbers(db, [{ id: "doc-1" }]); + expect(docs[0].latest_version_number).toBe(2); + }); +}); diff --git a/backend/src/lib/__tests__/llmModels.test.ts b/backend/src/lib/__tests__/llmModels.test.ts new file mode 100644 index 0000000..be3d7d5 --- /dev/null +++ b/backend/src/lib/__tests__/llmModels.test.ts @@ -0,0 +1,119 @@ +import { describe, it, expect } from "vitest"; +import { + CLAUDE_MAIN_MODELS, + GEMINI_MAIN_MODELS, + OPENAI_MAIN_MODELS, + CLAUDE_MID_MODELS, + GEMINI_MID_MODELS, + OPENAI_MID_MODELS, + CLAUDE_LOW_MODELS, + GEMINI_LOW_MODELS, + OPENAI_LOW_MODELS, + DEFAULT_MAIN_MODEL, + DEFAULT_TITLE_MODEL, + DEFAULT_TABULAR_MODEL, + providerForModel, + resolveModel, +} from "../llm/models"; + +// --------------------------------------------------------------------------- +// providerForModel +// --------------------------------------------------------------------------- + +describe("providerForModel", () => { + it("maps claude-* ids to the claude provider", () => { + for (const model of [...CLAUDE_MAIN_MODELS, ...CLAUDE_MID_MODELS, ...CLAUDE_LOW_MODELS]) { + expect(providerForModel(model)).toBe("claude"); + } + }); + + it("maps gemini-* ids to the gemini provider", () => { + for (const model of [...GEMINI_MAIN_MODELS, ...GEMINI_MID_MODELS, ...GEMINI_LOW_MODELS]) { + expect(providerForModel(model)).toBe("gemini"); + } + }); + + it("maps gpt-* ids to the openai provider", () => { + for (const model of [...OPENAI_MAIN_MODELS, ...OPENAI_MID_MODELS, ...OPENAI_LOW_MODELS]) { + expect(providerForModel(model)).toBe("openai"); + } + }); + + it("throws on an unknown model id", () => { + expect(() => providerForModel("llama-3")).toThrow(/Unknown model id/); + expect(() => providerForModel("")).toThrow(/Unknown model id/); + }); + + it("infers by prefix only, without validating against the catalog", () => { + // Documents current behavior: any claude-/gemini-/gpt- prefix is + // accepted even if the id is not a canonical model. + expect(providerForModel("claude-nonexistent")).toBe("claude"); + expect(providerForModel("gpt-nonexistent")).toBe("openai"); + }); +}); + +// --------------------------------------------------------------------------- +// resolveModel +// --------------------------------------------------------------------------- + +describe("resolveModel", () => { + it("returns a known model id unchanged", () => { + expect(resolveModel("claude-sonnet-4-6", DEFAULT_MAIN_MODEL)).toBe( + "claude-sonnet-4-6", + ); + expect(resolveModel("gpt-5.4-lite", DEFAULT_TITLE_MODEL)).toBe( + "gpt-5.4-lite", + ); + }); + + it("falls back for unknown model ids", () => { + expect(resolveModel("gpt-3.5-turbo", DEFAULT_MAIN_MODEL)).toBe( + DEFAULT_MAIN_MODEL, + ); + }); + + it("falls back for null, undefined, and empty ids", () => { + expect(resolveModel(null, DEFAULT_MAIN_MODEL)).toBe(DEFAULT_MAIN_MODEL); + expect(resolveModel(undefined, DEFAULT_TABULAR_MODEL)).toBe( + DEFAULT_TABULAR_MODEL, + ); + expect(resolveModel("", DEFAULT_TITLE_MODEL)).toBe(DEFAULT_TITLE_MODEL); + }); + + it("accepts models from every tier of the catalog", () => { + const catalog = [ + ...CLAUDE_MAIN_MODELS, + ...GEMINI_MAIN_MODELS, + ...OPENAI_MAIN_MODELS, + ...CLAUDE_MID_MODELS, + ...GEMINI_MID_MODELS, + ...OPENAI_MID_MODELS, + ...CLAUDE_LOW_MODELS, + ...GEMINI_LOW_MODELS, + ...OPENAI_LOW_MODELS, + ]; + for (const model of catalog) { + expect(resolveModel(model, "fallback-model")).toBe(model); + } + }); +}); + +// --------------------------------------------------------------------------- +// Default model sanity +// --------------------------------------------------------------------------- + +describe("default models", () => { + it("every default resolves to itself (defaults are in the catalog)", () => { + expect(resolveModel(DEFAULT_MAIN_MODEL, "x")).toBe(DEFAULT_MAIN_MODEL); + expect(resolveModel(DEFAULT_TITLE_MODEL, "x")).toBe(DEFAULT_TITLE_MODEL); + expect(resolveModel(DEFAULT_TABULAR_MODEL, "x")).toBe( + DEFAULT_TABULAR_MODEL, + ); + }); + + it("every default has a resolvable provider", () => { + expect(providerForModel(DEFAULT_MAIN_MODEL)).toBe("gemini"); + expect(providerForModel(DEFAULT_TITLE_MODEL)).toBe("gemini"); + expect(providerForModel(DEFAULT_TABULAR_MODEL)).toBe("gemini"); + }); +}); diff --git a/backend/src/lib/__tests__/safeError.test.ts b/backend/src/lib/__tests__/safeError.test.ts new file mode 100644 index 0000000..4bd8373 --- /dev/null +++ b/backend/src/lib/__tests__/safeError.test.ts @@ -0,0 +1,154 @@ +import { describe, it, expect } from "vitest"; +import { + redactSensitiveText, + safeErrorMessage, + safeErrorLog, +} from "../safeError"; + +// --------------------------------------------------------------------------- +// redactSensitiveText +// --------------------------------------------------------------------------- + +describe("redactSensitiveText", () => { + it('redacts the OpenAI "Incorrect API key provided" message', () => { + expect( + redactSensitiveText( + "Incorrect API key provided: sk-proj-abc123def456ghi789.", + ), + ).toBe("Incorrect API key provided: [redacted]."); + }); + + it("keeps the trailing period optional in the incorrect-key message", () => { + expect( + redactSensitiveText("Incorrect API key provided: badkey123"), + ).toBe("Incorrect API key provided: [redacted]"); + }); + + it("redacts secrets after api_key labels", () => { + expect(redactSensitiveText("api_key: mysecret123")).toBe( + "api_key: [redacted]", + ); + expect(redactSensitiveText("api key = mysecret123")).toBe( + "api key = [redacted]", + ); + }); + + it("redacts secrets after token/secret/authorization labels", () => { + expect(redactSensitiveText("token: abcdef123456")).toBe( + "token: [redacted]", + ); + expect(redactSensitiveText("secret is abcdef123456")).toBe( + "secret is [redacted]", + ); + expect(redactSensitiveText("authorization: abcdef123456")).toBe( + "authorization: [redacted]", + ); + }); + + it("leaves short values after labels alone (below 6 chars)", () => { + expect(redactSensitiveText("token: abc")).toBe("token: abc"); + }); + + it("redacts bare OpenAI-style sk- keys anywhere in the text", () => { + expect( + redactSensitiveText("request failed for sk-abc123def456ghi789 today"), + ).toBe("request failed for [redacted] today"); + }); + + it("redacts bare Anthropic-style sk-ant- keys", () => { + expect( + redactSensitiveText("used sk-ant-api03-abc123def456"), + ).toBe("used [redacted]"); + }); + + it("redacts bare Google AIza keys", () => { + expect( + redactSensitiveText("key AIzaSyA1234567890abcdefghij failed"), + ).toBe("key [redacted] failed"); + }); + + it("redacts multiple secrets in one string", () => { + const result = redactSensitiveText( + "first sk-abc123def456ghi789 then AIzaSyA1234567890abcdefghij", + ); + expect(result).toBe("first [redacted] then [redacted]"); + }); + + it("leaves ordinary text unchanged", () => { + expect(redactSensitiveText("Document not found")).toBe( + "Document not found", + ); + }); +}); + +// --------------------------------------------------------------------------- +// safeErrorMessage +// --------------------------------------------------------------------------- + +describe("safeErrorMessage", () => { + it("uses the message of an Error instance", () => { + expect(safeErrorMessage(new Error("boom"))).toBe("boom"); + }); + + it("redacts secrets inside an Error message", () => { + expect( + safeErrorMessage(new Error("bad key sk-abc123def456ghi789")), + ).toBe("bad key [redacted]"); + }); + + it("passes plain strings through (redacted)", () => { + expect(safeErrorMessage("token: abcdef123456")).toBe( + "token: [redacted]", + ); + }); + + it("falls back for non-Error, non-string values", () => { + expect(safeErrorMessage(42)).toBe("Unexpected error"); + expect(safeErrorMessage(null)).toBe("Unexpected error"); + expect(safeErrorMessage({ message: "obj" })).toBe("Unexpected error"); + }); + + it("falls back for an Error with an empty message", () => { + expect(safeErrorMessage(new Error(""))).toBe("Unexpected error"); + }); + + it("honors a custom fallback", () => { + expect(safeErrorMessage(undefined, "Chat failed")).toBe("Chat failed"); + }); +}); + +// --------------------------------------------------------------------------- +// safeErrorLog +// --------------------------------------------------------------------------- + +describe("safeErrorLog", () => { + it("captures name, message, and stack for an Error", () => { + const error = new Error("boom"); + const log = safeErrorLog(error); + expect(log.name).toBe("Error"); + expect(log.message).toBe("boom"); + expect(log.stack).toContain("boom"); + }); + + it("redacts secrets in the message and stack", () => { + const error = new Error("bad key sk-abc123def456ghi789"); + const log = safeErrorLog(error); + expect(log.message).toBe("bad key [redacted]"); + expect(log.stack).not.toContain("sk-abc123def456ghi789"); + }); + + it("falls back to 'Unexpected error' for an empty Error message", () => { + expect(safeErrorLog(new Error("")).message).toBe("Unexpected error"); + }); + + it("omits the stack when the Error has none", () => { + const error = new Error("boom"); + error.stack = undefined; + expect(safeErrorLog(error).stack).toBeUndefined(); + }); + + it("handles non-Error values with a null name and no stack", () => { + const log = safeErrorLog("plain failure"); + expect(log).toEqual({ name: null, message: "plain failure" }); + }); +}); diff --git a/backend/src/lib/__tests__/userDataCleanup.test.ts b/backend/src/lib/__tests__/userDataCleanup.test.ts new file mode 100644 index 0000000..35a8ed1 --- /dev/null +++ b/backend/src/lib/__tests__/userDataCleanup.test.ts @@ -0,0 +1,432 @@ +import { describe, it, expect, vi, beforeEach } from "vitest"; + +vi.mock("../storage", () => ({ + deleteFile: vi.fn(async () => {}), + listFiles: vi.fn(async () => [] as string[]), +})); + +import { deleteFile, listFiles } from "../storage"; +import { + deleteAllUserChats, + deleteAllUserTabularReviews, + deleteUserProjects, + deleteUserAccountData, +} from "../userDataCleanup"; + +const deleteFileMock = vi.mocked(deleteFile); +const listFilesMock = vi.mocked(listFiles); + +type Row = Record; + +/** + * Stateful Supabase mock: deletes and updates mutate `tables`, so tests can + * assert on exactly which rows survived a cleanup call. Supports the chains + * userDataCleanup uses (select/delete/update + eq/in/filter-cs) and can + * inject a delete error per table to exercise error propagation. + */ +function makeDb( + initialTables: Record, + options: { deleteErrors?: Record } = {}, +) { + const tables: Record = {}; + for (const [name, rows] of Object.entries(initialTables)) { + tables[name] = rows.map((row) => ({ ...row })); + } + const db = { + from(table: string) { + const rowsOf = () => tables[table] ?? (tables[table] = []); + let predicate: (row: Row) => boolean = () => true; + let mode: "select" | "delete" | "update" = "select"; + let patch: Row = {}; + const narrow = (next: (row: Row) => boolean) => { + const prev = predicate; + predicate = (row) => prev(row) && next(row); + }; + const query: any = { + select: () => query, + delete: () => { + mode = "delete"; + return query; + }, + update: (value: Row) => { + mode = "update"; + patch = value; + return query; + }, + eq: (column: string, value: unknown) => { + narrow((row) => row[column] === value); + return query; + }, + in: (column: string, values: unknown[]) => { + narrow((row) => values.includes(row[column])); + return query; + }, + filter: (column: string, operator: string, value: string) => { + if (operator !== "cs") return query; + const expected = (JSON.parse(value) as string[]).map((item) => + item.toLowerCase(), + ); + narrow((row) => { + const actual = row[column]; + if (!Array.isArray(actual)) return false; + const normalized = actual.map((item) => + String(item).toLowerCase(), + ); + return expected.every((item) => normalized.includes(item)); + }); + return query; + }, + then: ( + resolve: (value: { data: Row[] | null; error: unknown }) => unknown, + reject?: (reason: unknown) => unknown, + ) => { + let result: { data: Row[] | null; error: unknown }; + if (mode === "delete") { + const message = options.deleteErrors?.[table]; + if (message) { + result = { data: null, error: { message } }; + } else { + tables[table] = rowsOf().filter((row) => !predicate(row)); + result = { data: null, error: null }; + } + } else if (mode === "update") { + for (const row of rowsOf().filter(predicate)) { + Object.assign(row, patch); + } + result = { data: null, error: null }; + } else { + result = { + data: rowsOf().filter(predicate).map((row) => ({ ...row })), + error: null, + }; + } + return Promise.resolve(result).then(resolve, reject); + }, + }; + return query; + }, + }; + return { db: db as any, tables }; +} + +const ids = (rows: Row[] | undefined) => (rows ?? []).map((row) => row.id); + +beforeEach(() => { + deleteFileMock.mockClear(); + deleteFileMock.mockResolvedValue(undefined as never); + listFilesMock.mockClear(); + listFilesMock.mockResolvedValue([]); +}); + +// --------------------------------------------------------------------------- +// deleteAllUserChats +// --------------------------------------------------------------------------- + +describe("deleteAllUserChats", () => { + it("deletes only the target user's assistant and tabular chats", async () => { + const { db, tables } = makeDb({ + chats: [ + { id: "c1", user_id: "u1" }, + { id: "c2", user_id: "u2" }, + ], + tabular_review_chats: [ + { id: "tc1", user_id: "u1" }, + { id: "tc2", user_id: "u2" }, + ], + }); + await deleteAllUserChats(db, "u1"); + expect(ids(tables.chats)).toEqual(["c2"]); + expect(ids(tables.tabular_review_chats)).toEqual(["tc2"]); + }); + + it("surfaces delete failures with context", async () => { + const { db } = makeDb( + { chats: [{ id: "c1", user_id: "u1" }], tabular_review_chats: [] }, + { deleteErrors: { chats: "boom" } }, + ); + await expect(deleteAllUserChats(db, "u1")).rejects.toThrow( + "Failed to delete assistant chats: boom", + ); + }); +}); + +// --------------------------------------------------------------------------- +// deleteAllUserTabularReviews +// --------------------------------------------------------------------------- + +describe("deleteAllUserTabularReviews", () => { + const fixture = () => + makeDb({ + tabular_reviews: [ + { id: "r1", user_id: "u1" }, + { id: "r2", user_id: "u1" }, + { id: "r-other", user_id: "u2" }, + ], + tabular_review_chats: [ + { id: "rc1", review_id: "r1" }, + { id: "rc-other", review_id: "r-other" }, + ], + tabular_review_chat_messages: [ + { id: "rm1", chat_id: "rc1" }, + { id: "rm-other", chat_id: "rc-other" }, + ], + tabular_cells: [ + { id: "cell1", review_id: "r1" }, + { id: "cell-other", review_id: "r-other" }, + ], + }); + + it("cascades messages, chats, and cells before the reviews", async () => { + const { db, tables } = fixture(); + await expect(deleteAllUserTabularReviews(db, "u1")).resolves.toBe(2); + expect(ids(tables.tabular_reviews)).toEqual(["r-other"]); + expect(ids(tables.tabular_review_chats)).toEqual(["rc-other"]); + expect(ids(tables.tabular_review_chat_messages)).toEqual(["rm-other"]); + expect(ids(tables.tabular_cells)).toEqual(["cell-other"]); + }); + + it("returns 0 and deletes nothing for a user with no reviews", async () => { + const { db, tables } = fixture(); + await expect(deleteAllUserTabularReviews(db, "u3")).resolves.toBe(0); + expect(tables.tabular_reviews).toHaveLength(3); + expect(tables.tabular_cells).toHaveLength(2); + }); +}); + +// --------------------------------------------------------------------------- +// deleteUserProjects +// --------------------------------------------------------------------------- + +describe("deleteUserProjects", () => { + const fixture = () => + makeDb({ + projects: [ + { id: "p1", user_id: "u1" }, + { id: "p2", user_id: "u1" }, + { id: "p-other", user_id: "u2" }, + ], + documents: [ + { id: "d1", user_id: "u1", project_id: "p1" }, + { id: "d-loose", user_id: "u1", project_id: null }, + { id: "d-other", user_id: "u2", project_id: "p-other" }, + ], + document_versions: [ + { + id: "v1", + document_id: "d1", + storage_path: "documents/u1/d1/source.pdf", + pdf_storage_path: "documents/u1/d1/converted.pdf", + }, + { + id: "v-other", + document_id: "d-other", + storage_path: "documents/u2/d-other/source.pdf", + pdf_storage_path: null, + }, + ], + chats: [ + { id: "c1", project_id: "p1" }, + { id: "c-other", project_id: "p-other" }, + ], + chat_messages: [ + { id: "m1", chat_id: "c1" }, + { id: "m-other", chat_id: "c-other" }, + ], + tabular_reviews: [ + { id: "r1", project_id: "p1" }, + { id: "r-other", project_id: "p-other" }, + ], + tabular_review_chats: [ + { id: "rc1", review_id: "r1" }, + { id: "rc-other", review_id: "r-other" }, + ], + tabular_review_chat_messages: [ + { id: "rm1", chat_id: "rc1" }, + { id: "rm-other", chat_id: "rc-other" }, + ], + tabular_cells: [ + { id: "cell1", review_id: "r1" }, + { id: "cell-other", review_id: "r-other" }, + ], + project_subfolders: [ + { id: "f1", project_id: "p1" }, + { id: "f-other", project_id: "p-other" }, + ], + }); + + it("cascades project contents and storage files for owned projects", async () => { + const { db, tables } = fixture(); + await expect(deleteUserProjects(db, "u1")).resolves.toBe(2); + + expect(ids(tables.projects)).toEqual(["p-other"]); + expect(ids(tables.documents)).toEqual(["d-loose", "d-other"]); + expect(ids(tables.chats)).toEqual(["c-other"]); + expect(ids(tables.chat_messages)).toEqual(["m-other"]); + expect(ids(tables.tabular_reviews)).toEqual(["r-other"]); + expect(ids(tables.tabular_review_chats)).toEqual(["rc-other"]); + expect(ids(tables.tabular_review_chat_messages)).toEqual(["rm-other"]); + expect(ids(tables.tabular_cells)).toEqual(["cell-other"]); + expect(ids(tables.project_subfolders)).toEqual(["f-other"]); + + const deletedPaths = deleteFileMock.mock.calls.map(([path]) => path); + expect(deletedPaths.sort()).toEqual([ + "documents/u1/d1/converted.pdf", + "documents/u1/d1/source.pdf", + ]); + }); + + it("restricts deletion to the requested owned projects", async () => { + const { db, tables } = fixture(); + // p-other belongs to u2, so requesting it must not delete anything of theirs. + await expect( + deleteUserProjects(db, "u1", ["p2", "p-other", "p2"]), + ).resolves.toBe(1); + expect(ids(tables.projects)).toEqual(["p1", "p-other"]); + expect(ids(tables.documents)).toEqual(["d1", "d-loose", "d-other"]); + }); + + it("returns 0 for an explicitly empty project list", async () => { + const { db, tables } = fixture(); + await expect(deleteUserProjects(db, "u1", [])).resolves.toBe(0); + expect(tables.projects).toHaveLength(3); + }); + + it("returns 0 when the user owns no projects", async () => { + const { db, tables } = fixture(); + await expect(deleteUserProjects(db, "u3")).resolves.toBe(0); + expect(tables.projects).toHaveLength(3); + expect(deleteFileMock).not.toHaveBeenCalled(); + }); +}); + +// --------------------------------------------------------------------------- +// deleteUserAccountData +// --------------------------------------------------------------------------- + +describe("deleteUserAccountData", () => { + const fixture = () => + makeDb({ + projects: [ + { id: "p1", user_id: "u1", shared_with: [] }, + { + id: "p-other", + user_id: "u2", + shared_with: ["u1@example.com", " U1@Example.com ", "keep@example.com"], + }, + ], + tabular_reviews: [ + { id: "r1", user_id: "u1", shared_with: [] }, + { id: "r-other", user_id: "u2", shared_with: ["u1@example.com"] }, + ], + documents: [ + { id: "d1", user_id: "u1", project_id: null }, + // Guest doc uploaded by another user into u1's project: deleted too. + { id: "d-guest", user_id: "u2", project_id: "p1" }, + { id: "d-other", user_id: "u2", project_id: "p-other" }, + ], + document_versions: [ + { + id: "v1", + document_id: "d1", + storage_path: "documents/u1/d1/source.pdf", + pdf_storage_path: "documents/u1/d1/converted.pdf", + }, + { + id: "v-guest", + document_id: "d-guest", + storage_path: "documents/u2/d-guest/source.docx", + pdf_storage_path: null, + }, + { + id: "v-other", + document_id: "d-other", + storage_path: "documents/u2/d-other/source.pdf", + pdf_storage_path: null, + }, + ], + chats: [ + { id: "c1", user_id: "u1" }, + { id: "c-other", user_id: "u2" }, + ], + tabular_review_chats: [{ id: "rc1", user_id: "u1" }], + project_subfolders: [{ id: "f1", user_id: "u1" }], + hidden_workflows: [{ id: "h1", user_id: "u1" }], + workflow_open_source_submissions: [ + { id: "s1", submitted_by_user_id: "u1" }, + ], + workflow_shares: [ + { id: "ws-by", shared_by_user_id: "u1", shared_with_email: "x@y.z" }, + { + id: "ws-to", + shared_by_user_id: "u2", + shared_with_email: "u1@example.com", + }, + { + id: "ws-keep", + shared_by_user_id: "u2", + shared_with_email: "keep@example.com", + }, + ], + workflows: [ + { id: "w1", user_id: "u1" }, + { id: "w-other", user_id: "u2" }, + ], + }); + + it("removes the user's rows, files, and share references everywhere", async () => { + const { db, tables } = fixture(); + listFilesMock.mockResolvedValue(["documents/u1/orphan.bin"]); + + await deleteUserAccountData(db, "u1", " U1@Example.COM "); + + // Owned docs and guest docs inside owned projects are gone. + expect(ids(tables.documents)).toEqual(["d-other"]); + expect(ids(tables.projects)).toEqual(["p-other"]); + expect(ids(tables.chats)).toEqual(["c-other"]); + expect(tables.tabular_review_chats).toEqual([]); + expect(ids(tables.tabular_reviews)).toEqual(["r-other"]); + expect(tables.project_subfolders).toEqual([]); + expect(tables.hidden_workflows).toEqual([]); + expect(tables.workflow_open_source_submissions).toEqual([]); + expect(ids(tables.workflows)).toEqual(["w-other"]); + + // Shares by the user and shares to the user's email are both removed. + expect(ids(tables.workflow_shares)).toEqual(["ws-keep"]); + + // The email is scrubbed from other users' shared_with lists + // (case-insensitively), preserving other collaborators. + expect(tables.projects[0].shared_with).toEqual(["keep@example.com"]); + expect(tables.tabular_reviews[0].shared_with).toEqual([]); + + // Version files for deleted docs plus orphans under the user's prefix. + const deletedPaths = deleteFileMock.mock.calls.map(([path]) => path); + expect(deletedPaths.sort()).toEqual([ + "documents/u1/d1/converted.pdf", + "documents/u1/d1/source.pdf", + "documents/u1/orphan.bin", + "documents/u2/d-guest/source.docx", + ]); + expect(listFilesMock).toHaveBeenCalledWith("documents/u1/"); + }); + + it("treats storage prefix cleanup as best-effort", async () => { + const { db, tables } = fixture(); + listFilesMock.mockRejectedValue(new Error("storage unavailable")); + await expect( + deleteUserAccountData(db, "u1", "u1@example.com"), + ).resolves.toBeUndefined(); + expect(ids(tables.documents)).toEqual(["d-other"]); + }); + + it("skips shared_with scrubbing when no email is known", async () => { + const { db, tables } = fixture(); + await deleteUserAccountData(db, "u1", null); + // Rows referencing the email by value are left in place... + expect(tables.projects.find((row) => row.id === "p-other")?.shared_with) + .toContain("u1@example.com"); + expect(ids(tables.workflow_shares)).toEqual(["ws-to", "ws-keep"]); + // ...but the user's own data is still deleted. + expect(ids(tables.documents)).toEqual(["d-other"]); + expect(tables.workflows.map((row) => row.id)).toEqual(["w-other"]); + }); +}); diff --git a/backend/src/lib/__tests__/userLookup.test.ts b/backend/src/lib/__tests__/userLookup.test.ts new file mode 100644 index 0000000..c954bf4 --- /dev/null +++ b/backend/src/lib/__tests__/userLookup.test.ts @@ -0,0 +1,267 @@ +import { describe, it, expect } from "vitest"; +import { + normalizeEmail, + normalizeDisplayName, + loadProfileUsersByEmail, + findProfileUserByEmail, + findMissingUserEmails, + syncProfileEmail, +} from "../userLookup"; + +type Row = Record; + +/** + * Minimal user_profiles-shaped Supabase mock. Supports the query chains + * userLookup uses (select/eq/in/not + single-row readers) plus insert and + * update so syncProfileEmail can be exercised end to end. + */ +function makeDb(initialRows: Row[]) { + const tables: Record = { + user_profiles: initialRows.map((row) => ({ ...row })), + }; + return { + tables, + from(table: string) { + const all = () => tables[table] ?? []; + let predicate: (row: Row) => boolean = () => true; + let mode: "select" | "insert" | "update" = "select"; + let pendingRow: Row = {}; + const narrow = (next: (row: Row) => boolean) => { + const prev = predicate; + predicate = (row) => prev(row) && next(row); + }; + const query: any = { + select: () => query, + insert: (row: Row) => { + mode = "insert"; + pendingRow = row; + return query; + }, + update: (patch: Row) => { + mode = "update"; + pendingRow = patch; + return query; + }, + eq: (column: string, value: unknown) => { + narrow((row) => row[column] === value); + return query; + }, + in: (column: string, values: unknown[]) => { + narrow((row) => values.includes(row[column])); + return query; + }, + not: (column: string, operator: string, value: unknown) => { + if (operator === "is" && value === null) { + narrow((row) => row[column] != null); + } + return query; + }, + maybeSingle: async () => ({ + data: all().filter(predicate)[0] ?? null, + error: null, + }), + then: ( + resolve: (value: { data: Row[] | null; error: null }) => unknown, + reject?: (reason: unknown) => unknown, + ) => { + if (mode === "insert") { + all().push({ ...pendingRow }); + return Promise.resolve({ data: null, error: null }).then( + resolve, + reject, + ); + } + if (mode === "update") { + for (const row of all().filter(predicate)) { + Object.assign(row, pendingRow); + } + return Promise.resolve({ data: null, error: null }).then( + resolve, + reject, + ); + } + return Promise.resolve({ + data: all().filter(predicate), + error: null, + }).then(resolve, reject); + }, + }; + return query; + }, + }; +} + +// --------------------------------------------------------------------------- +// normalizeEmail / normalizeDisplayName +// --------------------------------------------------------------------------- + +describe("normalizeEmail", () => { + it("trims and lowercases", () => { + expect(normalizeEmail(" User@Example.COM ")).toBe("user@example.com"); + }); + + it("returns empty string for non-strings", () => { + expect(normalizeEmail(null)).toBe(""); + expect(normalizeEmail(undefined)).toBe(""); + expect(normalizeEmail(42)).toBe(""); + }); +}); + +describe("normalizeDisplayName", () => { + it("trims usable names", () => { + expect(normalizeDisplayName(" Ada Lovelace ")).toBe("Ada Lovelace"); + }); + + it("returns null for empty or non-string values", () => { + expect(normalizeDisplayName(" ")).toBeNull(); + expect(normalizeDisplayName("")).toBeNull(); + expect(normalizeDisplayName(null)).toBeNull(); + expect(normalizeDisplayName(7)).toBeNull(); + }); +}); + +// --------------------------------------------------------------------------- +// loadProfileUsersByEmail +// --------------------------------------------------------------------------- + +describe("loadProfileUsersByEmail", () => { + it("indexes profiles by normalized email and by id", async () => { + const db = makeDb([ + { user_id: "u1", email: "Alice@Example.com", display_name: " Alice " }, + { user_id: "u2", email: "bob@example.com", display_name: null }, + ]); + const { userByEmail, userById } = await loadProfileUsersByEmail( + db as any, + ); + expect(userByEmail.get("alice@example.com")).toEqual({ + id: "u1", + email: "alice@example.com", + display_name: "Alice", + }); + expect(userById.get("u2")).toEqual({ + id: "u2", + email: "bob@example.com", + display_name: null, + }); + expect(userByEmail.size).toBe(2); + }); + + it("skips rows whose email normalizes to empty", async () => { + const db = makeDb([ + { user_id: "u1", email: " ", display_name: null }, + { user_id: "u2", email: "ok@example.com", display_name: null }, + ]); + const { userByEmail } = await loadProfileUsersByEmail(db as any); + expect(userByEmail.size).toBe(1); + expect(userByEmail.has("ok@example.com")).toBe(true); + }); +}); + +// --------------------------------------------------------------------------- +// findProfileUserByEmail +// --------------------------------------------------------------------------- + +describe("findProfileUserByEmail", () => { + const rows = [ + { user_id: "u1", email: "alice@example.com", display_name: "Alice" }, + ]; + + it("finds a profile by normalized email", async () => { + const db = makeDb(rows); + await expect( + findProfileUserByEmail(db as any, " ALICE@example.com "), + ).resolves.toEqual({ + id: "u1", + email: "alice@example.com", + display_name: "Alice", + }); + }); + + it("returns null when no profile matches", async () => { + const db = makeDb(rows); + await expect( + findProfileUserByEmail(db as any, "missing@example.com"), + ).resolves.toBeNull(); + }); + + it("returns null without querying for empty input", async () => { + const db = makeDb(rows); + await expect(findProfileUserByEmail(db as any, " ")).resolves.toBeNull(); + }); +}); + +// --------------------------------------------------------------------------- +// findMissingUserEmails +// --------------------------------------------------------------------------- + +describe("findMissingUserEmails", () => { + const db = makeDb([ + { user_id: "u1", email: "alice@example.com", display_name: null }, + ]); + + it("returns only emails with no matching profile", async () => { + await expect( + findMissingUserEmails(db as any, [ + "Alice@Example.com", + "carol@example.com", + ]), + ).resolves.toEqual(["carol@example.com"]); + }); + + it("dedupes and drops empty entries before querying", async () => { + await expect( + findMissingUserEmails(db as any, [ + "carol@example.com", + " CAROL@example.com ", + "", + " ", + ]), + ).resolves.toEqual(["carol@example.com"]); + }); + + it("returns [] for an empty input list", async () => { + await expect(findMissingUserEmails(db as any, [])).resolves.toEqual([]); + }); +}); + +// --------------------------------------------------------------------------- +// syncProfileEmail +// --------------------------------------------------------------------------- + +describe("syncProfileEmail", () => { + it("inserts a profile row when none exists", async () => { + const db = makeDb([]); + const result = await syncProfileEmail(db as any, "u1", "New@Example.com"); + expect(result).toBeNull(); + expect(db.tables.user_profiles).toEqual([ + { user_id: "u1", email: "new@example.com" }, + ]); + }); + + it("is a no-op when the stored email already matches (case-insensitive)", async () => { + const db = makeDb([ + { user_id: "u1", email: "Same@Example.com", display_name: null }, + ]); + const result = await syncProfileEmail(db as any, "u1", "same@example.com"); + expect(result).toBeNull(); + expect(db.tables.user_profiles[0].email).toBe("Same@Example.com"); + }); + + it("updates the stored email when it changed", async () => { + const db = makeDb([ + { user_id: "u1", email: "old@example.com", display_name: null }, + ]); + const result = await syncProfileEmail(db as any, "u1", "New@Example.com"); + expect(result).toBeNull(); + expect(db.tables.user_profiles[0].email).toBe("new@example.com"); + expect(db.tables.user_profiles[0].updated_at).toEqual(expect.any(String)); + }); + + it("returns null without touching the table for missing inputs", async () => { + const db = makeDb([]); + await expect(syncProfileEmail(db as any, "", "a@b.com")).resolves.toBeNull(); + await expect(syncProfileEmail(db as any, "u1", null)).resolves.toBeNull(); + await expect(syncProfileEmail(db as any, "u1", " ")).resolves.toBeNull(); + expect(db.tables.user_profiles).toEqual([]); + }); +}); diff --git a/backend/vitest.config.mts b/backend/vitest.config.mts index 2f39f1d..b945289 100644 --- a/backend/vitest.config.mts +++ b/backend/vitest.config.mts @@ -16,20 +16,22 @@ export default defineConfig({ reporter: ["text", "lcov"], include: ["src/lib/**"], // No-regression RATCHET floor, not a target. src/lib/** spans the - // small, tested libs (access, storage keys/dispositions, - // downloadTokens, userApiKeys provider/env checks, chat doc - // resolution) AND the large, still-untested feature libs - // (courtlistener, mcp, chat tool dispatch, llm providers, - // spreadsheet/docx handling), so the global number is low. - // Measured on this tree: 2.58% statements, 2.00% branches, - // 4.61% functions, 2.58% lines. These floors sit just below - // that (rounded down to whole percents) so CI fails on a - // *drop*; raise them as the feature-lib backlog gets covered. + // tested libs (access, storage keys/dispositions, downloadTokens, + // userApiKeys provider/env checks, chat doc resolution, safeError, + // llm model resolution, chat citations, userLookup, + // documentVersions, userDataCleanup) AND the large, still-untested + // feature libs (courtlistener, mcp, chat tool dispatch, llm + // providers, spreadsheet/docx handling), so the global number is + // still low. Measured on this tree: 11.18% statements, 10.98% + // branches, 14.43% functions, 10.91% lines. These floors sit just + // below that (rounded down to whole percents) so CI fails on a + // *drop*. Floors only go up: when you add tests, raise them in the + // same PR. Backlog + per-area status: docs/testing-coverage.md. thresholds: { - statements: 2, - branches: 2, - functions: 4, - lines: 2, + statements: 11, + branches: 10, + functions: 14, + lines: 10, }, }, }, diff --git a/docs/testing-coverage.md b/docs/testing-coverage.md new file mode 100644 index 0000000..c62789c --- /dev/null +++ b/docs/testing-coverage.md @@ -0,0 +1,122 @@ +# Backend unit-test coverage + +The backend has a Vitest unit-test harness over `backend/src/lib/**`. This doc +tracks what is covered, what still needs tests, and how the coverage ratchet +works — so you can pick up a checkbox below and land it as a small PR. + +## Running the tests + +```bash +cd backend +npm install +npm test # run all unit tests +npm run test:coverage # same, plus the per-file coverage table + floor check +``` + +Tests live in `backend/src/lib/__tests__/*.test.ts`. Read a couple of the +existing suites first (`access.test.ts`, `userDataCleanup.test.ts`) and match +their conventions: plain in-memory Supabase query mocks (no network, no real +database), one `describe` block per exported function, and tests that assert +current behavior. + +## Current coverage (measured 2026-07) + +Per-area statement coverage from `npm run test:coverage`: + +| Lib area | % statements | Tested? | +| --- | ---: | :---: | +| `lib/safeError.ts` | 100 | ✓ | +| `lib/userDataCleanup.ts` | 100 | ✓ | +| `lib/llm/models.ts` | 100 | ✓ | +| `lib/documentVersions.ts` | 98 | ✓ | +| `lib/chat/citations.ts` | 98 | ✓ | +| `lib/userLookup.ts` | 91 | ✓ | +| `lib/downloadTokens.ts` | 87 | ✓ | +| `lib/chat/types.ts` | 85 | ✓ | +| `lib/access.ts` | 76 | ✓ | +| `lib/storage.ts` | 33 | partial — key/disposition helpers only | +| `lib/userApiKeys.ts` | 13 | partial — provider/env helpers only | +| `lib/documentTypes.ts`, `lib/userSettings.ts`, `lib/upload.ts`, `lib/officeText.ts` | 0 | ✗ | +| `lib/convert.ts`, `lib/spreadsheet.ts`, `lib/docxTrackedChanges.ts` | 0 | ✗ | +| `lib/userDataExport.ts` | 0 | ✗ | +| `lib/courtlistener.ts`, `lib/systemWorkflows.ts` | 0 | ✗ | +| `lib/chat/prompts.ts`, `lib/chat/contextBuilders.ts`, `lib/chat/streaming.ts` | 0 | ✗ | +| `lib/chat/tools/**` (schemas, documentOps, toolDispatcher) | 0 | ✗ | +| `lib/llm/**` (providers, tools, index, rawStreamLog) | ~4 | ✗ (only models.ts) | +| `lib/mcp/**` (client, servers, oauth, types) | 0 | ✗ | + +Global: **11.18% statements / 10.98% branches / 14.43% functions / 10.91% +lines**. The global number is low because `src/lib/**` includes several very +large feature libs (toolDispatcher, documentOps, systemWorkflows, +courtlistener, docxTrackedChanges) that dominate the line count. + +## TODO — untested libs, in priority order + +Each item is meant to be one self-contained PR: add the suite, then raise the +floors in `backend/vitest.config.mts` to just below the new measured numbers. +Size is a rough guess: S ≈ an hour, M ≈ an afternoon. + +- [ ] `lib/documentTypes.ts` — pure catalog/lookup of document types; assert + known types resolve and unknown inputs fall back sanely. (S) +- [ ] `lib/chat/prompts.ts` — pure prompt builders; assert key instructions and + interpolated values appear in the output strings. (S) +- [ ] `lib/userSettings.ts` — title/tabular model resolution from which API + keys a user has; reuse the Supabase mock pattern from + `userLookup.test.ts`. (S) +- [ ] `lib/upload.ts` — multer wrapper: assert LIMIT_FILE_SIZE maps to a 413 + with the right message and other errors pass through. (S) +- [ ] `lib/officeText.ts` — office XML text extraction; build a tiny in-memory + zip fixture with JSZip and assert extracted/decoded text. (S) +- [ ] `lib/chat/tools/toolSchemas.ts` — assert every tool schema has a name, + description, and well-formed parameters (guards against schema drift). (S) +- [ ] `lib/userApiKeys.ts` (rest) — encrypt/decrypt round-trip and DB + load/store paths with a mocked Supabase client. (M) +- [ ] `lib/storage.ts` (rest) — S3 upload/download/list/delete wrappers with a + mocked AWS SDK client. (M) +- [ ] `lib/userDataExport.ts` — export assembly: given seeded mock tables, + assert the export contains the user's data and nobody else's. (M) +- [ ] `lib/spreadsheet.ts` — parse a small in-memory xlsx fixture; assert sheet + and cell extraction, including empty/edge cells. (M) +- [ ] `lib/chat/contextBuilders.ts` — context assembly from doc stores; assert + doc labels, truncation, and ordering. (M) +- [ ] `lib/docxTrackedChanges.ts` — tracked-changes XML round-trip on a minimal + docx fixture: insert/delete runs, accept/reject. High value: document + integrity. (M) +- [ ] `lib/courtlistener.ts` — API client with mocked fetch: query building, + pagination, and error paths. Legal-research correctness. (M) +- [ ] `lib/systemWorkflows.ts` — mostly data: assert workflow definitions are + well-formed (unique ids, non-empty skill markdown). (S) +- [ ] `lib/llm/tools.ts` + `lib/llm/index.ts` — provider-neutral tool plumbing + and provider selection with mocked provider modules. (M) +- [ ] `lib/mcp/types.ts` + `lib/mcp/servers.ts` — server config validation and + allow-listing logic; security relevant. (M) +- [ ] `lib/mcp/client.ts` + `lib/mcp/oauth.ts` — connection lifecycle and OAuth + token handling with a mocked MCP SDK; security relevant. (M) +- [ ] `lib/llm/rawStreamLog.ts` — log path construction and redaction with a + mocked fs. (S) +- [ ] `lib/chat/tools/documentOps.ts` — start with the pure helpers (diff/match + utilities), not the full tool handlers. (M) +- [ ] `lib/chat/tools/toolDispatcher.ts` — dispatch table routing and argument + validation with stubbed tools; don't try to cover every tool body. (M) +- [ ] `lib/chat/streaming.ts` + `lib/llm/{claude,gemini,openai}.ts` — streaming + loops and provider adapters; hardest to unit test, consider extracting + pure chunk-parsing helpers first. (M) + +Not worth unit testing directly: `lib/supabase.ts` and `lib/convert.ts` are +thin wrappers around external services (Supabase auth, LibreOffice); they are +better exercised by the e2e suite. + +## Ratchet policy + +`backend/vitest.config.mts` enforces global coverage **floors** (currently +statements 11 / branches 10 / functions 14 / lines 10). They are a +no-regression ratchet, not a target: + +- **Floors only go up.** Never lower them to get a PR green — that means your + change removed tested behavior or added a large untested lib; add tests + instead. +- **Raise them in the same PR that adds tests.** After your suite passes, run + `npm run test:coverage`, take the new global numbers, and set each floor to + the measured value rounded down to a whole percent. +- Keep the measured numbers in the config comment and the table above honest + when you do.