From 1fb1841173ea1fec783294903e6b583e3f4835f9 Mon Sep 17 00:00:00 2001 From: tegwick Date: Thu, 30 Jul 2026 20:15:53 +0200 Subject: [PATCH] Rematch documents by content fingerprint on upload and session open. Re-uploading the same PDF attaches bytes to the existing document (keeping evidence) instead of creating a twin. On restore, merge duplicate fingerprint groups and migrate annotations onto the evidence-bearing primary. --- src/app/App.tsx | 2 + .../DocumentFingerprintReconciler.tsx | 50 ++++ src/app/sessions/SampleSessions.tsx | 13 +- src/app/sessions/UploadDropzone.tsx | 29 ++- src/app/sessions/document-rematch.test.ts | 150 ++++++++++++ src/app/sessions/document-rematch.ts | 213 ++++++++++++++++++ src/app/sessions/importSessionZip.ts | 21 +- 7 files changed, 464 insertions(+), 14 deletions(-) create mode 100644 src/app/sessions/DocumentFingerprintReconciler.tsx create mode 100644 src/app/sessions/document-rematch.test.ts create mode 100644 src/app/sessions/document-rematch.ts diff --git a/src/app/App.tsx b/src/app/App.tsx index 50fcee3..33d6ef1 100644 --- a/src/app/App.tsx +++ b/src/app/App.tsx @@ -38,6 +38,7 @@ import { CaptureLinkPersister } from "./forms/CaptureLinkPersister"; import { loadCaptureState } from "./forms/capture-persistence"; import { FormsApp } from "./forms/FormsApp"; import { ReviewLayout } from "./ReviewLayout"; +import { DocumentFingerprintReconciler } from "./sessions/DocumentFingerprintReconciler"; import { CreateFirstSession, @@ -176,6 +177,7 @@ function SessionScopedTree({ mode }: { mode: AppMode }) { ? { initialLinks: restoredCapture.evidenceLinks } : {})} > + {mode === "forms" ? ( (null); + + useEffect(() => { + // Run once after each engine restore/hydrate revision bump. + if (ranForRevision.current === revision) return; + ranForRevision.current = revision; + + const { remappedIds, documentsRemoved } = + mergeDuplicateDocumentsByFingerprint(engine, byteStore); + + if (activeId && remappedIds.has(activeId)) { + setId(remappedIds.get(activeId)!); + return; + } + + // Bytes may have moved onto the still-active primary; force the viewer + // to re-resolve the blob URL (same id alone may not remount). + if (documentsRemoved > 0 && activeId && byteStore.has(activeId)) { + const keep = activeId; + setId(null); + queueMicrotask(() => setId(keep)); + } + // Intentionally omit activeId from deps: we only reconcile on revision. + // eslint-disable-next-line react-hooks/exhaustive-deps -- see above + }, [engine, byteStore, revision, setId]); + + return null; +} diff --git a/src/app/sessions/SampleSessions.tsx b/src/app/sessions/SampleSessions.tsx index dfb2a79..94125f1 100644 --- a/src/app/sessions/SampleSessions.tsx +++ b/src/app/sessions/SampleSessions.tsx @@ -21,6 +21,8 @@ import { usePdfByteStore, } from "@work/index"; +import { registerOrRematchDocument } from "./document-rematch"; + import manifest from "../../../fixtures/pdfs/manifest.json"; interface Fixture { @@ -60,16 +62,15 @@ export function SampleSessions() { const { document, representation } = await ingestPdf(bytes, { filename: fixture.filename, }); - // Push the bytes into the byte store so the viewer can mount them via - // the same blob URL machinery used by the upload path. The document - // record carries the blob URL on `uri` for the viewer adapter. + // Same blob URL machinery as the upload path, with fingerprint rematch + // so reloading a sample reattaches to an existing session document. const record = byteStore.put(document.id, bytes); - engine.documents.register({ + const outcome = registerOrRematchDocument(engine, byteStore, { document: { ...document, uri: record.blobUrl }, representation, }); - setByFixture((prev) => ({ ...prev, [fixture.id]: document.id })); - setId(document.id); + setByFixture((prev) => ({ ...prev, [fixture.id]: outcome.documentId })); + setId(outcome.documentId); } catch (err) { setError(err instanceof Error ? err.message : String(err)); } finally { diff --git a/src/app/sessions/UploadDropzone.tsx b/src/app/sessions/UploadDropzone.tsx index d3d0d02..6273f57 100644 --- a/src/app/sessions/UploadDropzone.tsx +++ b/src/app/sessions/UploadDropzone.tsx @@ -6,9 +6,9 @@ * 1. read each File as bytes, * 2. run the source-layer `ingestPdfFromFile` (mints the blob URL * via the session's `PdfByteStore`), - * 3. register the resulting `{document, representation}` with the - * engine, - * 4. activate the most-recently-uploaded document. + * 3. register — or rematch by content fingerprint to an existing + * document (so re-upload reattaches PDF bytes to prior evidence), + * 4. activate the most-recently-uploaded / rematched document. * * Failures (non-PDFs, ingest errors) are surfaced inline above the * dropzone; the caller doesn't need a separate toast for them. @@ -23,10 +23,14 @@ import { usePdfByteStore, } from "@work/index"; +import { registerOrRematchDocument } from "./document-rematch"; + interface UploadEntry { readonly file: File; status: "queued" | "uploading" | "done" | "error"; error?: string; + /** True when upload reattached bytes to an existing fingerprint match. */ + rematched?: boolean; } export interface UploadDropzoneProps { @@ -67,10 +71,18 @@ export function UploadDropzone({ onUploaded }: UploadDropzoneProps) { entry.file, byteStore, ); - engine.documents.register({ document, representation }); + const outcome = registerOrRematchDocument(engine, byteStore, { + document, + representation, + }); entry.status = "done"; - lastDocumentId = document.id; - onUploaded?.(document.id); + if (outcome.kind === "rematched") { + entry.error = undefined; + // Surface rematch in the progress list without treating it as failure. + (entry as UploadEntry & { rematched?: boolean }).rematched = true; + } + lastDocumentId = outcome.documentId; + onUploaded?.(outcome.documentId); } catch (err) { entry.status = "error"; entry.error = err instanceof Error ? err.message : String(err); @@ -178,7 +190,10 @@ export function UploadDropzone({ onUploaded }: UploadDropzoneProps) { : "#333", }} > - {entry.file.name} — {entry.status} + {entry.file.name} —{" "} + {entry.status === "done" && entry.rematched + ? "done (matched existing document by content hash)" + : entry.status} {entry.error ? `: ${entry.error}` : ""} ))} diff --git a/src/app/sessions/document-rematch.test.ts b/src/app/sessions/document-rematch.test.ts new file mode 100644 index 0000000..441eb2f --- /dev/null +++ b/src/app/sessions/document-rematch.test.ts @@ -0,0 +1,150 @@ +import { describe, expect, it, vi } from "vitest"; + +import type { Document, DocumentRepresentation } from "@shared/document"; +import type { DocumentId, RepresentationId } from "@shared/ids"; +import { createEngine } from "@engine/index"; +import { createPdfByteStore } from "@source/index"; + +import { + mergeDuplicateDocumentsByFingerprint, + pickPrimaryDocument, + registerOrRematchDocument, + rekeyByteStore, +} from "./document-rematch"; + +function makeDoc( + id: string, + fingerprint: string, + title = "doc.pdf", +): Document { + const now = "2026-07-30T00:00:00.000Z"; + return { + id: id as DocumentId, + mediaType: "application/pdf", + title, + fingerprint, + createdAt: now, + updatedAt: now, + }; +} + +function makeRep(docId: string, repId: string): DocumentRepresentation { + return { + id: repId as RepresentationId, + documentId: docId as DocumentId, + representationType: "pdf-text", + contentHash: "hash", + canonicalText: "hello", + pageMap: [{ page: 1, width: 1, height: 1 }], + offsetMap: [{ page: 1, globalStart: 0, globalEnd: 5, pageLength: 5 }], + generatedAt: "2026-07-30T00:00:00.000Z", + }; +} + +describe("document-rematch", () => { + it("rekeys byte store from new id onto existing id", () => { + const store = createPdfByteStore({ + createObjectURL: () => "blob:x", + revokeObjectURL: () => {}, + }); + const from = "doc_new" as DocumentId; + const to = "doc_old" as DocumentId; + store.put(from, new Uint8Array([1, 2, 3])); + expect(rekeyByteStore(store, from, to)).toBe(true); + expect(store.has(from)).toBe(false); + expect(store.get(to)?.bytes).toEqual(new Uint8Array([1, 2, 3])); + }); + + it("registerOrRematch attaches bytes to existing fingerprint and skips twin", () => { + const engine = createEngine(); + const store = createPdfByteStore({ + createObjectURL: vi.fn(() => `blob:${Math.random()}`), + revokeObjectURL: () => {}, + }); + const existing = makeDoc("doc_old", "fp-aaa", "old.pdf"); + engine.documents.register({ + document: existing, + representation: makeRep("doc_old", "rep_old"), + }); + + const fresh = makeDoc("doc_new", "fp-aaa", "reupload.pdf"); + store.put("doc_new" as DocumentId, new Uint8Array([9, 9])); + const result = registerOrRematchDocument(engine, store, { + document: { ...fresh, uri: "blob:new" }, + representation: makeRep("doc_new", "rep_new"), + }); + + expect(result.kind).toBe("rematched"); + if (result.kind !== "rematched") return; + expect(result.documentId).toBe("doc_old"); + expect(engine.documents.list()).toHaveLength(1); + expect(store.has("doc_new" as DocumentId)).toBe(false); + expect(store.get("doc_old" as DocumentId)?.bytes).toEqual( + new Uint8Array([9, 9]), + ); + }); + + it("registerOrRematch registers when fingerprint is new", () => { + const engine = createEngine(); + const store = createPdfByteStore({ + createObjectURL: () => "blob:y", + revokeObjectURL: () => {}, + }); + const doc = makeDoc("doc_1", "fp-new"); + store.put("doc_1" as DocumentId, new Uint8Array([1])); + const result = registerOrRematchDocument(engine, store, { + document: doc, + representation: makeRep("doc_1", "rep_1"), + }); + expect(result.kind).toBe("registered"); + expect(engine.documents.list()).toHaveLength(1); + }); + + it("mergeDuplicateDocumentsByFingerprint prefers evidence-bearing primary", () => { + const engine = createEngine(); + const store = createPdfByteStore({ + createObjectURL: vi.fn(() => `blob:${Math.random()}`), + revokeObjectURL: () => {}, + }); + + const withEvidence = makeDoc("doc_a", "fp-same", "a.pdf"); + const withBytesOnly = makeDoc("doc_b", "fp-same", "b.pdf"); + engine.documents.register({ + document: withEvidence, + representation: makeRep("doc_a", "rep_a"), + }); + engine.documents.register({ + document: withBytesOnly, + representation: makeRep("doc_b", "rep_b"), + }); + store.put("doc_b" as DocumentId, new Uint8Array([7, 7, 7])); + + const ann = engine.annotations.create({ + documentId: "doc_a" as DocumentId, + selectors: [{ type: "TextQuoteSelector", exact: "hello" }], + quote: "hello", + }); + engine.evidence.create({ + annotationIds: [ann.id], + commentary: "note", + }); + + const primary = pickPrimaryDocument(engine, store, [ + withBytesOnly, + withEvidence, + ]); + expect(primary.id).toBe("doc_a"); + + const stats = mergeDuplicateDocumentsByFingerprint(engine, store); + expect(stats.groupsMerged).toBe(1); + expect(stats.documentsRemoved).toBe(1); + expect(engine.documents.list().map((d) => d.id)).toEqual(["doc_a"]); + expect(store.get("doc_a" as DocumentId)?.bytes).toEqual( + new Uint8Array([7, 7, 7]), + ); + expect(engine.annotations.listByDocument("doc_a" as DocumentId)).toHaveLength( + 1, + ); + expect(engine.evidence.listByDocument("doc_a" as DocumentId)).toHaveLength(1); + }); +}); diff --git a/src/app/sessions/document-rematch.ts b/src/app/sessions/document-rematch.ts new file mode 100644 index 0000000..6ace0a3 --- /dev/null +++ b/src/app/sessions/document-rematch.ts @@ -0,0 +1,213 @@ +/** + * Document identity by content fingerprint (SHA-256 of source bytes). + * + * PDFs and evidence are stored under different lifecycle concerns (bytes in + * the byte store / IndexedDB; annotations + evidence in the engine snapshot). + * They re-join via `Document.fingerprint` so a re-upload of the same file + * attaches bytes to the existing document instead of creating a twin. + * + * Also collapses accidental duplicates (e.g. upload after a reload lost + * bytes but kept the evidence-bearing document record). + */ + +import type { Document, DocumentRepresentation } from "@shared/document"; +import type { DocumentId } from "@shared/ids"; +import type { Engine } from "@engine/index"; +import type { PdfByteStore } from "@source/index"; + +export type RegisterOrRematchResult = + | { + readonly kind: "registered"; + readonly documentId: DocumentId; + readonly fingerprint: string | undefined; + } + | { + readonly kind: "rematched"; + readonly documentId: DocumentId; + readonly fingerprint: string; + readonly discardedDocumentId: DocumentId; + }; + +function evidenceCount(engine: Engine, documentId: DocumentId): number { + return engine.evidence.listByDocument(documentId).length; +} + +function annotationCount(engine: Engine, documentId: DocumentId): number { + return engine.annotations.listByDocument(documentId).length; +} + +/** + * Prefer the document that already carries user work (evidence / annotations), + * then one that already has bytes, then the first seen. + */ +export function pickPrimaryDocument( + engine: Engine, + byteStore: PdfByteStore, + candidates: readonly Document[], +): Document { + if (candidates.length === 0) { + throw new Error("pickPrimaryDocument: empty candidates"); + } + if (candidates.length === 1) return candidates[0]!; + + const scored = candidates.map((doc) => ({ + doc, + evidence: evidenceCount(engine, doc.id), + annotations: annotationCount(engine, doc.id), + hasBytes: byteStore.has(doc.id) ? 1 : 0, + })); + scored.sort((a, b) => { + if (b.evidence !== a.evidence) return b.evidence - a.evidence; + if (b.annotations !== a.annotations) return b.annotations - a.annotations; + if (b.hasBytes !== a.hasBytes) return b.hasBytes - a.hasBytes; + return a.doc.createdAt.localeCompare(b.doc.createdAt); + }); + return scored[0]!.doc; +} + +export function findDocumentsByFingerprint( + engine: Engine, + fingerprint: string, +): readonly Document[] { + if (!fingerprint) return []; + return engine.documents.list().filter((d) => d.fingerprint === fingerprint); +} + +/** + * Move in-memory / durable bytes from `fromId` onto `toId` and drop `fromId`. + */ +export function rekeyByteStore( + byteStore: PdfByteStore, + fromId: DocumentId, + toId: DocumentId, +): boolean { + if (fromId === toId) return byteStore.has(toId); + const record = byteStore.get(fromId); + if (!record) return byteStore.has(toId); + byteStore.put(toId, record.bytes); + byteStore.delete(fromId); + return true; +} + +/** + * After ingest, either register a new document or attach bytes to an existing + * one with the same fingerprint (evidence stays on the existing id). + */ +export function registerOrRematchDocument( + engine: Engine, + byteStore: PdfByteStore, + input: { + readonly document: Document; + readonly representation: DocumentRepresentation; + }, +): RegisterOrRematchResult { + const fingerprint = input.document.fingerprint; + if (fingerprint) { + const matches = findDocumentsByFingerprint(engine, fingerprint); + if (matches.length > 0) { + const primary = pickPrimaryDocument(engine, byteStore, matches); + rekeyByteStore(byteStore, input.document.id, primary.id); + + // Refresh title/uri on the primary when useful (viewer prefers byte store). + const live = byteStore.get(primary.id); + if (live || input.document.title) { + engine.repos.documents.update({ + ...primary, + ...(input.document.title !== undefined + ? { title: input.document.title } + : {}), + ...(live ? { uri: live.blobUrl } : {}), + updatedAt: new Date().toISOString(), + }); + } + + // Never register the discarded twin. + return { + kind: "rematched", + documentId: primary.id, + fingerprint, + discardedDocumentId: input.document.id, + }; + } + } + + engine.documents.register({ + document: input.document, + representation: input.representation, + }); + return { + kind: "registered", + documentId: input.document.id, + fingerprint, + }; +} + +/** + * Collapse documents that share a fingerprint into one primary record. + * Migrates annotations onto the primary and copies PDF bytes when present. + */ +export function mergeDuplicateDocumentsByFingerprint( + engine: Engine, + byteStore: PdfByteStore, +): { + readonly groupsMerged: number; + readonly documentsRemoved: number; + /** secondary documentId → primary documentId */ + readonly remappedIds: ReadonlyMap; +} { + const byFp = new Map(); + for (const doc of engine.documents.list()) { + if (!doc.fingerprint) continue; + const list = byFp.get(doc.fingerprint) ?? []; + list.push(doc); + byFp.set(doc.fingerprint, list); + } + + let groupsMerged = 0; + let documentsRemoved = 0; + const remappedIds = new Map(); + + for (const [, group] of byFp) { + if (group.length < 2) continue; + groupsMerged += 1; + const primary = pickPrimaryDocument(engine, byteStore, group); + const primaryRep = + engine.documents.listRepresentations(primary.id)[0] ?? null; + + for (const secondary of group) { + if (secondary.id === primary.id) continue; + + rekeyByteStore(byteStore, secondary.id, primary.id); + remappedIds.set(secondary.id, primary.id); + + for (const ann of engine.annotations.listByDocument(secondary.id)) { + engine.repos.annotations.update({ + ...ann, + documentId: primary.id, + ...(primaryRep + ? { representationId: primaryRep.id } + : ann.representationId !== undefined + ? { representationId: ann.representationId } + : {}), + updatedAt: new Date().toISOString(), + }); + } + + // Drop secondary record (and its representations). Evidence items stay: + // they key off annotation ids, which now point at primary. + engine.documents.remove(secondary.id); + documentsRemoved += 1; + } + + const live = byteStore.get(primary.id); + if (live) { + engine.repos.documents.update({ + ...engine.documents.get(primary.id)!, + uri: live.blobUrl, + updatedAt: new Date().toISOString(), + }); + } + } + + return { groupsMerged, documentsRemoved, remappedIds }; +} diff --git a/src/app/sessions/importSessionZip.ts b/src/app/sessions/importSessionZip.ts index bd7e31c..06e46e1 100644 --- a/src/app/sessions/importSessionZip.ts +++ b/src/app/sessions/importSessionZip.ts @@ -154,10 +154,29 @@ export async function importSessionZip( const targetByteStore = services.getOrCreateByteStore(targetSessionId); // 5. Build the document remap. + // Prefer fingerprint targets that already hold annotations/evidence so a + // re-import reattaches to the "work" document, not an empty twin. const docRemap = new Map(); const existingByFingerprint = new Map(); + const fingerprintCandidates = new Map(); for (const doc of targetEngine.documents.list()) { - if (doc.fingerprint) existingByFingerprint.set(doc.fingerprint, doc.id); + if (!doc.fingerprint) continue; + const list = fingerprintCandidates.get(doc.fingerprint) ?? []; + list.push(doc); + fingerprintCandidates.set(doc.fingerprint, list); + } + for (const [fp, candidates] of fingerprintCandidates) { + const scored = candidates.map((doc) => ({ + id: doc.id, + evidence: targetEngine.evidence.listByDocument(doc.id).length, + annotations: targetEngine.annotations.listByDocument(doc.id).length, + })); + scored.sort((a, b) => { + if (b.evidence !== a.evidence) return b.evidence - a.evidence; + if (b.annotations !== a.annotations) return b.annotations - a.annotations; + return 0; + }); + existingByFingerprint.set(fp, scored[0]!.id); } let documentsAdded = 0;