Implement CE-WP-0002 T03-T09: ingest, anchor resolution, engine, UI, persistence, e2e
Completes the PDF review slice end-to-end. After this commit a user can
open a fixture, select text, save an evidence item with commentary, see
it in the sidebar, reload the page, click the item, and the viewer
scrolls to the passage.
- T03 src/source/pdf/{fingerprint,extract,ingest}.ts + 39 fixture tests
- SHA-256 fingerprint over a fresh ArrayBuffer (TS BufferSource-safe)
- PDF.js text extract; per-page normalize then join with "\n\n"
- PageMap + OffsetMap (gap-free coverage); pageLength = end - start
- Updated manifest's Betriebskosten quote to one PDF.js extracts cleanly
- T04 src/anchor/selectors/{create,resolve}.ts + 25 unit + 7 fixture tests
- createSelectors emits the maximal redundant set (TextQuote +
TextPosition + PdfRect + PdfPageText when available)
- resolveSelectors implements the SharedContracts §7 ladder; confidence
1.0 (pos+quote) → 0.7 (rect-only) → 0 (unresolved)
- Cross-module integration test moved to tests/integration/ to honor
the anchor↛source boundary lint rule
- T05 engine: sync event bus over the closed §4 vocabulary, Map-backed
repos, services, createEngine() composition root, 12 tests
- T06 work + app: three-pane shell (CollectionList | ViewerShell |
EvidenceSidebar) wired through EngineProvider; EngineContext lives in
src/work/ to respect the work↛app boundary; SpikeApp deleted
- T07 AnnotationToolbar: pendingSelection in context; Save runs
createSelectors → engine.annotations.create → engine.evidence.create
- T08 click-to-reopen + localStorage persistence
- scrollToAnnotation state in context with a version counter so a
second click on the same item re-fires the viewer scroll
- captureSnapshot/restoreSnapshot/attachPersister/restoreFromStorage;
restore bypasses services to avoid event-loops
- active-document id persisted alongside the snapshot so reload lands
on the same fixture; ADR-0005 written
- 9 persistence tests
- T09 tests/integration/app-prd-scenario.dom.test.tsx
- end-to-end happy-dom test of PRD scenario steps 1-8 through the real
React tree; viewer + ingest mocked per ADR-0004's headless-Chromium
limitation. Fixed memo-deps bug in EvidenceSidebar/ViewerShell where
useEngineEventTick values were not included in the useMemo deps,
leaving stale memoization across event-driven re-renders
- vitest.config.ts: happy-dom for *.dom.test.{ts,tsx} files
- noEmit added to tsconfig so tsc -b doesn't litter src/ with .js outputs
Gates: typecheck ✓ lint ✓ test 109/109 across 11 files ✓ build ✓
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
parent
2a7b05c190
commit
d54daf2e61
45 changed files with 3655 additions and 277 deletions
122
src/source/pdf/extract.ts
Normal file
122
src/source/pdf/extract.ts
Normal file
|
|
@ -0,0 +1,122 @@
|
|||
/**
|
||||
* PDF text extraction → canonical text + PageMap + OffsetMap.
|
||||
*
|
||||
* Implements `wiki/ArchitectureOverview.md` §3.4 ("extract canonical text /
|
||||
* build format-specific maps") for the `pdf-text` representation
|
||||
* (`wiki/SharedContracts.md` §1, §3) and §6 (canonical normalization).
|
||||
*
|
||||
* Runtime independence: the PDF.js worker must be configured by the host
|
||||
* application (`GlobalWorkerOptions.workerSrc`) before this module is
|
||||
* called. In Vite/browser code the worker is bundled via the viewer; in
|
||||
* Node tests the test setup file points it at
|
||||
* `pdfjs-dist/legacy/build/pdf.worker.mjs`. No worker setup happens here
|
||||
* so the same module loads cleanly in both runtimes.
|
||||
*
|
||||
* Page boundary semantics: canonical text concatenates per-page normalized
|
||||
* text with a single "\n\n" paragraph separator. The separator is treated
|
||||
* as belonging to the *preceding* page in `OffsetMap`, so the map covers
|
||||
* `[0, canonicalText.length)` with no gaps. The last page has no trailing
|
||||
* separator. This means `pageLength = globalEnd - globalStart` for
|
||||
* every page; for non-last pages it equals (normalized page text length +
|
||||
* 2). See `PageOffsetRange` in `@shared/document.ts`.
|
||||
*/
|
||||
|
||||
import { getDocument } from "pdfjs-dist";
|
||||
import type { PDFPageProxy } from "pdfjs-dist";
|
||||
import type {
|
||||
OffsetMap,
|
||||
PageInfo,
|
||||
PageMap,
|
||||
PageOffsetRange,
|
||||
} from "@shared/document";
|
||||
import { normalize } from "@shared/text/normalize";
|
||||
|
||||
const PAGE_SEPARATOR = "\n\n";
|
||||
|
||||
export interface PdfExtractionResult {
|
||||
readonly canonicalText: string;
|
||||
readonly pageMap: PageMap;
|
||||
readonly offsetMap: OffsetMap;
|
||||
readonly pageCount: number;
|
||||
}
|
||||
|
||||
export async function extractPdf(bytes: Uint8Array): Promise<PdfExtractionResult> {
|
||||
// PDF.js mutates the bytes buffer (transfers ownership). Pass a fresh copy
|
||||
// so the caller's Uint8Array stays usable for fingerprinting after extract.
|
||||
const data = new Uint8Array(bytes);
|
||||
const loadingTask = getDocument({ data });
|
||||
const doc = await loadingTask.promise;
|
||||
|
||||
try {
|
||||
const pageCount = doc.numPages;
|
||||
const pageInfos: PageInfo[] = [];
|
||||
const pageNormalizedTexts: string[] = [];
|
||||
|
||||
for (let pageNumber = 1; pageNumber <= pageCount; pageNumber++) {
|
||||
const page = await doc.getPage(pageNumber);
|
||||
try {
|
||||
const viewport = page.getViewport({ scale: 1 });
|
||||
pageInfos.push({
|
||||
page: pageNumber,
|
||||
width: viewport.width,
|
||||
height: viewport.height,
|
||||
});
|
||||
|
||||
const rawText = await extractPageText(page);
|
||||
pageNormalizedTexts.push(normalize(rawText).text);
|
||||
} finally {
|
||||
page.cleanup();
|
||||
}
|
||||
}
|
||||
|
||||
const { canonicalText, offsetMap } = buildOffsetMap(pageNormalizedTexts);
|
||||
|
||||
return {
|
||||
canonicalText,
|
||||
pageMap: pageInfos,
|
||||
offsetMap,
|
||||
pageCount,
|
||||
};
|
||||
} finally {
|
||||
await doc.destroy();
|
||||
}
|
||||
}
|
||||
|
||||
async function extractPageText(page: PDFPageProxy): Promise<string> {
|
||||
const content = await page.getTextContent();
|
||||
// textContent.items are TextItem | TextMarkedContent. We want only the
|
||||
// TextItem strings (those have a `str` field); marked-content entries are
|
||||
// structural anchors and have no visible text.
|
||||
const parts: string[] = [];
|
||||
for (const item of content.items) {
|
||||
if ("str" in item) {
|
||||
parts.push(item.str);
|
||||
if (item.hasEOL) parts.push("\n");
|
||||
}
|
||||
}
|
||||
return parts.join("");
|
||||
}
|
||||
|
||||
function buildOffsetMap(pageTexts: readonly string[]): {
|
||||
canonicalText: string;
|
||||
offsetMap: OffsetMap;
|
||||
} {
|
||||
const ranges: PageOffsetRange[] = [];
|
||||
let offset = 0;
|
||||
for (let i = 0; i < pageTexts.length; i++) {
|
||||
const text = pageTexts[i]!;
|
||||
const isLast = i === pageTexts.length - 1;
|
||||
const segmentLength = text.length + (isLast ? 0 : PAGE_SEPARATOR.length);
|
||||
const globalStart = offset;
|
||||
const globalEnd = offset + segmentLength;
|
||||
ranges.push({
|
||||
page: i + 1,
|
||||
globalStart,
|
||||
globalEnd,
|
||||
pageLength: segmentLength,
|
||||
});
|
||||
offset = globalEnd;
|
||||
}
|
||||
const canonicalText = pageTexts.join(PAGE_SEPARATOR);
|
||||
return { canonicalText, offsetMap: ranges };
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue