From b0ef516b1841efc5f0e41e4a20aef6cba4cd6ba3 Mon Sep 17 00:00:00 2001 From: mrsibe Date: Fri, 25 Sep 2026 13:39:43 +0800 Subject: [PATCH] test(parsers): pin loader structure with golden-file tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `test/fixtures/` shipped sample.pdf/docx/html but no test under `test/` ever exercised a loader, so nothing caught a loader dropping `DocumentLoadResult.structure` — which the provenance work in #66/#67 is about to start trusting. Add `test/parsers.test.ts`, which parses the fixtures and asserts the shape of the result (page counts, page boundaries, heading outlines, offset arithmetic) rather than the extracted prose, so minor extraction differences still pass while lost structure fails. Fixtures: - sample.pdf / sample.html / sample.docx: reused as-is. - multipage.pdf: new, three pages, so page boundaries are exercised across pages rather than on a single page. - headings.docx: new, minimal OOXML with Heading1/Heading2 styles; sample.docx is heading-less and only proves structure is still emitted. - sample.md: new, one h1 and two h2. `npm test` now passes `--experimental-transform-types`: Node's strip-only type support rejects the enums on every loader's import path (`src/shared/utils/logger.ts`), and transforming lets the tests import the real loaders instead of a test-only copy. The rationale is recorded under package.json's `//test` key. Refs #64 --- package.json | 9 +- test/fixtures/headings.docx | Bin 0 -> 1885 bytes test/fixtures/multipage.pdf | Bin 0 -> 1124 bytes test/fixtures/sample.md | 11 ++ test/parsers.test.ts | 258 ++++++++++++++++++++++++++++++++++++ 5 files changed, 277 insertions(+), 1 deletion(-) create mode 100644 test/fixtures/headings.docx create mode 100644 test/fixtures/multipage.pdf create mode 100644 test/fixtures/sample.md create mode 100644 test/parsers.test.ts diff --git a/package.json b/package.json index 6fbd153..4e3826e 100644 --- a/package.json +++ b/package.json @@ -21,7 +21,7 @@ "typecheck:web": "tsc --noEmit -p tsconfig.web.json --composite false", "typecheck:test": "tsc -p tsconfig.test.json", "typecheck": "npm run typecheck:node && npm run typecheck:web && npm run typecheck:test", - "test": "node --import ./test/ts-resolve.mjs --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --test \"test/**/*.test.ts\"", + "test": "node --experimental-transform-types --disable-warning=ExperimentalWarning --import ./test/ts-resolve.mjs --disable-warning=MODULE_TYPELESS_PACKAGE_JSON --test \"test/**/*.test.ts\"", "start": "electron-vite preview", "dev": "electron-vite dev", "build": "npm run typecheck && electron-vite build", @@ -36,6 +36,13 @@ "db:push": "drizzle-kit push", "db:studio": "drizzle-kit studio" }, + "//test": [ + "`node --test` strips TypeScript types rather than compiling them, and strip-only", + "mode rejects the enums the production sources use (src/shared/utils/logger.ts is", + "on every loader's import path). `--experimental-transform-types` lets those", + "modules load unmodified, so parser tests can import the real loaders instead of", + "a test-only copy." + ], "//dependencies": [ "ONLY packages that cannot be bundled belong here.", "electron-vite defaults build.externalizeDeps to true, so every `dependencies`", diff --git a/test/fixtures/headings.docx b/test/fixtures/headings.docx new file mode 100644 index 0000000000000000000000000000000000000000..0d8bd685d3c6d27d3e590fa2de4a247ea5a424db GIT binary patch literal 1885 zcmWIWW@Zs#U|`^2SdwEI`(ONy{cj-8n2CWw7)VDu=jWBA=9R>UR2HNb$Ldw&=B%B# z+wZV}NZWV2)~8#&WF!Qfe60&48yDIfm{>b0;;XM+Ugo`dli92mFmW&b8^169TUPNG zKN*9`y$h#?mnyTe$@cQrYTw`He|4=$h=N3cVS={1%Famru-k@;oP6wAPXoMm1$TXE zkXzBm$*diFMvhCqc)idI*)J>B&3@+gQi7{3XzEJmqMOWR8_Hd6*nO}c7>mKDrs~3O7$7H znJd1}Tm1FCt@yNK>o!ZB|G854>57RR@ppIsT2rwh@#k{^t;Np2{!13WpSj*IeqUS- zLjW+WxPYO;01hiK14^(0>GJ%d6n)%ExM51-i&Arn!3q)K{&IP#_Bx;fIYtHs8B~QS z`N^fZz$gHPy&gzw@A-{fhYSSR9_;lO%-fzR>#`(MTyU3HZNr4P%L%huVnlv_V~Yuq zk*V+dUH5NUwa&D=hLb+}ImCt?xS(k-wVXAk?Q$A_&!m#YTkb7v5arypE#q89>M47Z z*$z6&eJp1*TbZ9PNsCdLYPaC@^pnS3m&q)C=D2G0l@r@Tw?6)Ow?Ti(*ToGt`b{?^ z^8Vm|xBR4rO;u_1g{AVcb1r3lW-vRk!jQA#sgd=8V*CFQSH=860fO*n&iiFv2Z8>Z z4fH2J%%8<2l{u-!Apb4xJD7FYfT!(!E!X1TuT)x%oaWp)RTVGqpb{mtAvJmCvdZ5( zw=4~Kk@!aKf6#!`iZytEV2|95kij(d7}Hx(Rxb#7_3%I(=I)eZ|wV%TQ({(rvXuk!J3 zL#3Y6GMftG70g$?TQ&gYbb+Y_lzkv-pqXGVUz33V z!-KiYGxDZ4H5;}aO7e-|wR5;Ew{+saC2Q^X?3I<7VqtObetEg?cN62C$sM;M*-9fB z9ku6N2{YWmv;3@_htC}^JL_&9v7QTCU5k&+Ty7(*{#N;9M=#H^DSsL-X{edlOgp#u z?9_{_hlBZ=7tQO|4)*|Ng2fLnH;8&womBX;iOony?*GBN7Fmt%>UnIrzcwacotP>o zqQ=HAt}^v~$;`+FKV^P>O^MzEEIJsOv&4zw%+*@;~XC^!*X{9#&Pv4HAW^wJ5W5#rbiRE@}K8yrpO-bT;=2$K?k zc?9fgL}`Go89h58H2X0_H6uF!Ux?^*oe}^5 literal 0 HcmV?d00001 diff --git a/test/fixtures/multipage.pdf b/test/fixtures/multipage.pdf new file mode 100644 index 0000000000000000000000000000000000000000..f3d6431381ef38630a59a4c7387b06f2aad4befe GIT binary patch literal 1124 zcmchXO;5r=5Qgvl6>}lcgSOjGnh*{sjfolv?2UL>=l~_yHSMD4uXm<^RGRp4p@&U7 z`|i8*%xt$cyiKmewQmVKF0DGtwSb|G}5p=s2<|zyL`}=~O z&c4sm5+*?KOuU9kqDf+pH&at3!Knzad#c3U%pI;@(PT4KbMb2~ z5122^5`3TMOH-hFwt5omM1bpqoI~V7P<1D0&-<5hU!I79d-q!M7cD%A-58krC#Q+ zw)?-?92-7z@biBqm&H)oCnl#@v{Gw#K5mBgWYW c5*r join(FIXTURES, name) + +/** The committed outline each fixture is expected to produce. */ +const DOCX_OUTLINE = [ + { level: 1, title: 'Overview' }, + { level: 2, title: 'Details' }, + { level: 1, title: 'Summary' } +] +const MARKDOWN_OUTLINE = [ + { level: 1, title: 'Sample Markdown' }, + { level: 2, title: 'Section One' }, + { level: 2, title: 'Section Two' } +] + +/** `structure` is optional on the result, so every structural test has to ask for it. */ +function requirePages(result: DocumentLoadResult) { + assert.ok(result.structure, 'the loader must populate `structure`') + assert.equal(result.structure.type, 'pages') + assert.ok(result.structure.pages, 'a `pages` structure must carry `pages`') + return result.structure.pages +} + +function requireSections(result: DocumentLoadResult) { + assert.ok(result.structure, 'the loader must populate `structure`') + assert.equal(result.structure.type, 'sections') + assert.ok(result.structure.sections, 'a `sections` structure must carry `sections`') + return result.structure.sections +} + +function flattenSections(sections: SectionInfo[]): SectionInfo[] { + const flat: SectionInfo[] = [] + const walk = (nodes: SectionInfo[]): void => { + for (const node of nodes) { + flat.push(node) + if (node.children?.length) walk(node.children) + } + } + walk(sections) + return flat +} + +function outline(sections: SectionInfo[]): { level: number; title: string }[] { + return flattenSections(sections).map(({ level, title }) => ({ level, title })) +} + +/** + * Siblings must not overlap, children must stay inside their parent, and no + * range may be inverted. This is the invariant #66/#67 are most likely to break + * while they rework how offsets are computed. + */ +function assertSectionOffsets(nodes: SectionInfo[], label: string, parent?: SectionInfo): void { + nodes.forEach((node, index) => { + assert.ok(node.endOffset >= node.startOffset, `${label}: ${node.title} has an inverted range`) + if (parent) { + assert.ok( + node.startOffset >= parent.startOffset && node.endOffset <= parent.endOffset, + `${label}: ${node.title} escapes its parent ${parent.title}` + ) + } + if (index > 0) { + assert.ok( + node.startOffset >= nodes[index - 1].endOffset, + `${label}: ${node.title} overlaps ${nodes[index - 1].title}` + ) + } + if (node.children?.length) assertSectionOffsets(node.children, label, node) + }) +} + +// --- PDF --------------------------------------------------------------------- + +test('PDF: pages are numbered in order and their offsets slice back to the page text', async () => { + const result = await new PdfLoader().loadFromPath(fixture('multipage.pdf')) + assert.equal(result.mimeType, 'application/pdf') + assert.equal(result.metadata?.pageCount, 3) + + const pages = requirePages(result) + assert.equal(pages.length, 3) + assert.deepEqual( + pages.map((page) => page.pageNumber), + [1, 2, 3] + ) + + for (const page of pages) { + assert.equal( + result.content.slice(page.startOffset, page.endOffset), + page.content, + `page ${page.pageNumber} offsets do not slice back to its text` + ) + } + + for (let i = 1; i < pages.length; i++) { + assert.ok(pages[i].startOffset >= pages[i - 1].endOffset, 'page boundaries overlap') + assert.ok(pages[i].startOffset >= pages[i - 1].startOffset, 'page offsets are not monotonic') + } + assert.ok( + pages[pages.length - 1].endOffset <= result.content.length, + 'the last page boundary runs past the document' + ) +}) + +test('PDF: the existing single-page fixture keeps its text and page box', async () => { + const result = await new PdfLoader().loadFromPath(fixture('sample.pdf')) + assert.equal(result.metadata?.pageCount, 1) + + const pages = requirePages(result) + assert.equal(pages.length, 1) + assert.match(pages[0].content, /KnowNote Import Test/) + assert.equal(result.content.slice(pages[0].startOffset, pages[0].endOffset), pages[0].content) + assert.equal(pages[0].metadata?.width, 612) + assert.equal(pages[0].metadata?.height, 792) +}) + +// --- DOCX -------------------------------------------------------------------- + +test('DOCX: heading levels and their nesting survive the round trip', async () => { + const result = await new DocxLoader().loadFromPath(fixture('headings.docx')) + assert.equal(result.title, 'Overview') + assert.equal( + result.mimeType, + 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' + ) + + const sections = requireSections(result) + assert.deepEqual(outline(sections), DOCX_OUTLINE) + assert.equal(sections.length, 2, 'Overview and Summary are the roots') + assert.equal(sections[0].children?.length, 1, 'Details nests under Overview') + assertSectionOffsets(sections, 'DOCX') + + for (const section of flattenSections(sections)) { + assert.ok( + result.content.slice(section.startOffset).startsWith(section.title), + `DOCX: ${section.title} startOffset does not point at its title` + ) + } +}) + +test('DOCX: a heading-less document still reports an (empty) section structure', async () => { + const result = await new DocxLoader().loadFromPath(fixture('sample.docx')) + assert.match(result.content, /KnowNote Import Test/) + assert.deepEqual(requireSections(result), []) +}) + +// --- Markdown ---------------------------------------------------------------- + +test('Markdown: ATX headings become a nested outline with consistent offsets', async () => { + const result = await new MarkdownLoader().loadFromPath(fixture('sample.md')) + assert.equal(result.title, 'Sample Markdown') + assert.equal(result.mimeType, 'text/markdown') + + const sections = requireSections(result) + assert.deepEqual(outline(sections), MARKDOWN_OUTLINE) + assertSectionOffsets(sections, 'Markdown') + + // Heading offsets are positions in the raw fixture text while `content` is + // trimmed, so the root section can end a newline past `result.content`; the + // assertions stay relative to the headings themselves. + for (const section of flattenSections(sections)) { + const marker = `${'#'.repeat(section.level)} ${section.title}` + assert.ok( + result.content.slice(section.startOffset).startsWith(marker), + `Markdown: ${section.title} startOffset does not point at its ATX marker` + ) + } +}) + +// --- HTML -------------------------------------------------------------------- + +test('HTML: Readability extracts the title and body and still carries a structure', async () => { + const html = await readFile(fixture('sample.html')) + const result = await new WebLoader().loadFromBuffer(html) + assert.equal(result.title, 'Sample Document') + assert.equal(result.mimeType, 'text/html') + assert.equal(result.metadata?.excerpt, 'Fixture for the packaged-app smoke test') + assert.match(result.content, /KnowNote Import Test/) + assert.match(result.content, /Hello World/) + + // Readability hoists the document's lone

into `title` and drops it from + // the body, so this fixture has no heading left to become a section. The test + // still guards the presence of the structure object itself. + assert.deepEqual(requireSections(result), []) +}) + +// --- structure opt-out ------------------------------------------------------- + +test('preserveStructure: false drops the structure without dropping the text', async () => { + const buffer = await readFile(fixture('multipage.pdf')) + const result = await new PdfLoader().loadFromBuffer(buffer, { preserveStructure: false }) + assert.equal(result.structure, undefined) + assert.match(result.content, /Page One Text/) +}) + +// --- dispatch ---------------------------------------------------------------- + +test('FileParserService routes each extension and MIME type to the matching loader', async () => { + const parser = new FileParserService() + + const pdf = await parser.parseFile(fixture('sample.pdf')) + assert.equal(pdf.mimeType, 'application/pdf') + assert.equal(pdf.structure?.type, 'pages') + + const docx = await parser.parseFile(fixture('sample.docx')) + assert.equal( + docx.mimeType, + 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' + ) + assert.equal(docx.structure?.type, 'sections') + + const markdown = await parser.parseBuffer(await readFile(fixture('sample.md')), 'md') + assert.equal(markdown.mimeType, 'text/markdown') + assert.equal(markdown.structure?.type, 'sections') + + for (const supported of ['pdf', 'docx', 'md', 'html', 'htm', 'pptx']) { + assert.ok(parser.isSupported(supported), `${supported} should be routable`) + } + assert.equal(parser.isSupported('exe'), false) + + assert.equal(parser.getFileTypeFromMime('application/pdf'), 'pdf') + assert.equal(parser.getFileTypeFromMime('text/markdown'), 'md') + assert.equal(parser.getMimeType('md'), 'text/markdown') + assert.equal( + parser.getMimeType('docx'), + 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' + ) +})