From 25cac3014d5f12380aad081add0c1910354e9a08 Mon Sep 17 00:00:00 2001 From: Asiones Jia Date: Wed, 26 Aug 2026 13:14:25 +0800 Subject: [PATCH 1/2] fix(pdf): keep footnotes out of body reading order Raised markers after punctuation stay with the word. Numbered note blocks are peeled from the page before XY-cut and table detection so they do not interleave with columns or glue onto the last paragraph. fixes #74 --- packages/pdf/src/layout.ts | 179 ++++++++++++++++++++++++ packages/pdf/src/pdf.ts | 148 ++++++++++++++------ packages/pdf/test/layout.test.ts | 125 +++++++++++++++++ packages/pdf/test/reading-order.test.ts | 76 +++++++++- 4 files changed, 484 insertions(+), 44 deletions(-) diff --git a/packages/pdf/src/layout.ts b/packages/pdf/src/layout.ts index 5227c03..d72b586 100644 --- a/packages/pdf/src/layout.ts +++ b/packages/pdf/src/layout.ts @@ -580,3 +580,182 @@ function partitionX(boxes: T[], splitX: number): [T[], T[]] if (left.length + right.length !== boxes.length) return undefined; return [left, right]; } + +const SUP_DIGITS = '⁰¹²³⁴⁵⁶⁷⁸⁹'; + +export function peelFootnoteLines( + lines: T[][], +): { body: T[][]; notes: T[][]; footer: T[][]; drop: T[][] } { + if (lines.length < 2) return { body: lines, notes: [], footer: [], drop: [] }; + const body: T[][] = []; + const notes: T[][] = []; + const footer: T[][] = []; + const drop: T[][] = []; + const byPage = new Map(); + for (const line of lines) { + const page = line[0]!.page; + const list = byPage.get(page) ?? []; + list.push(line); + byPage.set(page, list); + } + for (const page of [...byPage.keys()].sort((a, b) => a - b)) { + const split = splitPageFootnotes(byPage.get(page)!); + body.push(...split.body); + notes.push(...split.notes); + footer.push(...split.footer); + drop.push(...split.drop); + } + return { body, notes, footer, drop }; +} + +function splitPageFootnotes( + lines: T[][], +): { body: T[][]; notes: T[][]; footer: T[][]; drop: T[][] } { + if (lines.length < 2) return { body: lines, notes: [], footer: [], drop: [] }; + const bodyFont = bodyFontSize(lines); + const ys = lines.map((line) => line[0]!.y); + const pageTop = Math.max(...ys); + const pageBottom = Math.min(...ys); + const height = Math.max(pageTop - pageBottom, 1); + const floor = pageBottom + height * 0.5; + const bodyLines = lines.filter((line) => line[0]!.y > floor); + const markers = new Set([ + ...collectMarkers(bodyLines, bodyFont), + ...collectTrailingMarkers(lines), + ]); + + const starts: T[][] = []; + for (const line of lines) { + if (line[0]!.y > floor) continue; + const text = linePlain(line); + const n = noteStartNumber(text); + if (n !== undefined) { + if (markers.has(n) || isSmallNoteLine(line, bodyFont)) starts.push(line); + continue; + } + if (isOrphanMarkerLine(text, markers)) starts.push(line); + } + if (starts.length === 0) return { body: lines, notes: [], footer: [], drop: [] }; + + const regionTop = Math.max(...starts.map((line) => line[0]!.y)); + const noteSet = new Set(starts); + for (const line of lines) { + if (noteSet.has(line)) continue; + if (line[0]!.y > regionTop + 6) continue; + const text = linePlain(line); + if (isFooterText(text)) continue; + const fs = Math.max(...line.map((it) => it.fontSize)); + if (fs > bodyFont * 0.98 && noteStartNumber(text) === undefined) continue; + noteSet.add(line); + } + if (noteSet.size > lines.length * 0.55) return { body: lines, notes: [], footer: [], drop: [] }; + + const notes = lines.filter((line) => noteSet.has(line)); + const noteNums = new Set(); + for (const line of notes) { + const n = noteStartNumber(linePlain(line)); + if (n !== undefined) noteNums.add(n); + } + const rest = lines.filter((line) => !noteSet.has(line)); + const noteBottom = Math.min(...notes.map((line) => line[0]!.y)); + const body: T[][] = []; + const footer: T[][] = []; + const drop: T[][] = []; + for (const line of rest) { + const text = linePlain(line); + if (isOrphanMarkerLine(text, noteNums) || isOrphanMarkerLine(text, markers)) { + drop.push(line); + continue; + } + if (line[0]!.y < noteBottom && isFooterText(text)) footer.push(line); + else body.push(line); + } + return { body, notes, footer, drop }; +} + +function isOrphanMarkerLine(text: string, noteNums: Set): boolean { + const ascii = text.replace(/[⁰¹²³⁴⁵⁶⁷⁸⁹]/g, (c) => String(SUP_DIGITS.indexOf(c))); + const parts = ascii.split(/\s+/).filter(Boolean); + if (parts.length === 0 || parts.length > 3) return false; + return parts.every((p) => /^\d{1,3}$/.test(p) && noteNums.has(p)); +} + +function bodyFontSize(lines: T[][]): number { + const counts = new Map(); + for (const line of lines) { + const fs = Math.round(Math.max(...line.map((it) => it.fontSize)) * 2) / 2; + if (fs < 9) continue; + counts.set(fs, (counts.get(fs) ?? 0) + 1); + } + let best = 12; + let bestN = -1; + for (const [fs, n] of counts) { + if (n > bestN || (n === bestN && fs > best)) { + best = fs; + bestN = n; + } + } + return best; +} + +function collectTrailingMarkers(lines: T[][]): Set { + const nums = new Set(); + const glued = /(?<=[\p{L}.!?”"'”'])(\d{1,3})(?!\d)/gu; + for (const line of lines) { + const text = linePlain(line); + for (const m of text.matchAll(glued)) nums.add(m[1]!); + } + return nums; +} + +function collectMarkers(lines: T[][], bodyFont: number): Set { + const nums = new Set(); + const re = /[⁰¹²³⁴⁵⁶⁷⁸⁹]+/g; + for (const line of lines) { + const text = linePlain(line); + for (const m of text.matchAll(re)) { + nums.add([...m[0]!].map((c) => String(SUP_DIGITS.indexOf(c))).join('')); + } + for (const it of line) { + const t = it.text.trim(); + if (t.length === 0 || t.length > 4) continue; + if (![...t].every((c) => c >= '0' && c <= '9')) continue; + if (it.fontSize <= 0 || it.fontSize > bodyFont * 0.85) continue; + nums.add(t); + } + } + return nums; +} + +function isSmallNoteLine(line: T[], bodyFont: number): boolean { + const first = line[0]!; + const mark = first.text.trim().replace(/[.)]$/, ''); + const firstIsMark = + mark.length > 0 && mark.length <= 3 && [...mark].every((c) => c >= '0' && c <= '9'); + if (firstIsMark && first.fontSize <= bodyFont * 0.85) return true; + const fs = Math.max(...line.map((it) => it.fontSize)); + return fs <= bodyFont * 0.92; +} + +function noteStartNumber(text: string): string | undefined { + const ascii = text.replace(/[⁰¹²³⁴⁵⁶⁷⁸⁹]/g, (c) => String(SUP_DIGITS.indexOf(c))); + const m = ascii.match(/^(\d{1,3})(?:[.)]\s+|\s+|(?=[A-Z]))/); + if (!m) return undefined; + const after = ascii.slice(m[0].length).trim(); + if (after.length < 6) return undefined; + if (/^(figure|table|fig\.?)\b/i.test(ascii)) return undefined; + return m[1]; +} + +function isFooterText(text: string): boolean { + if (/^\d{1,4}$/.test(text)) return true; + return /^page\s+\d+$/i.test(text); +} + +function linePlain(line: T[]): string { + return line + .map((it) => it.text) + .join('') + .replace(/\s+/g, ' ') + .trim(); +} diff --git a/packages/pdf/src/pdf.ts b/packages/pdf/src/pdf.ts index 40688db..0487c16 100644 --- a/packages/pdf/src/pdf.ts +++ b/packages/pdf/src/pdf.ts @@ -17,7 +17,13 @@ import { import { applyDifferences, applyNamedEncoding } from './encodings.js'; import { decryptBytes, type EncryptParams, type FileCrypt, openEncrypt } from './encrypt.js'; import { xObjectToImage } from './images.js'; -import { groupIntoLines, type LayoutBox, lineBox, orderBoxes } from './layout.js'; +import { + groupIntoLines, + type LayoutBox, + lineBox, + orderBoxes, + peelFootnoteLines, +} from './layout.js'; import { detectTables } from './tables.js'; import { cmapFromTrueType } from './truetype.js'; import { looksLikeXfdf, xfdfToMarkdown } from './xfdf.js'; @@ -2924,46 +2930,66 @@ function dedupeOverlappingItems(items: TextItem[]): TextItem[] { const SUP = ['⁰', '¹', '²', '³', '⁴', '⁵', '⁶', '⁷', '⁸', '⁹']; const SUB = ['₀', '₁', '₂', '₃', '₄', '₅', '₆', '₇', '₈', '₉']; +function scriptDigits(text: string): string | undefined { + const t = text.trim(); + if (t.length === 0 || t.length > 4) return undefined; + let out = ''; + for (const c of t) { + if (c >= '0' && c <= '9') { + out += c; + continue; + } + const i = SUP.indexOf(c); + if (i < 0) return undefined; + out += String(i); + } + return out; +} + +function canAttachScript(parent: TextItem, item: TextItem): boolean { + if (parent.page !== item.page) return false; + if (item.fontSize <= 0) return false; + if (scriptDigits(item.text) === undefined) return false; + if (parent.fontSize < item.fontSize * 1.15) return false; + const last = [...parent.text.trimEnd()].at(-1); + if (last === undefined) return false; + if (!/[\p{L}\p{N}\p{P}]/u.test(last)) return false; + const fs = Math.max(parent.fontSize, 8); + const gap = item.x - (parent.x + parent.width); + if (gap >= fs * 0.35 || gap <= -fs * 0.5) return false; + if (Math.abs(item.y - parent.y) > Math.max(fs * 0.8, 8)) return false; + return true; +} + function mergeScriptItems(items: TextItem[]): TextItem[] { if (items.length < 2) return items; - const groups: TextItem[][] = []; + const byPage = new Map(); for (const item of items) { - const found = groups.find((g) => g[0]!.page === item.page && Math.abs(g[0]!.y - item.y) < 5); - if (found) found.push(item); - else groups.push([item]); + const list = byPage.get(item.page) ?? []; + list.push(item); + byPage.set(item.page, list); } const result: TextItem[] = []; - for (const group of groups) { - group.sort((a, b) => a.x - b.x); - const maxFs = group.reduce((m, it) => Math.max(m, it.fontSize), 0); - if (maxFs < 1) { - result.push(...group); - continue; - } - const subTh = maxFs * 0.75; + for (const pageItems of byPage.values()) { + pageItems.sort((a, b) => a.x - b.x || b.y - a.y); const merged: TextItem[] = []; - for (const item of group) { - const parent = merged[merged.length - 1]; - if ( - parent && - item.fontSize < subTh && - item.fontSize > 0 && - item.text.length <= 4 && - [...item.text].every((c) => c >= '0' && c <= '9') - ) { - const last = parent.text.at(-1); - if (parent.fontSize >= subTh && last !== undefined && /\p{L}/u.test(last)) { - const gap = item.x - (parent.x + parent.width); - if (gap < parent.fontSize * 0.2 && gap > -parent.fontSize * 0.3) { - const raised = item.y > parent.y + parent.fontSize * 0.1; - const table = raised ? SUP : SUB; - parent.text += [...item.text].map((c) => table[Number(c)]!).join(''); - parent.width = item.x + item.width - parent.x; - continue; - } + for (const item of pageItems) { + let attached = false; + if (scriptDigits(item.text) !== undefined) { + for (let i = merged.length - 1; i >= 0; i -= 1) { + const parent = merged[i]!; + if (parent.x > item.x) continue; + if (!canAttachScript(parent, item)) continue; + const digits = scriptDigits(item.text)!; + const lowered = item.y < parent.y - parent.fontSize * 0.12; + const table = lowered ? SUB : SUP; + parent.text += [...digits].map((c) => table[Number(c)]!).join(''); + parent.width = Math.max(parent.width, item.x + item.width - parent.x); + attached = true; + break; } } - merged.push(item); + if (!attached) merged.push(item); } result.push(...merged); } @@ -3101,6 +3127,12 @@ interface FlowBlock { y2: number; items?: TextItem[]; markdown?: string; + role?: 'note' | 'footer'; +} + +function lineFlow(line: TextItem[], role?: 'note' | 'footer'): FlowBlock { + const box = lineBox(line); + return { kind: 'line', page: line[0]!.page, ...box, items: line, role }; } function itemsToMarkdown( @@ -3109,18 +3141,22 @@ function itemsToMarkdown( pageRects: PdfRect[], imageBlocks: MarkdownBlock[] = [], ): string { - const tables = detectTables(items, strokeLines, pageRects); + const peeled = peelFootnoteLines(groupIntoLines(items)); + const reserved = new Set([ + ...peeled.notes.flat(), + ...peeled.footer.flat(), + ...peeled.drop.flat(), + ]); + const bodyItems = items.filter((it) => !reserved.has(it)); + const tables = detectTables(bodyItems, strokeLines, pageRects); const claimed = new Set(); for (const table of tables) { for (const idx of table.itemIndices) claimed.add(idx); } - const remaining = items.filter((_, i) => !claimed.has(i)); + const remaining = bodyItems.filter((_, i) => !claimed.has(i)); const rawLines = groupIntoLines(remaining); const flow: FlowBlock[] = [ - ...rawLines.map((line) => { - const box = lineBox(line); - return { kind: 'line' as const, page: line[0]!.page, ...box, items: line }; - }), + ...rawLines.map((line) => lineFlow(line)), ...tables.map((t) => ({ kind: 'table' as const, page: t.page, @@ -3146,9 +3182,29 @@ function itemsToMarkdown( list.push(block); byPage.set(block.page, list); } + const notesByPage = new Map(); + for (const line of peeled.notes) { + const page = line[0]!.page; + const list = notesByPage.get(page) ?? []; + list.push(lineFlow(line, 'note')); + notesByPage.set(page, list); + } + const footerByPage = new Map(); + for (const line of peeled.footer) { + const page = line[0]!.page; + const list = footerByPage.get(page) ?? []; + list.push(lineFlow(line, 'footer')); + footerByPage.set(page, list); + } const ordered: FlowBlock[] = []; - for (const page of [...byPage.keys()].sort((a, b) => a - b)) { - ordered.push(...orderBoxes(byPage.get(page)!)); + const pages = new Set([...byPage.keys(), ...notesByPage.keys(), ...footerByPage.keys()]); + for (const page of [...pages].sort((a, b) => a - b)) { + const body = byPage.get(page) ?? []; + if (body.length > 0) ordered.push(...orderBoxes(body)); + const notes = notesByPage.get(page) ?? []; + if (notes.length > 0) ordered.push(...orderBoxes(notes)); + const footer = footerByPage.get(page) ?? []; + if (footer.length > 0) ordered.push(...orderBoxes(footer)); } const lines = ordered.filter((b) => b.kind === 'line').map((b) => b.items!); if (lines.length === 0 && tables.length === 0 && imageBlocks.length === 0) return ''; @@ -3246,6 +3302,7 @@ function itemsToMarkdown( let prevY = Number.POSITIVE_INFINITY; let prevPage = 0; let prevBox: LayoutBox | undefined; + let prevRole: FlowBlock['role']; let lastListX: number | undefined; let lineIndex = 0; @@ -3293,6 +3350,7 @@ function itemsToMarkdown( prevY = Number.POSITIVE_INFINITY; prevBox = undefined; prevPage = page; + prevRole = undefined; inList = false; lastListX = undefined; } @@ -3301,14 +3359,18 @@ function itemsToMarkdown( const yGap = prevY - y; const em = Math.max(line[0]!.fontSize, base); const indented = prevBox !== undefined && isFirstLineIndent(prevBox, box, em); - if (inPara && (yGap > paraTh || yGap < -base * 0.8 || indented)) { + const noteStart = + block.role === 'note' && /^(?:\d{1,3}|[⁰¹²³⁴⁵⁶⁷⁸⁹]{1,3})[.)]?(?:\s+|[A-Z])/.test(plain); + const roleChange = block.role !== prevRole && block.role !== undefined; + if (inPara && (yGap > paraTh || yGap < -base * 0.8 || indented || roleChange || noteStart)) { out += '\n\n'; inPara = false; } prevY = y; prevBox = box; + prevRole = block.role; - const lvl = headerLevel(i, line, plain); + const lvl = block.role === undefined ? headerLevel(i, line, plain) : undefined; if (lvl !== undefined) { if (inPara) out += '\n\n'; out += `${'#'.repeat(lvl)} ${plain}\n\n`; diff --git a/packages/pdf/test/layout.test.ts b/packages/pdf/test/layout.test.ts index 07004dd..9b01e20 100644 --- a/packages/pdf/test/layout.test.ts +++ b/packages/pdf/test/layout.test.ts @@ -417,3 +417,128 @@ ET expect(md).not.toMatch(/list item\. After the list/); }); }); + +describe('footnotes', () => { + it('attaches a raised digit after punctuation instead of dropping it into the next line', () => { + const md = toMarkdownFromPdf( + pagePdf( + `BT +/F1 12 Tf +1 0 0 1 20 220 Tm +(from FY2019.) Tj +/F1 7 Tf +4 Ts +(4) Tj +0 Ts +/F1 12 Tf +( The data collected after implementation) Tj +1 0 0 1 20 200 Tm +(of the FIT scheme revealed the costs.) Tj +/F1 8 Tf +1 0 0 1 20 40 Tm +(4 Biomass of waste is not eligible from FY2021.) Tj +ET +`, + [0, 0, 420, 280], + ), + ); + expect(md).toContain('FY2019.⁴'); + expect(md).not.toMatch(/implementation 4 of/); + expect(md).toMatch(/costs\.\n\n4 Biomass of waste/); + }); + + it('reads two-column notes after both columns of body, not mixed into the other column', () => { + const md = toMarkdownFromPdf( + pagePdf( + `BT +/F1 12 Tf +1 0 0 1 20 240 Tm +(Alpha) Tj +1 0 0 1 20 220 Tm +(Bravo) Tj +1 0 0 1 220 240 Tm +(Charlie) Tj +1 0 0 1 220 220 Tm +(Delta) Tj +/F1 8 Tf +1 0 0 1 20 50 Tm +(25 Left note continues here.) Tj +1 0 0 1 20 35 Tm +(26 Left note also continues.) Tj +1 0 0 1 220 50 Tm +(30 Right note continues here.) Tj +1 0 0 1 220 35 Tm +(31 Right note also continues.) Tj +ET +`, + [0, 0, 420, 280], + ), + ); + expect(md.indexOf('Alpha')).toBeLessThan(md.indexOf('Bravo')); + expect(md.indexOf('Bravo')).toBeLessThan(md.indexOf('Charlie')); + expect(md.indexOf('Charlie')).toBeLessThan(md.indexOf('Delta')); + expect(md.indexOf('Delta')).toBeLessThan(md.indexOf('25 Left note')); + expect(md.indexOf('25 Left note')).toBeLessThan(md.indexOf('26 Left note')); + expect(md.indexOf('26 Left note')).toBeLessThan(md.indexOf('30 Right note')); + expect(md).not.toMatch(/Bravo[\s\S]*25 Left note[\s\S]*Charlie/); + expect(md).toMatch(/25 Left note continues here\.\n\n26 Left note also continues/); + expect(md).toMatch(/30 Right note continues here\.\n\n31 Right note also continues/); + }); + + it('does not glue footer notes onto the last body paragraph', () => { + const md = toMarkdownFromPdf( + pagePdf( + `BT +/F1 12 Tf +1 0 0 1 20 120 Tm +(This report surveys land.) Tj +/F1 7 Tf +4 Ts +(1) Tj +0 Ts +/F1 12 Tf +1 0 0 1 20 100 Tm +(Coverage is selected to stay representative.) Tj +/F1 8 Tf +1 0 0 1 20 40 Tm +(1 The surveyed jurisdictions are listed in the appendix text.) Tj +1 0 0 1 20 25 Tm +(2 World Bank Databank Gross Domestic Product figures.) Tj +ET +`, + [0, 0, 420, 180], + ), + ); + expect(md).toContain('land.¹'); + expect(md).toMatch(/representative\.\n\n1 The surveyed jurisdictions/); + expect(md).toMatch(/appendix text\.\n\n2 World Bank Databank/); + expect(md).not.toMatch(/representative\. 1 The surveyed/); + }); + + it('keeps a raised note marker at the end of the sentence, not prepended', () => { + const md = toMarkdownFromPdf( + pagePdf( + `BT +/F1 12 Tf +1 0 0 1 20 160 Tm +(Now, how do we solve for the analytical equilibrium?) Tj +/F1 7 Tf +4 Ts +(12) Tj +0 Ts +/F1 12 Tf +1 0 0 1 20 140 Tm +(Player two applies backward induction to find the equilibrium.) Tj +/F1 8 Tf +1 0 0 1 20 40 Tm +(12. This equilibrium is known as a Perfect Bayesian Equilibrium.) Tj +ET +`, + [0, 0, 420, 220], + ), + ); + expect(md).toContain('equilibrium?¹²'); + expect(md).not.toMatch(/^12 Now,/m); + expect(md).toMatch(/equilibrium\.\n\n12\. This equilibrium/); + }); +}); diff --git a/packages/pdf/test/reading-order.test.ts b/packages/pdf/test/reading-order.test.ts index bc78f34..4787b2f 100644 --- a/packages/pdf/test/reading-order.test.ts +++ b/packages/pdf/test/reading-order.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest'; -import { groupIntoLines, orderBoxes } from '../src/layout.js'; +import { groupIntoLines, orderBoxes, peelFootnoteLines } from '../src/layout.js'; function word( text: string, @@ -226,3 +226,77 @@ describe('orderBoxes', () => { expect(ordered.map((b) => b.id)).toEqual(['L1', 'IMG', 'L2', 'R1', 'R2']); }); }); + +describe('peelFootnoteLines', () => { + it('keeps two-column notes after both columns of body', () => { + const lines = [ + [word('Left body one', 20, 200, 120)], + [word('Left body two', 20, 180, 120)], + [word('Right body one', 200, 200, 120)], + [word('Right body two', 200, 180, 120)], + [word('25 Left note one continues here.', 20, 50, 120, { fontSize: 8 })], + [word('26 Left note two continues here.', 20, 35, 120, { fontSize: 8 })], + [word('30 Right note one continues here.', 200, 50, 120, { fontSize: 8 })], + [word('31 Right note two continues here.', 200, 35, 120, { fontSize: 8 })], + ]; + const { body, notes, footer } = peelFootnoteLines(lines); + expect(body.map((line) => line[0]!.text)).toEqual([ + 'Left body one', + 'Left body two', + 'Right body one', + 'Right body two', + ]); + expect(notes.map((line) => line[0]!.text)).toEqual([ + '25 Left note one continues here.', + '26 Left note two continues here.', + '30 Right note one continues here.', + '31 Right note two continues here.', + ]); + expect(footer).toHaveLength(0); + }); + + it('matches a bottom note to a body superscript and peels a page number', () => { + const lines = [ + [word('from FY2019.⁴ The data collected after implementation', 20, 200, 300)], + [word('of the FIT scheme revealed the costs.', 20, 180, 280)], + [word('4 Biomass of waste is not eligible from FY2021.', 20, 40, 280, { fontSize: 9 })], + [word('31', 300, 18, 12, { fontSize: 9 })], + ]; + const { body, notes, footer } = peelFootnoteLines(lines); + expect(body.map((line) => line[0]!.text)).toEqual([ + 'from FY2019.⁴ The data collected after implementation', + 'of the FIT scheme revealed the costs.', + ]); + expect(notes).toHaveLength(1); + expect(notes[0]![0]!.text).toContain('Biomass of waste'); + expect(footer.map((line) => line[0]!.text)).toEqual(['31']); + }); + + it('drops a stray marker line that already appears as a body superscript', () => { + const lines = [ + [word('Coffee is called the wine of Islam.²⁶', 20, 200, 260)], + [word('26', 80, 204, 8, { fontSize: 7 })], + [word('Body continues after the marker.', 20, 180, 240)], + [word('26 For the association between coffee and wine.', 20, 40, 240, { fontSize: 8 })], + ]; + const { body, notes } = peelFootnoteLines(lines); + expect(body.map((line) => line[0]!.text)).toEqual([ + 'Coffee is called the wine of Islam.²⁶', + 'Body continues after the marker.', + ]); + expect(notes).toHaveLength(1); + expect(notes[0]![0]!.text).toContain('For the association'); + }); + + it('does not peel a top heading or a same-size numbered list', () => { + const lines = [ + [word('1. Introduction to land ownership', 20, 200, 260)], + [word('Body paragraph about the survey results.', 20, 160, 260)], + [word('1. First list item stays in the body.', 20, 40, 260)], + [word('2. Second list item stays in the body.', 20, 24, 260)], + ]; + const { body, notes } = peelFootnoteLines(lines); + expect(notes).toHaveLength(0); + expect(body).toHaveLength(4); + }); +}); From 90928eb7b453fc708f9867356796a65aa99936c0 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 26 Aug 2026 05:26:48 +0000 Subject: [PATCH 2/2] fix(pdf): keep notes, page numbers, and captions distinct Close lists before a note so footnotes do not glue onto the last item. Classify bottom page numbers as footers before dropping orphan markers. Test figure/table captions against the text after the leading number. Limit trailing marker scans to end-of-word digits. --- packages/pdf/src/layout.ts | 12 ++++-- packages/pdf/src/pdf.ts | 8 +++- packages/pdf/test/layout.test.ts | 56 +++++++++++++++++++++++++ packages/pdf/test/reading-order.test.ts | 37 ++++++++++++++++ 4 files changed, 108 insertions(+), 5 deletions(-) diff --git a/packages/pdf/src/layout.ts b/packages/pdf/src/layout.ts index d72b586..8c4f177 100644 --- a/packages/pdf/src/layout.ts +++ b/packages/pdf/src/layout.ts @@ -628,6 +628,7 @@ function splitPageFootnotes( for (const line of lines) { if (line[0]!.y > floor) continue; const text = linePlain(line); + if (isFooterText(text)) continue; const n = noteStartNumber(text); if (n !== undefined) { if (markers.has(n) || isSmallNoteLine(line, bodyFont)) starts.push(line); @@ -663,12 +664,15 @@ function splitPageFootnotes( const drop: T[][] = []; for (const line of rest) { const text = linePlain(line); + if (line[0]!.y < noteBottom && isFooterText(text)) { + footer.push(line); + continue; + } if (isOrphanMarkerLine(text, noteNums) || isOrphanMarkerLine(text, markers)) { drop.push(line); continue; } - if (line[0]!.y < noteBottom && isFooterText(text)) footer.push(line); - else body.push(line); + body.push(line); } return { body, notes, footer, drop }; } @@ -700,7 +704,7 @@ function bodyFontSize(lines: T[][]): number { function collectTrailingMarkers(lines: T[][]): Set { const nums = new Set(); - const glued = /(?<=[\p{L}.!?”"'”'])(\d{1,3})(?!\d)/gu; + const glued = /(?<=\p{L}{2,})(\d{1,3})(?=$|[^\d\p{L}])/gu; for (const line of lines) { const text = linePlain(line); for (const m of text.matchAll(glued)) nums.add(m[1]!); @@ -743,7 +747,7 @@ function noteStartNumber(text: string): string | undefined { if (!m) return undefined; const after = ascii.slice(m[0].length).trim(); if (after.length < 6) return undefined; - if (/^(figure|table|fig\.?)\b/i.test(ascii)) return undefined; + if (/^(figure|table|fig\.?)\b/i.test(after)) return undefined; return m[1]; } diff --git a/packages/pdf/src/pdf.ts b/packages/pdf/src/pdf.ts index 0487c16..d5c6d41 100644 --- a/packages/pdf/src/pdf.ts +++ b/packages/pdf/src/pdf.ts @@ -3362,7 +3362,13 @@ function itemsToMarkdown( const noteStart = block.role === 'note' && /^(?:\d{1,3}|[⁰¹²³⁴⁵⁶⁷⁸⁹]{1,3})[.)]?(?:\s+|[A-Z])/.test(plain); const roleChange = block.role !== prevRole && block.role !== undefined; - if (inPara && (yGap > paraTh || yGap < -base * 0.8 || indented || roleChange || noteStart)) { + if (roleChange || noteStart) { + if (inList) closeList(); + else if (inPara) { + out += '\n\n'; + inPara = false; + } + } else if (inPara && (yGap > paraTh || yGap < -base * 0.8 || indented)) { out += '\n\n'; inPara = false; } diff --git a/packages/pdf/test/layout.test.ts b/packages/pdf/test/layout.test.ts index 9b01e20..952ad50 100644 --- a/packages/pdf/test/layout.test.ts +++ b/packages/pdf/test/layout.test.ts @@ -515,6 +515,62 @@ ET expect(md).not.toMatch(/representative\. 1 The surveyed/); }); + it('does not glue footer notes onto the last list item', () => { + const md = toMarkdownFromPdf( + pagePdf( + `BT +/F1 12 Tf +1 0 0 1 20 200 Tm +(Coverage is selected to stay representative.) Tj +/F1 7 Tf +4 Ts +(1) Tj +0 Ts +/F1 12 Tf +1 0 0 1 20 90 Tm +(- First list item on this page.) Tj +1 0 0 1 20 72 Tm +(- Second list item on this page.) Tj +/F1 8 Tf +1 0 0 1 20 54 Tm +(1 The surveyed jurisdictions are listed in the appendix text.) Tj +ET +`, + [0, 0, 420, 240], + ), + ); + expect(md).toMatch(/Second list item on this page\.\n\n1 The surveyed jurisdictions/); + expect(md).not.toMatch(/Second list item on this page\. 1 The surveyed/); + }); + + it('does not treat a following note as the next list item', () => { + const md = toMarkdownFromPdf( + pagePdf( + `BT +/F1 12 Tf +1 0 0 1 20 200 Tm +(Coverage is selected to stay representative.) Tj +/F1 7 Tf +4 Ts +(1) Tj +0 Ts +/F1 12 Tf +1 0 0 1 20 160 Tm +(1. First list item stays in the body.) Tj +1 0 0 1 20 140 Tm +(2. Second list item stays in the body.) Tj +/F1 8 Tf +1 0 0 1 20 40 Tm +(1. The surveyed jurisdictions are listed in the appendix text.) Tj +ET +`, + [0, 0, 420, 240], + ), + ); + expect(md).toMatch(/stays in the body\.\n\n1\. The surveyed jurisdictions/); + expect(md).not.toMatch(/stays in the body\.\n1\. The surveyed/); + }); + it('keeps a raised note marker at the end of the sentence, not prepended', () => { const md = toMarkdownFromPdf( pagePdf( diff --git a/packages/pdf/test/reading-order.test.ts b/packages/pdf/test/reading-order.test.ts index 4787b2f..577b7de 100644 --- a/packages/pdf/test/reading-order.test.ts +++ b/packages/pdf/test/reading-order.test.ts @@ -288,6 +288,43 @@ describe('peelFootnoteLines', () => { expect(notes[0]![0]!.text).toContain('For the association'); }); + it('keeps a page number that matches a footnote number', () => { + const lines = [ + [word('Body with a citation.¹ More text follows after that.', 20, 200, 300)], + [word('1 The first note explains the method used here.', 20, 40, 280, { fontSize: 8 })], + [word('1', 300, 18, 12, { fontSize: 9 })], + ]; + const { body, notes, footer, drop } = peelFootnoteLines(lines); + expect(body.map((line) => line[0]!.text)).toEqual([ + 'Body with a citation.¹ More text follows after that.', + ]); + expect(notes).toHaveLength(1); + expect(footer.map((line) => line[0]!.text)).toEqual(['1']); + expect(drop).toHaveLength(0); + }); + + it('does not treat a numbered figure caption as a footnote', () => { + const lines = [ + [word('The results appear below the fold on this page.', 20, 200, 280)], + [word('1 Figure of the experimental setup on page two.', 20, 40, 280, { fontSize: 8 })], + ]; + const { body, notes } = peelFootnoteLines(lines); + expect(notes).toHaveLength(0); + expect(body).toHaveLength(2); + }); + + it('does not peel a numbered list when the body has figure and version digits', () => { + const lines = [ + [word('See Fig.1 and p.45 and v2 and 3.14 in the text.', 20, 200, 300)], + [word('Body paragraph about the survey results here.', 20, 160, 260)], + [word('1. First list item stays in the body.', 20, 40, 260)], + [word('2. Second list item stays in the body.', 20, 24, 260)], + ]; + const { body, notes } = peelFootnoteLines(lines); + expect(notes).toHaveLength(0); + expect(body).toHaveLength(4); + }); + it('does not peel a top heading or a same-size numbered list', () => { const lines = [ [word('1. Introduction to land ownership', 20, 200, 260)],