From 0e9f24bd1b509aa4682df8d3a7281e2fd9ddb66f Mon Sep 17 00:00:00 2001 From: Asiones Jia Date: Wed, 26 Aug 2026 20:55:30 +0800 Subject: [PATCH 1/2] fix(pdf): keep extractable chart labels out of broken tables Chart ticks, bar values, and legends were detected as tables, and axis gridlines were marked as underline/strike. Skip rotated items in table detection, reject chart-like grids, join rotated date labels, and ignore rules that run far past a glyph. fixes #76 --- packages/pdf/src/layout.ts | 2 +- packages/pdf/src/pdf.ts | 53 +++++- packages/pdf/src/tables.ts | 125 ++++++++++++++- packages/pdf/test/tables.test.ts | 267 +++++++++++++++++++++++++++++++ 4 files changed, 441 insertions(+), 6 deletions(-) diff --git a/packages/pdf/src/layout.ts b/packages/pdf/src/layout.ts index 8c4f177..942d89a 100644 --- a/packages/pdf/src/layout.ts +++ b/packages/pdf/src/layout.ts @@ -49,7 +49,7 @@ export function groupIntoLines(items: T[]): T[][] { return lines; } -function isUpright(item: LineItem): boolean { +export function isUpright(item: { dx?: number; dy?: number }): boolean { const dx = item.dx ?? 1; const dy = item.dy ?? 0; return Math.abs(dy) <= Math.abs(dx) * 0.35; diff --git a/packages/pdf/src/pdf.ts b/packages/pdf/src/pdf.ts index d5c6d41..b20737e 100644 --- a/packages/pdf/src/pdf.ts +++ b/packages/pdf/src/pdf.ts @@ -19,6 +19,7 @@ import { decryptBytes, type EncryptParams, type FileCrypt, openEncrypt } from '. import { xObjectToImage } from './images.js'; import { groupIntoLines, + isUpright, type LayoutBox, lineBox, orderBoxes, @@ -142,7 +143,7 @@ async function convertUniqueImages( function finishPdf(extracted: ExtractedPdf, imageBlocks: MarkdownBlock[]): string { const markdown = itemsToMarkdown( - mergeScriptItems(dedupeOverlappingItems(extracted.items)), + mergeNumericFragments(mergeScriptItems(dedupeOverlappingItems(extracted.items))), extracted.strokeLines, extracted.pageRects, imageBlocks, @@ -2868,6 +2869,14 @@ function markDecorations(items: TextItem[], lines: StrokeLine[], httpRects: Rect for (const rule of rules) { const overlap = Math.min(item.x + item.width, rule.x2) - Math.max(item.x, rule.x1); if (overlap < item.width * 0.6) continue; + const ruleW = rule.x2 - rule.x1; + const leftHang = Math.max(0, item.x - rule.x1); + const rightHang = Math.max(0, rule.x2 - (item.x + item.width)); + const fs = Math.max(item.fontSize, 8); + const maxHang = Math.max(leftHang, rightHang); + if (ruleW > Math.max(item.width * 6, fs * 12) && maxHang > Math.max(item.width * 4, fs * 8)) { + continue; + } const below = Math.max(item.fontSize * 0.72, 3); if (!skipHttp && rule.y >= item.y - below && rule.y <= item.y + 1) { item.isUnderline = true; @@ -2961,6 +2970,41 @@ function canAttachScript(parent: TextItem, item: TextItem): boolean { return true; } +function isNumericFragment(text: string): boolean { + const t = text.trim(); + return t.length > 0 && t.length <= 8 && /^[\d.,\s]+%?$/.test(t) && /[\d%]/.test(t); +} + +function canMergeNumeric(prev: TextItem, item: TextItem): boolean { + if (prev.page !== item.page) return false; + if (!isUpright(prev) || !isUpright(item)) return false; + if (!isNumericFragment(prev.text) || !isNumericFragment(item.text)) return false; + const fs = Math.max(prev.fontSize, item.fontSize, 8); + if (Math.abs(prev.y - item.y) > fs * 0.6) return false; + const prevW = Math.max(prev.width, fs * 0.35); + const gap = item.x - (prev.x + prevW); + if (gap > fs * 0.55 || gap < -fs * 0.35) return false; + return true; +} + +function mergeNumericFragments(items: TextItem[]): TextItem[] { + if (items.length < 2) return items; + const sorted = items.slice().sort((a, b) => a.page - b.page || b.y - a.y || a.x - b.x); + const out: TextItem[] = []; + for (const item of sorted) { + const prev = out[out.length - 1]; + if (prev && canMergeNumeric(prev, item)) { + const next = item.text.trim(); + prev.text = `${prev.text.trimEnd()}${next}`; + prev.width = Math.max(prev.width, item.x + item.width - prev.x); + prev.height = Math.max(prev.height, item.height); + continue; + } + out.push(item); + } + return out; +} + function mergeScriptItems(items: TextItem[]): TextItem[] { if (items.length < 2) return items; const byPage = new Map(); @@ -3006,13 +3050,14 @@ function shouldJoinItems(prev: TextItem, curr: TextItem): boolean { if (currFirst !== undefined && ".,;!?)]}'".includes(currFirst)) return true; const prevLast = [...prev.text.trimEnd()].at(-1); if (prevLast === ':' && currFirst !== undefined && /[\p{L}\p{N}]/u.test(currFirst)) return false; - if (prev.width > 0) { + const fs = Math.max(prev.fontSize, curr.fontSize, 8); + if (isUpright(prev) && isUpright(curr) && prev.width > 0) { const gap = curr.x - (prev.x + prev.width); - const fs = prev.fontSize; if (gap > fs * 3 || gap < -fs) return false; return gap < fs * 0.12; } - return false; + const dist = Math.hypot(curr.x - prev.x, curr.y - prev.y); + return dist < fs * 1.8; } function needsSpace(prev: TextItem, curr: TextItem, result: string): boolean { diff --git a/packages/pdf/src/tables.ts b/packages/pdf/src/tables.ts index 93b5861..9a6449c 100644 --- a/packages/pdf/src/tables.ts +++ b/packages/pdf/src/tables.ts @@ -1,5 +1,7 @@ /** Ruled-grid and borderless table detection from lines, thin `re` borders, and aligned text. */ +import { isUpright } from './layout.js'; + export interface TableTextItem { text: string; x: number; @@ -7,6 +9,8 @@ export interface TableTextItem { width: number; page: number; fontSize?: number; + dx?: number; + dy?: number; } export interface TableLine { @@ -229,6 +233,7 @@ function buildGridTable( if (claimed.has(idx)) continue; const item = items[idx]!; if (item.page !== page || item.text.trim().length === 0) continue; + if (!isUpright(item)) continue; const cx = item.x + Math.max(item.width, 0) / 2; const cy = item.y; if (cx < xLeft - 4 || cx > xRight + 4 || cy < yBottom - 4 || cy > yTop + 4) continue; @@ -314,6 +319,7 @@ function acceptGrid(grid: string[][]): boolean { if (avg > 220) return false; const long = lengths.filter((n) => n > 160).length; if (long > 0 && long / lengths.length > 0.2 && longCols > 2) return false; + if (looksLikeDateFragments(grid)) return false; return true; } @@ -376,6 +382,7 @@ function collectAlignedRows(items: TableTextItem[], page: number, claimed: Set b.item.y - a.item.y || a.item.x - b.item.x); @@ -671,10 +678,21 @@ function acceptBorderless( if (words > MAX_TABLE_WORDS) return false; const avgChars = cells.reduce((n, c) => n + [...c].length, 0) / Math.max(cells.length, 1); const nCols = grid[0]!.length; - if (nCols === 2 && avgChars > PROSE_AVG_CELL_CHARS) { + if (nCols >= 2 && avgChars > PROSE_AVG_CELL_CHARS) { const avgWords = words / Math.max(cells.length, 1); if (avgWords > 4) return false; } + if ( + looksLikeDateFragments(grid) || + looksLikeChartLegend(grid) || + looksLikeChartValues(grid, run) || + looksLikeNumericScatter(grid) || + looksLikeLabeledTicks(grid) || + looksLikeHeaderlessJunk(grid) + ) { + return false; + } + if (mixedChartAndProse(grid)) return false; if (pageBox && nCols <= 2) { const x1 = Math.min(...run.flatMap((r) => r.segs.map((s) => s.x))); const x2 = Math.max(...run.flatMap((r) => r.segs.map((s) => s.x2))); @@ -687,6 +705,111 @@ function acceptBorderless( return true; } +function looksLikeDateFragments(grid: string[][]): boolean { + const cells = grid.flat().filter((c) => c.length > 0); + if (cells.length < 8) return false; + if (grid[0]!.length < 6) return false; + if (cells.some((c) => [...c].length > 5)) return false; + const frag = cells.filter((c) => /^(?:\d{1,4}\/?|\/?\d{1,4})$/.test(c.trim())).length; + return frag / cells.length >= 0.6; +} + +function looksLikeChartLegend(grid: string[][]): boolean { + if (grid.length < 2 || grid.length > 3) return false; + if (grid[0]!.length < 3) return false; + const first = grid[0]!.filter((c) => c.length > 0); + const last = grid[grid.length - 1]!.filter((c) => c.length > 0); + if (first.length < 3 || last.length === 0) return false; + const categories = first.every((c) => wordCount(c) <= 2 && /[\p{L}]/u.test(c)); + const years = last.every((c) => /^(?:19|20)\d{2}$/.test(c.trim())); + return categories && years; +} + +function looksLikeChartValues(grid: string[][], run: RowCand[]): boolean { + const cells = grid.flat().filter((c) => c.length > 0); + if (cells.length < 4) return false; + const numeric = cells.filter((c) => /^-?[\d.,]+%?$/.test(c.trim())).length; + if (numeric / cells.length < 0.8) return false; + const avgLen = cells.reduce((n, c) => n + [...c].length, 0) / cells.length; + if (avgLen > 8) return false; + const nCols = grid[0]!.length; + const fill = cells.length / (grid.length * nCols); + return fill < 0.55 || rowPitchIrregular(run); +} + +function mixedChartAndProse(grid: string[][]): boolean { + const cells = grid.flat().filter((c) => c.length > 0); + const long = cells.filter((c) => wordCount(c) >= 6).length; + const chartish = cells.filter((c) => /%/.test(c) || /^-?[\d.,]+%?$/.test(c.trim())).length; + return long >= 2 && chartish >= 3; +} + +function isNumericToken(text: string): boolean { + const t = text.trim(); + return /^-?[\d.,]+%?$/.test(t) || /^(?:19|20)\d{2}$/.test(t); +} + +function looksLikeNumericScatter(grid: string[][]): boolean { + const header = grid[0]!.filter((c) => c.length > 0); + const headerLabels = header.filter((c) => /[\p{L}]/u.test(c) && !isNumericToken(c)).length; + if (header.length >= 2 && headerLabels >= Math.ceil(header.length * 0.6)) return false; + const col0 = grid.map((row) => row[0] ?? '').filter((c) => c.length > 0); + const col0Labels = col0.filter((c) => /[\p{L}]/u.test(c) && !isNumericToken(c)).length; + if (col0.length >= 2 && col0Labels / col0.length >= 0.6) return false; + const cells = grid.flat().filter((c) => c.length > 0); + const tokens = cells.flatMap((c) => c.split(/\s+/).filter((w) => w.length > 0)); + if (tokens.length < 6) return false; + const numeric = tokens.filter((t) => isNumericToken(t)).length; + if (numeric / tokens.length < 0.45) return false; + const letters = tokens.filter((t) => /[\p{L}]{3,}/u.test(t) && !isNumericToken(t)).length; + return letters <= tokens.length * 0.4; +} + +function looksLikeLabeledTicks(grid: string[][]): boolean { + const filled = grid.map((row) => row.filter((c) => c.length > 0)).filter((r) => r.length >= 2); + const percentRow = filled.some( + (r) => + r.filter((c) => /%/.test(c) || /^-?[\d.,]+$/.test(c.trim())).length >= + Math.ceil(r.length * 0.7) && r.every((c) => wordCount(c) <= 2), + ); + const labelRow = filled.some( + (r) => + r.length >= 3 && r.every((c) => wordCount(c) <= 2 && /[\p{L}]/u.test(c) && !/\d/.test(c)), + ); + return percentRow && labelRow; +} + +function looksLikeHeaderlessJunk(grid: string[][]): boolean { + if (grid.length !== 2 || grid[0]!.length < 3) return false; + const r0 = grid[0]!.filter((c) => c.length > 0); + const r1 = grid[1]!.filter((c) => c.length > 0); + if (r0.length < 2 || r1.length === 0) return false; + const allLabels = (row: string[]): boolean => + row.every((c) => !/\d/.test(c) && wordCount(c) <= 6); + const hasDigit = (row: string[]): boolean => row.some((c) => /\d/.test(c)); + if (allLabels(r0) && hasDigit(r1)) return false; + if (!hasDigit(r0) || !hasDigit(r1)) return false; + const labels = r0.filter((c) => /[\p{L}]/u.test(c) && !isNumericToken(c)).length; + const years = r0.filter((c) => isNumericToken(c)).length; + if (labels >= 1 && years >= 2) return false; + return true; +} + +function rowPitchIrregular(run: RowCand[]): boolean { + if (run.length < 3) return false; + const gaps: number[] = []; + for (let i = 1; i < run.length; i += 1) { + const g = run[i - 1]!.y - run[i]!.y; + if (g > 0.5) gaps.push(g); + } + if (gaps.length < 2) return false; + const mean = gaps.reduce((a, b) => a + b, 0) / gaps.length; + let sumSq = 0; + for (const g of gaps) sumSq += (g - mean) ** 2; + const std = Math.sqrt(sumSq / gaps.length); + return std > mean * 0.45 && std > 6; +} + function nearestCol(colXs: number[], x: number): number { let best = 0; let bestD = Number.POSITIVE_INFINITY; diff --git a/packages/pdf/test/tables.test.ts b/packages/pdf/test/tables.test.ts index b3f5a11..1a5c6ae 100644 --- a/packages/pdf/test/tables.test.ts +++ b/packages/pdf/test/tables.test.ts @@ -348,6 +348,102 @@ describe('ruled table detection', () => { expect(md).toContain('|Item|Qty|'); expect(md).toContain('|Pen|2|'); }); + + it('does not treat bar-chart categories and year ticks as a table', () => { + const items = [ + { text: 'Event', x: 40, y: 80, width: 28, page: 1 }, + { text: 'Celebration', x: 100, y: 80, width: 50, page: 1 }, + { text: 'Information', x: 180, y: 80, width: 52, page: 1 }, + { text: 'Videograph', x: 260, y: 80, width: 48, page: 1 }, + { text: '2019', x: 120, y: 58, width: 22, page: 1 }, + { text: '2020', x: 180, y: 58, width: 22, page: 1 }, + ]; + expect(detectTables(items, [], [])).toHaveLength(0); + }); + + it('does not treat stacked date-axis fragments as a table', () => { + const xs = [40, 70, 100, 130, 160, 190, 220, 250]; + const items = [ + ...xs.map((x) => ({ text: '9', x, y: 90, width: 6, page: 1 })), + ...xs.map((x) => ({ text: '201', x, y: 80, width: 16, page: 1 })), + ...xs.map((x, i) => ({ text: i % 2 === 0 ? '1/' : '3/', x, y: 70, width: 10, page: 1 })), + ...xs.map((x) => ({ text: '0', x, y: 60, width: 6, page: 1 })), + ]; + expect(detectTables(items, [], [])).toHaveLength(0); + }); + + it('does not treat irregular bar-top numbers as a table', () => { + const items = [ + { text: '1,450', x: 20, y: 120, width: 24, page: 1 }, + { text: '1,427', x: 80, y: 118, width: 24, page: 1 }, + { text: '1,393', x: 20, y: 96, width: 24, page: 1 }, + { text: '1,386', x: 80, y: 84, width: 24, page: 1 }, + { text: '1,368', x: 20, y: 70, width: 24, page: 1 }, + { text: '1,232', x: 80, y: 40, width: 24, page: 1 }, + ]; + expect(detectTables(items, [], [])).toHaveLength(0); + }); + + it('does not treat a row of chart section titles as a table', () => { + const items = [ + { text: 'Comparison with Beauty Commerce', x: 20, y: 120, width: 140, page: 1 }, + { text: 'Domestic Subscription Platform Case', x: 180, y: 120, width: 160, page: 1 }, + { text: 'Education Content Platform PoC Case', x: 360, y: 120, width: 150, page: 1 }, + { text: 'Hit Ratio comparison of models', x: 20, y: 96, width: 140, page: 1 }, + { text: 'Quantitative evaluations among content', x: 180, y: 96, width: 160, page: 1 }, + { text: 'Prediction rates of student answers', x: 360, y: 96, width: 150, page: 1 }, + ]; + expect(detectTables(items, [], [])).toHaveLength(0); + }); + + it('keeps a two-row table whose header mixes a label with year columns', () => { + const items = [ + { text: 'Product', x: 20, y: 80, width: 40, page: 1 }, + { text: '2019', x: 90, y: 80, width: 22, page: 1 }, + { text: '2020', x: 140, y: 80, width: 22, page: 1 }, + { text: 'Widgets', x: 20, y: 60, width: 40, page: 1 }, + { text: '12', x: 90, y: 60, width: 12, page: 1 }, + { text: '15', x: 140, y: 60, width: 12, page: 1 }, + ]; + const tables = detectTables(items, [], []); + expect(tables).toHaveLength(1); + expect(tables[0]!.markdown).toContain('|Product|2019|2020|'); + expect(tables[0]!.markdown).toContain('|Widgets|12|15|'); + }); + + it('does not treat percent bars with category ticks as a table', () => { + const items = [ + { text: '7%', x: 40, y: 80, width: 16, page: 1 }, + { text: '7%', x: 200, y: 80, width: 16, page: 1 }, + { text: '5,4%', x: 250, y: 80, width: 22, page: 1 }, + { text: 'OFTEN', x: 30, y: 50, width: 36, page: 1 }, + { text: 'SOMETIMES', x: 90, y: 50, width: 60, page: 1 }, + { text: 'RARELY', x: 170, y: 50, width: 40, page: 1 }, + { text: 'NEVER', x: 240, y: 50, width: 36, page: 1 }, + ]; + expect(detectTables(items, [], [])).toHaveLength(0); + }); + + it('does not treat rotated axis labels as table cells', () => { + const items = [ + { text: '0', x: 40, y: 50, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '1', x: 40, y: 58, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '/', x: 40, y: 66, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '2', x: 40, y: 74, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '0', x: 70, y: 50, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '3', x: 70, y: 58, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '/', x: 70, y: 66, width: 0, page: 1, dx: 0, dy: 1 }, + { text: '2', x: 70, y: 74, width: 0, page: 1, dx: 0, dy: 1 }, + { text: 'Body', x: 20, y: 120, width: 24, page: 1 }, + { text: 'text', x: 80, y: 120, width: 20, page: 1 }, + { text: 'here', x: 20, y: 100, width: 22, page: 1 }, + { text: 'now', x: 80, y: 100, width: 20, page: 1 }, + ]; + const tables = detectTables(items, [], []); + for (const t of tables) { + expect(t.markdown).not.toMatch(/0.*1.*\//); + } + }); }); function filledPathGridPdf(): Uint8Array { @@ -466,4 +562,175 @@ ET expect(md).toMatch(/\|Ada\|36\|London\|/); expect(md).toMatch(/\|Bob\|41\|Paris\|/); }); + + it('keeps chart axis ticks out of tables and does not underline them', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 24 80 Tm (35) Tj +1 0 0 1 24 60 Tm (0) Tj +1 0 0 1 70 40 Tm (Event) Tj +1 0 0 1 130 40 Tm (Celebration) Tj +1 0 0 1 210 40 Tm (Information) Tj +1 0 0 1 300 40 Tm (Videograph) Tj +1 0 0 1 150 22 Tm (2019) Tj +1 0 0 1 210 22 Tm (2020) Tj +ET +20 58 360 1 re S +`; + const objects = [ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 400 120] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>\nendobj\n', + ]; + let body = '%PDF-1.4\n'; + const offsets = [0]; + for (const obj of objects) { + offsets.push(body.length); + body += obj; + } + const xrefAt = body.length; + let xref = `xref\n0 6\n0000000000 65535 f \n`; + for (let i = 1; i <= 5; i += 1) xref += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + body += `${xref}trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n${xrefAt}\n%%EOF\n`; + const md = toMarkdownFromPdf(new TextEncoder().encode(body)); + expect(md).toContain('Event'); + expect(md).toContain('Celebration'); + expect(md).toContain('Information'); + expect(md).toContain('Videograph'); + expect(md).toContain('2019'); + expect(md).toContain('2020'); + expect(md).not.toMatch(/\|Event\|/); + expect(md).not.toContain(''); + expect(md).not.toContain(''); + }); + + it('still underlines a short rule under a word', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 70 Tm (underlined) Tj +1 0 0 1 20 50 Tm (Body line follows here) Tj +ET +18 68 58 0.6 re S +`; + const objects = [ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 80] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>\nendobj\n', + ]; + let body = '%PDF-1.4\n'; + const offsets = [0]; + for (const obj of objects) { + offsets.push(body.length); + body += obj; + } + const xrefAt = body.length; + let xref = `xref\n0 6\n0000000000 65535 f \n`; + for (let i = 1; i <= 5; i += 1) xref += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + body += `${xref}trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n${xrefAt}\n%%EOF\n`; + const md = toMarkdownFromPdf(new TextEncoder().encode(body)); + expect(md).toContain('underlined'); + }); + + it('keeps a short underline across several words', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 70 Tm (a ) Tj +(sibling) Tj +( file) Tj +1 0 0 1 20 50 Tm (Body line follows here) Tj +ET +20 68 78 0.6 re S +`; + const objects = [ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 80] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>\nendobj\n', + ]; + let body = '%PDF-1.4\n'; + const offsets = [0]; + for (const obj of objects) { + offsets.push(body.length); + body += obj; + } + const xrefAt = body.length; + let xref = `xref\n0 6\n0000000000 65535 f \n`; + for (let i = 1; i <= 5; i += 1) xref += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + body += `${xref}trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n${xrefAt}\n%%EOF\n`; + const md = toMarkdownFromPdf(new TextEncoder().encode(body)); + expect(md).toContain('a sibling file'); + }); + + it('assembles rotated date labels instead of a fragment table', () => { + const content = `BT +/F1 10 Tf +0 1 -1 0 80 40 Tm +(0) Tj (1) Tj (/) Tj (2) Tj (0) Tj (1) Tj (9) Tj +ET +BT +/F1 10 Tf +0 1 -1 0 110 40 Tm +(0) Tj (3) Tj (/) Tj (2) Tj (0) Tj (1) Tj (9) Tj +ET +`; + const objects = [ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 120] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>\nendobj\n', + ]; + let body = '%PDF-1.4\n'; + const offsets = [0]; + for (const obj of objects) { + offsets.push(body.length); + body += obj; + } + const xrefAt = body.length; + let xref = `xref\n0 6\n0000000000 65535 f \n`; + for (let i = 1; i <= 5; i += 1) xref += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + body += `${xref}trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n${xrefAt}\n%%EOF\n`; + const md = toMarkdownFromPdf(new TextEncoder().encode(body)); + expect(md).toContain('01/2019'); + expect(md).toContain('03/2019'); + expect(md).not.toMatch(/0\s+1\//); + expect(md).not.toMatch(/\|9\|/); + expect(md).not.toMatch(/\|201\|/); + }); + + it('joins a TJ-split bar percent into one token', () => { + const content = `BT +/F1 8 Tf +1 0 0 1 80 50 Tm +[(5) 5 (3,9%)] TJ +1 0 0 1 20 20 Tm +(Body follows here) Tj +ET +`; + const objects = [ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 80] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>\nendobj\n', + ]; + let body = '%PDF-1.4\n'; + const offsets = [0]; + for (const obj of objects) { + offsets.push(body.length); + body += obj; + } + const xrefAt = body.length; + let xref = `xref\n0 6\n0000000000 65535 f \n`; + for (let i = 1; i <= 5; i += 1) xref += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + body += `${xref}trailer\n<< /Size 6 /Root 1 0 R >>\nstartxref\n${xrefAt}\n%%EOF\n`; + const md = toMarkdownFromPdf(new TextEncoder().encode(body)); + expect(md).toContain('53,9%'); + expect(md).not.toMatch(/\b5\b[\s\S]*3,9%/); + }); }); From c04205f608a93dee733d112ddae9cbbf36fbbac7 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 26 Aug 2026 13:07:30 +0000 Subject: [PATCH 2/2] fix(pdf): keep percent, quarter, and numeric tables Chart-junk checks were dropping ordinary summary tables, Q1-style headers, and ruled numeric matrices. --- packages/pdf/src/tables.ts | 20 ++++++------ packages/pdf/test/tables.test.ts | 56 ++++++++++++++++++++++++++++++++ 2 files changed, 66 insertions(+), 10 deletions(-) diff --git a/packages/pdf/src/tables.ts b/packages/pdf/src/tables.ts index 9a6449c..1ca3d6d 100644 --- a/packages/pdf/src/tables.ts +++ b/packages/pdf/src/tables.ts @@ -710,6 +710,7 @@ function looksLikeDateFragments(grid: string[][]): boolean { if (cells.length < 8) return false; if (grid[0]!.length < 6) return false; if (cells.some((c) => [...c].length > 5)) return false; + if (!cells.some((c) => c.includes('/'))) return false; const frag = cells.filter((c) => /^(?:\d{1,4}\/?|\/?\d{1,4})$/.test(c.trim())).length; return frag / cells.length >= 0.6; } @@ -767,16 +768,14 @@ function looksLikeNumericScatter(grid: string[][]): boolean { function looksLikeLabeledTicks(grid: string[][]): boolean { const filled = grid.map((row) => row.filter((c) => c.length > 0)).filter((r) => r.length >= 2); - const percentRow = filled.some( - (r) => - r.filter((c) => /%/.test(c) || /^-?[\d.,]+$/.test(c.trim())).length >= - Math.ceil(r.length * 0.7) && r.every((c) => wordCount(c) <= 2), - ); - const labelRow = filled.some( - (r) => - r.length >= 3 && r.every((c) => wordCount(c) <= 2 && /[\p{L}]/u.test(c) && !/\d/.test(c)), - ); - return percentRow && labelRow; + const isPercentRow = (r: string[]): boolean => + r.filter((c) => /%/.test(c) || /^-?[\d.,]+$/.test(c.trim())).length >= + Math.ceil(r.length * 0.7) && r.every((c) => wordCount(c) <= 2); + const isLabelRow = (r: string[]): boolean => + r.length >= 3 && r.every((c) => wordCount(c) <= 2 && /[\p{L}]/u.test(c) && !/\d/.test(c)); + const percentAt = filled.findIndex(isPercentRow); + const labelAt = filled.findIndex(isLabelRow); + return percentAt >= 0 && labelAt >= 0 && percentAt < labelAt; } function looksLikeHeaderlessJunk(grid: string[][]): boolean { @@ -792,6 +791,7 @@ function looksLikeHeaderlessJunk(grid: string[][]): boolean { const labels = r0.filter((c) => /[\p{L}]/u.test(c) && !isNumericToken(c)).length; const years = r0.filter((c) => isNumericToken(c)).length; if (labels >= 1 && years >= 2) return false; + if (labels >= Math.ceil(r0.length * 0.6)) return false; return true; } diff --git a/packages/pdf/test/tables.test.ts b/packages/pdf/test/tables.test.ts index 1a5c6ae..32c6b21 100644 --- a/packages/pdf/test/tables.test.ts +++ b/packages/pdf/test/tables.test.ts @@ -424,6 +424,62 @@ describe('ruled table detection', () => { expect(detectTables(items, [], [])).toHaveLength(0); }); + it('keeps a borderless percent summary with category headers', () => { + const items = [ + { text: 'North', x: 20, y: 80, width: 30, page: 1 }, + { text: 'South', x: 80, y: 80, width: 30, page: 1 }, + { text: 'East', x: 140, y: 80, width: 24, page: 1 }, + { text: 'West', x: 200, y: 80, width: 24, page: 1 }, + { text: '12%', x: 22, y: 60, width: 18, page: 1 }, + { text: '18%', x: 82, y: 60, width: 18, page: 1 }, + { text: '9%', x: 142, y: 60, width: 14, page: 1 }, + { text: '21%', x: 202, y: 60, width: 18, page: 1 }, + ]; + const tables = detectTables(items, [], []); + expect(tables).toHaveLength(1); + expect(tables[0]!.markdown).toContain('|North|South|East|West|'); + expect(tables[0]!.markdown).toContain('|12%|18%|9%|21%|'); + }); + + it('keeps a two-row table with quarter or ordinal headers', () => { + const items = [ + { text: 'Q1', x: 20, y: 80, width: 14, page: 1 }, + { text: 'Q2', x: 80, y: 80, width: 14, page: 1 }, + { text: 'Q3', x: 140, y: 80, width: 14, page: 1 }, + { text: '10', x: 22, y: 60, width: 14, page: 1 }, + { text: '20', x: 82, y: 60, width: 14, page: 1 }, + { text: '30', x: 142, y: 60, width: 14, page: 1 }, + ]; + const tables = detectTables(items, [], []); + expect(tables).toHaveLength(1); + expect(tables[0]!.markdown).toContain('|Q1|Q2|Q3|'); + expect(tables[0]!.markdown).toContain('|10|20|30|'); + }); + + it('keeps a ruled numeric matrix', () => { + const xs = [20, 50, 80, 110, 140, 170, 200]; + const items = [ + ...xs.slice(0, 6).map((x, i) => ({ text: String(i), x: x + 4, y: 82, width: 10, page: 1 })), + ...xs.slice(0, 6).map((x, i) => ({ + text: String(10 + i), + x: x + 4, + y: 52, + width: 14, + page: 1, + })), + ]; + const rects = [ + { x: 20, y: 90, width: 180, height: 1, page: 1 }, + { x: 20, y: 70, width: 180, height: 1, page: 1 }, + { x: 20, y: 40, width: 180, height: 1, page: 1 }, + ...xs.map((x) => ({ x, y: 40, width: 1, height: 50, page: 1 })), + ]; + const tables = detectTables(items, [], rects); + expect(tables).toHaveLength(1); + expect(tables[0]!.markdown).toContain('|0|1|2|3|4|5|'); + expect(tables[0]!.markdown).toContain('|10|11|12|13|14|15|'); + }); + it('does not treat rotated axis labels as table cells', () => { const items = [ { text: '0', x: 40, y: 50, width: 0, page: 1, dx: 0, dy: 1 },