diff --git a/packages/pdf/src/pdf.ts b/packages/pdf/src/pdf.ts index 4114fb7..a88ddf0 100644 --- a/packages/pdf/src/pdf.ts +++ b/packages/pdf/src/pdf.ts @@ -1397,7 +1397,10 @@ function loadFont(doc: PdfDocument, obj: PdfValue | undefined): FontInfo { const tu = d.map.get('/ToUnicode'); if (tu !== undefined) { const parsed = parseToUnicode(decodeStream(doc, tu)); - for (const [k, v] of parsed.map) cmap.set(k, v); + for (const [k, v] of parsed.map) { + if (!cmapValueUsable(v)) continue; + cmap.set(k, v); + } if (!encCmap && !uni) codeByteLength = parsed.codeByteLength; hasToUnicode = parsed.map.size > 0; } @@ -2002,6 +2005,14 @@ function loadPageFonts(doc: PdfDocument, page: PdfDict): Map { return fonts; } +function cmapValueUsable(text: string): boolean { + if (text.length === 0) return false; + for (const ch of text) { + if (ch !== '\uFFFD') return true; + } + return false; +} + function mapDecoded( font: FontInfo, code: number, @@ -2009,7 +2020,7 @@ function mapDecoded( unicodePassthrough: boolean, ): DecodedChar { const fromCode = font.cmap.get(code); - if (fromCode !== undefined) { + if (fromCode !== undefined && cmapValueUsable(fromCode)) { return { ch: stripInvisibles(fromCode), code: cid || code, mapped: true }; } // Uni* encodings put Unicode in the content stream. That number is not a CID. @@ -2025,12 +2036,12 @@ function mapDecoded( } if (cid !== code) { const fromCidAsToUnicode = font.cmap.get(cid); - if (fromCidAsToUnicode !== undefined) { + if (fromCidAsToUnicode !== undefined && cmapValueUsable(fromCidAsToUnicode)) { return { ch: stripInvisibles(fromCidAsToUnicode), code: cid, mapped: true }; } } const fromAdobe = font.cidToUnicode.get(cid); - if (fromAdobe !== undefined) { + if (fromAdobe !== undefined && cmapValueUsable(fromAdobe)) { return { ch: stripInvisibles(fromAdobe), code: cid, mapped: true }; } if (code >= 32 && code < 127 && !font.isCid) { @@ -2123,15 +2134,162 @@ function decodeFontBytes(font: FontInfo, raw: Uint8Array): DecodedChar[] { return out; } +function markedIsArtifact(op: string, args: PdfValue[]): boolean { + if (op === 'BMC') return nameOf(args[args.length - 1]) === '/Artifact'; + const tag = args.length >= 2 ? args[args.length - 2] : args[args.length - 1]; + return nameOf(tag) === '/Artifact'; +} + function trackMarkedArtifact(stack: boolean[], op: string, args: PdfValue[]): void { - if (op === 'BMC') { - stack.push(nameOf(args[args.length - 1]) === '/Artifact'); - } else if (op === 'BDC') { - const tag = args.length >= 2 ? args[args.length - 2] : args[args.length - 1]; - stack.push(nameOf(tag) === '/Artifact'); - } else if (op === 'EMC') { + if (op === 'EMC') { stack.pop(); + } else if (op === 'BMC' || op === 'BDC') { + stack.push(markedIsArtifact(op, args)); + } +} + +interface MarkedCapture { + x: number; + y: number; + widthAcc: number; + height: number; + font: string; + fontSize: number; + isBold: boolean; + isItalic: boolean; + dx: number; + dy: number; +} + +interface StructActual { + text: string; + emitted: boolean; + capture?: MarkedCapture; +} + +interface MarkedFrame { + artifact: boolean; + actual?: StructActual; +} + +function markedString(v: PdfValue | undefined): string | undefined { + const text = stripInvisibles(pdfText(v) ?? ''); + return text.length > 0 ? text : undefined; +} + +function resolveBdcDict( + doc: PdfDocument, + resources: PdfValue | undefined, + args: PdfValue[], +): PdfDict | undefined { + if (args.length < 2) return undefined; + const props = args[args.length - 1]; + if (isDict(props)) return props; + const key = nameOf(props); + if (!key) return undefined; + const res = isDict(resources) ? resources : undefined; + const propRes = dictGet(doc, res, '/Properties'); + if (!isDict(propRes)) return undefined; + const got = dictGet(doc, propRes, key); + return isDict(got) ? got : undefined; +} + +function openMarkedFrame( + doc: PdfDocument, + resources: PdfValue | undefined, + structActual: Map, + op: string, + args: PdfValue[], +): MarkedFrame { + const artifact = markedIsArtifact(op, args); + if (op !== 'BDC') return { artifact }; + const props = resolveBdcDict(doc, resources, args); + if (!props) return { artifact }; + const direct = markedString(dictGet(doc, props, '/ActualText')); + if (direct) return { artifact, actual: { text: direct, emitted: false } }; + const mcid = asNumber(dictGet(doc, props, '/MCID')); + if (mcid === undefined) return { artifact }; + const fromStruct = structActual.get(mcid); + if (fromStruct) return { artifact, actual: fromStruct }; + return { artifact }; +} + +function findActualFrame(stack: MarkedFrame[]): MarkedFrame | undefined { + for (let i = stack.length - 1; i >= 0; i -= 1) { + if (stack[i]!.actual) return stack[i]; + } + return undefined; +} + +function loadStructActualText(doc: PdfDocument, page: PdfDict): Map { + const out = new Map(); + const root = deref(doc, doc.trailer.map.get('/Root')); + if (!isDict(root)) return out; + const structRoot = dictGet(doc, root, '/StructTreeRoot'); + if (!isDict(structRoot)) return out; + const seen = new Set(); + walkStructActual(doc, dictGet(doc, structRoot, '/K'), page, undefined, undefined, out, seen); + return out; +} + +function bindStructActual( + out: Map, + mcid: number, + actual: string | undefined, + matchesPage: boolean, + group?: { bind: StructActual }, +): void { + if (!actual || !matchesPage) return; + if (group) { + out.set(mcid, group.bind); + return; + } + out.set(mcid, { text: actual, emitted: false }); +} + +function walkStructActual( + doc: PdfDocument, + node: PdfValue | undefined, + page: PdfDict, + inheritedPage: PdfDict | undefined, + inheritedActual: string | undefined, + out: Map, + seen: Set, + group?: { bind: StructActual }, +): void { + if (node === undefined) return; + if (typeof node === 'number' && Number.isFinite(node)) { + bindStructActual( + out, + Math.trunc(node), + inheritedActual, + inheritedPage === undefined || inheritedPage === page, + group, + ); + return; + } + if (Array.isArray(node)) { + for (const item of node) { + walkStructActual(doc, item, page, inheritedPage, inheritedActual, out, seen, group); + } + return; } + const dict = isDict(node) ? node : deref(doc, node); + if (!isDict(dict) || seen.has(dict)) return; + seen.add(dict); + const pg = dictGet(doc, dict, '/Pg'); + const thisPage = isDict(pg) ? pg : inheritedPage; + const own = markedString(dictGet(doc, dict, '/ActualText')); + const actual = own ?? inheritedActual; + const nextGroup = own ? { bind: { text: own, emitted: false } } : group; + if (nameOf(dictGet(doc, dict, '/Type')) === '/MCR') { + const mcid = asNumber(dictGet(doc, dict, '/MCID')); + if (mcid !== undefined) { + bindStructActual(out, mcid, actual, thisPage === undefined || thisPage === page, nextGroup); + } + return; + } + walkStructActual(doc, dictGet(doc, dict, '/K'), page, thisPage, actual, out, seen, nextGroup); } type ClipState = @@ -2272,19 +2430,57 @@ function extractPage( let rise = 0; let path: [number, number][] = []; let pathStart: [number, number] | undefined; - const marked: boolean[] = []; + const marked: MarkedFrame[] = []; const stats: DecodeStats = { mapped: 0, unmapped: 0 }; const args: PdfValue[] = []; const pageResources = dictGet(doc, page, '/Resources'); const xobjects = loadXObjects(doc, isDict(pageResources) ? pageResources : undefined); const formFonts = fonts; + const structActual = loadStructActualText(doc, page); + const inArtifact = (): boolean => marked.some((frame) => frame.artifact); const emitText = (raw: Uint8Array): void => { if (!font) return; const decoded = decodeFontBytes(font, raw); - countDecoded(stats, decoded); const dirMat = mulMat(tm, ctm); const rendered = Math.abs(fontSize) * matrixScale(dirMat); + const actual = findActualFrame(marked)?.actual; + if (actual) { + stats.mapped += 1; + const take = (x: number, y: number, w: number): void => { + if (!actual.capture) { + actual.capture = { + x, + y, + widthAcc: 0, + height: rendered, + font: font!.name, + fontSize: rendered, + isBold: font!.bold, + isItalic: font!.italic, + dx: dirMat[0]!, + dy: dirMat[1]!, + }; + } + actual.capture.widthAcc += w; + }; + if (decoded.length === 0) { + const [x, y] = applyMat(dirMat, 0, rise); + take(x, y, 0); + return; + } + for (const { ch, code } of decoded) { + let w = (font.widths.get(code) ?? font.defaultWidth) * font.unitsScale * fontSize; + if (ch === ' ') w += wordSpace; + w = (w + charSpace) * hscale; + const trm = mulMat(tm, ctm); + const [x, y] = applyMat(trm, 0, rise); + take(x, y, w); + tm = translateTextMatrix(tm, w, 0); + } + return; + } + countDecoded(stats, decoded); let buf = ''; let startX: number | undefined; let startY = 0; @@ -2431,7 +2627,41 @@ function extractPage( } } } else if (op === 'BMC' || op === 'BDC' || op === 'EMC') { - trackMarkedArtifact(marked, op, args); + if (op === 'EMC') { + const frame = marked.pop(); + const actual = frame?.actual; + if ( + actual && + !actual.emitted && + actual.capture && + !marked.some((open) => open.actual === actual) + ) { + const cap = actual.capture; + const width = pageAdvanceX(cap.widthAcc, tm, ctm); + const h = cap.height || 1; + if (overlapsClip(clip, cap.x, cap.y - h, cap.x + width, cap.y + h)) { + actual.emitted = true; + items.push({ + text: actual.text, + x: cap.x, + y: cap.y, + width, + height: cap.height, + font: cap.font, + fontSize: cap.fontSize, + page: pageNo, + isBold: cap.isBold, + isItalic: cap.isItalic, + isUnderline: false, + isStrikeout: false, + dx: cap.dx, + dy: cap.dy, + }); + } + } + } else { + marked.push(openMarkedFrame(doc, pageResources, structActual, op, args)); + } } else if (op === 'm' && args.length >= 2) { const p = applyMat(ctm, lastNum(2), lastNum(1)); path = [p]; @@ -2449,7 +2679,7 @@ function extractPage( const p3 = applyMat(ctm, x, y + h); path = [p0, p1, p2, p3, p0]; pathStart = p0; - if (!marked.includes(true)) { + if (!inArtifact()) { const xs = [p0[0], p1[0], p2[0], p3[0]]; const ys = [p0[1], p1[1], p2[1], p3[1]]; const minX = Math.min(...xs); @@ -2473,7 +2703,7 @@ function extractPage( if (box) clip = intersectClip(clip, box); } else if (op === 'S' || op === 's') { if (op === 's' && pathStart) path.push(pathStart); - if (!marked.includes(true)) { + if (!inArtifact()) { for (let i = 1; i < path.length; i += 1) { const a = path[i - 1]!; const b = path[i]!; @@ -2495,7 +2725,7 @@ function extractPage( typeof args[args.length - 1] === 'string' ? (args[args.length - 1] as string) : ''; const xobj = xobjects.get(name); if (xobj !== undefined) { - const inArtifact = marked.includes(true); + const skipRules = inArtifact(); extractFormText( doc, xobj.dict, @@ -2503,8 +2733,8 @@ function extractPage( ctm, pageNo, items, - inArtifact ? [] : lines, - inArtifact ? [] : rects, + skipRules ? [] : lines, + skipRules ? [] : rects, stats, depth, clip, @@ -2520,7 +2750,7 @@ function extractPage( op === 'b' || op === 'b*' ) { - if (op !== 'n' && !marked.includes(true)) { + if (op !== 'n' && !inArtifact()) { if (op === 'b' || op === 'b*') { if (pathStart) path.push(pathStart); } diff --git a/packages/pdf/test/encoding.test.ts b/packages/pdf/test/encoding.test.ts index 0f0b5a8..8da815f 100644 --- a/packages/pdf/test/encoding.test.ts +++ b/packages/pdf/test/encoding.test.ts @@ -301,3 +301,279 @@ ET expect(md).not.toContain('ayoung'); }); }); + +describe('ActualText and ToUnicode FFFD fallback', () => { + function objectsPdf(objects: string[]): Uint8Array { + let body = '%PDF-1.4\n'; + const offsets = [0]; + for (const obj of objects) { + offsets.push(body.length); + body += obj; + } + const xrefAt = body.length; + let xref = `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`; + for (let i = 1; i <= objects.length; i += 1) { + xref += `${String(offsets[i]).padStart(10, '0')} 00000 n \n`; + } + body += `${xref}trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xrefAt}\n%%EOF\n`; + return new TextEncoder().encode(body); + } + + const fffdToUnicode = `%!PS-Adobe-3.0 Resource-CMap +/CIDInit /ProcSet findresource begin +12 dict begin +begincmap +/CIDSystemInfo << /Registry (Adobe) /Ordering (UCS) /Supplement 0 >> def +/CMapName /Adobe-Identity-UCS def +/CMapType 2 def +1 begincodespacerange +<00> +endcodespacerange +3 beginbfchar +<0d> <0033> +<13> +<0a> <0034> +endbfchar +endcmap +CMapName currentdict /CMap defineresource pop +end +end +`; + + it('falls back to Differences names when ToUnicode is U+FFFD', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(\r\x13\n) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /GKDCHH+Brill-Roman /Encoding 6 0 R /ToUnicode 7 0 R >>\nendobj\n', + '6 0 obj\n<< /Type /Encoding /BaseEncoding /WinAnsiEncoding /Differences [10 /four.SP 13 /three.SP 19 /one.SP] >>\nendobj\n', + `7 0 obj\n<< /Length ${fffdToUnicode.length} >>\nstream\n${fffdToUnicode}endstream\nendobj\n`, + ]), + ); + expect(md).toContain('314'); + expect(md).not.toContain('\uFFFD'); + }); + + it('uses inline ActualText when ToUnicode is U+FFFD', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(\r) Tj +/Span << /ActualText >> BDC +(\x13) Tj +EMC +(\n) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /GKDCHH+Brill-Roman /Encoding /WinAnsiEncoding /ToUnicode 6 0 R >>\nendobj\n', + `6 0 obj\n<< /Length ${fffdToUnicode.length} >>\nstream\n${fffdToUnicode}endstream\nendobj\n`, + ]), + ); + expect(md).toContain('314'); + expect(md).not.toContain('\uFFFD'); + }); + + it('uses named Properties ActualText', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(\r) Tj +/Span /AT1 BDC +(\x13) Tj +EMC +(\n) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> /Properties << /AT1 7 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /GKDCHH+Brill-Roman /Encoding /WinAnsiEncoding /ToUnicode 6 0 R >>\nendobj\n', + `6 0 obj\n<< /Length ${fffdToUnicode.length} >>\nstream\n${fffdToUnicode}endstream\nendobj\n`, + '7 0 obj\n<< /ActualText >>\nendobj\n', + ]), + ); + expect(md).toContain('314'); + }); + + it('uses StructTreeRoot ActualText for an MCID', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(\r) Tj +/Span << /MCID 0 >> BDC +(\x13) Tj +EMC +(\n) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 8 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /GKDCHH+Brill-Roman /Encoding /WinAnsiEncoding /ToUnicode 6 0 R >>\nendobj\n', + `6 0 obj\n<< /Length ${fffdToUnicode.length} >>\nstream\n${fffdToUnicode}endstream\nendobj\n`, + '7 0 obj\n<< >>\nendobj\n', + '8 0 obj\n<< /Type /StructTreeRoot /K [9 0 R] >>\nendobj\n', + '9 0 obj\n<< /Type /StructElem /S /Span /P 8 0 R /Pg 3 0 R /K 0 /ActualText >>\nendobj\n', + ]), + ); + expect(md).toContain('314'); + }); + + it('uses parent StructTreeRoot ActualText once across child MCIDs', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(3) Tj +/Span << /MCID 0 >> BDC +(x) Tj +EMC +/Span << /MCID 1 >> BDC +(y) Tj +EMC +(4) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 6 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>\nendobj\n', + '6 0 obj\n<< /Type /StructTreeRoot /K [7 0 R] >>\nendobj\n', + '7 0 obj\n<< /Type /StructElem /S /Span /P 6 0 R /Pg 3 0 R /K [0 1] /ActualText >>\nendobj\n', + ]), + ); + expect(md).toContain('314'); + expect(md).not.toContain('311'); + expect(md).not.toContain('x'); + expect(md).not.toContain('y'); + }); + + it('uses parent StructTreeRoot ActualText when the first MCID has no text', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(3) Tj +/Span << /MCID 0 >> BDC +EMC +/Span << /MCID 1 >> BDC +(x) Tj +EMC +(4) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 6 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>\nendobj\n', + '6 0 obj\n<< /Type /StructTreeRoot /K [7 0 R] >>\nendobj\n', + '7 0 obj\n<< /Type /StructElem /S /Span /P 6 0 R /Pg 3 0 R /K [0 1] /ActualText >>\nendobj\n', + ]), + ); + expect(md).toContain('314'); + expect(md).not.toContain('x'); + }); + + it('uses parent StructTreeRoot ActualText for nested child MCIDs', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(3) Tj +/Span << /MCID 0 >> BDC +/Span << /MCID 1 >> BDC +(x) Tj +EMC +EMC +(4) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 6 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>\nendobj\n', + '6 0 obj\n<< /Type /StructTreeRoot /K [7 0 R] >>\nendobj\n', + '7 0 obj\n<< /Type /StructElem /S /Span /P 6 0 R /Pg 3 0 R /K [0 1] /ActualText >>\nendobj\n', + ]), + ); + expect(md).toContain('314'); + expect(md).not.toContain('x'); + }); + + it('uses parent ActualText from the outer nested MCID span', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 80 Tm +(3) Tj +/Span << /MCID 0 >> BDC +(A) Tj +1 0 0 1 20 10 Tm +/Span << /MCID 1 >> BDC +(x) Tj +EMC +EMC +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R /StructTreeRoot 6 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>\nendobj\n', + '6 0 obj\n<< /Type /StructTreeRoot /K [7 0 R] >>\nendobj\n', + '7 0 obj\n<< /Type /StructElem /S /Span /P 6 0 R /Pg 3 0 R /K [0 1] /ActualText >>\nendobj\n', + ]), + ); + expect(md).toContain('31'); + expect(md).not.toMatch(/3\s*\n/); + expect(md).not.toContain('x'); + expect(md).not.toContain('A'); + }); + + it('maps small-cap Differences names to the printed letters', () => { + const content = `BT +/F1 12 Tf +1 0 0 1 20 50 Tm +(Yarrow) Tj +ET +`; + const md = toMarkdownFromPdf( + objectsPdf([ + '1 0 obj\n<< /Type /Catalog /Pages 2 0 R >>\nendobj\n', + '2 0 obj\n<< /Type /Pages /Kids [3 0 R] /Count 1 >>\nendobj\n', + '3 0 obj\n<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 100] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>\nendobj\n', + `4 0 obj\n<< /Length ${content.length} >>\nstream\n${content}endstream\nendobj\n`, + '5 0 obj\n<< /Type /Font /Subtype /Type1 /BaseFont /GKDCHH+Brill-Roman /Encoding 6 0 R >>\nendobj\n', + '6 0 obj\n<< /Type /Encoding /BaseEncoding /WinAnsiEncoding /Differences [89 /Y.c2sc 97 /a.smcp 111 /o.smcp 114 /r.smcp 119 /w.smcp] >>\nendobj\n', + ]), + ); + expect(md).toContain('YARROW'); + }); +}); diff --git a/packages/pdf/test/layout.test.ts b/packages/pdf/test/layout.test.ts index ad14dc6..63511f2 100644 --- a/packages/pdf/test/layout.test.ts +++ b/packages/pdf/test/layout.test.ts @@ -311,6 +311,23 @@ EMC expect(md).toContain('a b c d e f g'); expect(md).not.toMatch(/\|---\|/); }); + + it('uses marked-content ActualText instead of the shown glyphs', () => { + const md = toMarkdownFromPdf( + pagePdf(`BT +/F1 12 Tf +1 0 0 1 20 90 Tm +(3) Tj +/Span << /ActualText >> BDC +(x) Tj +EMC +(4) Tj +ET +`), + ); + expect(md).toContain('314'); + expect(md).not.toContain('3x4'); + }); }); describe('paragraph breaks', () => {