diff --git a/README.md b/README.md
index 6375d4c..1c35f15 100644
--- a/README.md
+++ b/README.md
@@ -8,7 +8,7 @@ Tagging runs offline and makes no network or model calls. An optional `review` a
## Install
-Node 24 or later. Tesseract 5 is optional; it is needed only for scanned pages.
+Node 24 or later. Tesseract 5 is optional. It places a scan's text over its image; without it, a scan is still tagged, its text placed approximately.
```sh
git clone https://github.com/EqualifyEverything/equalify-iris-pdf && cd equalify-iris-pdf
@@ -60,7 +60,7 @@ Text fields take strings, checkboxes `true`/`false`, radio groups and lists one
## How it works
1. The page's original drawing is kept byte for byte and marked as an artifact. A PDF that is already tagged loses its old tags first (warning `retagged`); each retag adds to the file's size.
-2. Iris's words are matched to the words on the page (from the text layer, or from Tesseract on a scan).
+2. Iris's words are matched to the words on the page (from the text layer, or from Tesseract on a scan). With neither, the words flow down the page in reading order.
3. An invisible text layer is added with Iris's words at those positions, tagged with the structure from the HTML: headings, lists, tables with their headers, links, figures with alt text, form fields.
4. The file is saved incrementally: the original bytes are the start of the output. A damaged file is instead rewritten from mupdf's repair of it, with warning `repaired`.
@@ -73,7 +73,7 @@ Then two checks run, and if either fails nothing is written (exit 2):
A page that could not be tagged is left as it was, with a warning, and the run still exits 0. Read the report to catch it; the output then makes no PDF/UA-1 claim.
-`--report` writes JSON: per page, where the text came from and how many words matched; the structure written; fields set and skipped; the check results; and every warning. Warnings name what could not be done, for example `unmatched_text` (page text missing from the HTML, kept as a paragraph), `missing_alt`, `field_not_in_html`, `unmatched_link`, `duplicate_text_layer`, `page_not_in_html` (a blank page needs no HTML and is not warned), `page_not_tagged`, `no_text_positions` (a scan, with Tesseract missing or failing), `no_title`, `font_not_embedded` (a source font has no embedded program, which PDF/UA-1 requires; the source drawing is not changed), `source_marked_content` (the page drawing has marked-content ids left from an earlier tag tree), `alignment_incomplete` (the page and the HTML differ too much to match every word in time; the rest is kept as unmatched text).
+`--report` writes JSON: per page, where the text came from and how many words matched; the structure written; fields set and skipped; the check results; and every warning. Warnings name what could not be done, for example `unmatched_text` (page text missing from the HTML, kept as a paragraph), `missing_alt`, `field_not_in_html`, `unmatched_link`, `duplicate_text_layer`, `page_not_in_html` (a blank page needs no HTML and is not warned), `page_not_tagged`, `no_text_positions` (a scan, with Tesseract missing or failing: tagged, but its text does not line up with its image), `no_title`, `font_not_embedded` (a source font has no embedded program, which PDF/UA-1 requires; the source drawing is not changed), `source_marked_content` (the page drawing has marked-content ids left from an earlier tag tree), `alignment_incomplete` (the page and the HTML differ too much to match every word in time; the rest is kept as unmatched text).
## Review
diff --git a/src/html/build.ts b/src/html/build.ts
index 4f3487d..0fc5fd1 100644
--- a/src/html/build.ts
+++ b/src/html/build.ts
@@ -143,6 +143,7 @@ function build(e: Elem | string, parent: Node, ctx: Ctx) {
}
const type = STRUCT[e.tag];
if (!type) ctx.warn({ code: "unmapped_element", detail: e.tag });
+ if (type === "LBody") { const w = lastWord(parent); if (w) w.space = true; } // after its
kids(add(type ?? "P"), inner);
}
@@ -214,6 +215,7 @@ function tableHeaders(table: Node, prefix: string) {
function addText(parent: Node, text: string) {
const words = splitWords(text).map((w) => ({ ...word(w.text), space: w.space }));
+ if (/^\s/.test(text)) { const w = lastWord(parent); if (w) w.space = true; } // "form for"
if (!words.length) return;
const last = parent.kids.at(-1);
if (last && isRun(last)) last.words.push(...words);
@@ -221,6 +223,7 @@ function addText(parent: Node, text: string) {
}
const word = (text: string): Word => ({ text, norm: normalize(text) });
+const lastWord = (n: Node | Run): Word | undefined => (isRun(n) ? n.words.at(-1) : n.kids.length ? lastWord(n.kids.at(-1)!) : undefined);
const clean = (s: string) => s.replace(/\s+/g, " ").trim();
function walk(e: Elem, fn: (e: Elem) => void) {
@@ -229,11 +232,16 @@ function walk(e: Elem, fn: (e: Elem) => void) {
}
// Every word in reading order, each with the index of its top-level block.
-export function wordsInOrder(top: Node): { word: Word; block: number }[] {
- const out: { word: Word; block: number }[] = [];
+// newLine: the word is the first in, or after, a block element.
+const INLINE = new Set(["Link", "Reference", "Code", "Lbl", "LBody"]);
+export function wordsInOrder(top: Node): { word: Word; block: number; newLine: boolean }[] {
+ const out: { word: Word; block: number; newLine: boolean }[] = [];
+ let newLine = true;
const visit = (n: Node | Run, block: number) => {
- if (isRun(n)) n.words.forEach((word) => out.push({ word, block }));
- else n.kids.forEach((k) => visit(k, block));
+ if (isRun(n)) return n.words.forEach((word) => (out.push({ word, block, newLine }), (newLine = false)));
+ if (!INLINE.has(n.type)) newLine = true;
+ n.kids.forEach((k) => visit(k, block));
+ if (!INLINE.has(n.type)) newLine = true;
};
top.kids.forEach((k, i) => visit(k, i));
return out;
diff --git a/src/report.ts b/src/report.ts
index 50e7a2b..4f74422 100644
--- a/src/report.ts
+++ b/src/report.ts
@@ -7,7 +7,7 @@ export type Warning = { code: string; page?: number; detail?: string };
export type PageReport = {
page: number;
- textSource: "pdf-text" | "ocr" | "none";
+ textSource: "pdf-text" | "ocr" | "approximate" | "none";
words: number;
matched: number;
addedFromHtml: number;
diff --git a/src/tag.ts b/src/tag.ts
index 4bb29bb..acf7ad8 100644
--- a/src/tag.ts
+++ b/src/tag.ts
@@ -152,11 +152,13 @@ function tagPage(page: mupdf.PDFPage, i: number, html: string, ctx: PageCtx): {
else if (ordered.length) {
const ocr = ocrWords(page);
if (typeof ocr === "string") {
- warn({ code: "no_text_positions", detail: `The page has no text layer and ${ocr}; it was left untagged.` });
- return { report, overlay: null, untagged: true };
+ // Still tagged: the words flow down the page in reading order, not over their image.
+ warn({ code: "no_text_positions", detail: `The page has no text layer and ${ocr}; its words are placed approximately.` });
+ report.textSource = "approximate";
+ } else {
+ words = ocr;
+ report.textSource = "ocr";
}
- words = ocr;
- report.textSource = "ocr";
}
// Text drawn inside a form field belongs to the field, which is tagged by reference.
const inField = (w: PageWord) => ctx.widgets.some((f) => overlaps(f.box, w.box, 0.5));
@@ -180,7 +182,9 @@ function tagPage(page: mupdf.PDFPage, i: number, html: string, ctx: PageCtx): {
blockOf.set(t, ordered[k].block);
});
const bounds = page.getBounds();
- fillPositions(ordered.map((o) => o.word), bounds, (s) => ctx.fonts.width(s));
+ // With no page words, each block starts a new line, so blocks do not run together.
+ const breaks = new Set(words.length ? [] : ordered.filter((o, k) => k && o.newLine).map((o) => o.word));
+ fillPositions(ordered.map((o) => o.word), bounds, (s) => ctx.fonts.width(s), breaks);
// Page words Iris left out: furniture, or lost content reported and kept as a P.
const lost = new Map(); // after which top-level block
@@ -247,8 +251,8 @@ const placed = (w: PageWord): Placed => ({ box: w.box, baseline: w.baseline, siz
// An HTML word with no page word follows the word before it, at its natural
// width, wrapping at the page edge: extractors drop text off the page, and
// merge repeated letters piled into one spot. With no word before it, it
-// starts at the first placed word, or the page's top left.
-function fillPositions(words: Word[], page: Box, width: (text: string) => number) {
+// starts at the first placed word, or the page's top left. A word in `breaks` starts a new line.
+function fillPositions(words: Word[], page: Box, width: (text: string) => number, breaks: Set) {
const fallback: Placed = { box: [page[0] + 10, page[1] + 10, page[0] + 10, page[1] + 20], baseline: page[1] + 20, size: 10 };
const first = words.find((w) => w.at)?.at ?? fallback;
let prev: Word | null = null;
@@ -258,8 +262,8 @@ function fillPositions(words: Word[], page: Box, width: (text: string) => number
const wide = Math.min(width(w.text) * size, page[2] - page[0] - 2);
let x = prev ? from.box[2] + (prev.space ?? true ? width(" ") * size : 0) : first.box[0];
let dy = 0;
- if (x + wide > page[2] - 1) {
- x = page[0] + 1;
+ if (breaks.has(w) || x + wide > page[2] - 1) {
+ x = breaks.has(w) && first.box[0] + wide <= page[2] - 1 ? first.box[0] : page[0] + 1;
dy = from.box[3] + line > page[3] ? page[1] + line - from.box[3] : line; // off the bottom: back to the top
}
w.at = { size, baseline: from.baseline + dy, box: [x, from.box[1] + dy, x + wide, from.box[3] + dy] };
diff --git a/test/cli.test.ts b/test/cli.test.ts
index 4a47770..258bc3b 100644
--- a/test/cli.test.ts
+++ b/test/cli.test.ts
@@ -41,7 +41,7 @@ test("a tagged PDF is retagged", () => {
assert.equal(run(...again).code, 0);
});
-test("with Tesseract missing or failing, a scan is left untagged with a warning", () => {
+test("with Tesseract missing or failing, a scan is still tagged, its words placed approximately", () => {
const failing = mkdtempSync(join(tmpdir(), "iris-pdf-bin-"));
writeFileSync(join(failing, "tesseract"), '#!/bin/sh\n[ "$1" = --version ] && exit 0\necho no eng >&2; exit 1\n', { mode: 0o755 });
for (const [PATH, why] of [["", /not installed/], [failing, /failed: no eng/]] as const) {
@@ -49,11 +49,25 @@ test("with Tesseract missing or failing, a scan is left untagged with a warning"
const r = spawnSync(process.execPath, [cli, ...tagArgs("mixed", out, "--report", report)], { encoding: "utf8", env: { PATH } });
assert.equal(r.status, 0, r.stderr);
const json = JSON.parse(readFileSync(report, "utf8"));
- assert.deepEqual(json.pages.map((p: { textSource: string }) => p.textSource), ["pdf-text", "none"]);
+ assert.deepEqual(json.pages.map((p: { textSource: string }) => p.textSource), ["pdf-text", "approximate"]);
assert.ok(json.warnings.some((w: { code: string; page?: number; detail: string }) => w.code === "no_text_positions" && w.page === 2 && why.test(w.detail)));
+ assert.ok(json.pages[1].mcids >= 2);
+ assert.match(new mupdf.PDFDocument(readFileSync(out)).loadPage(1).toStructuredText("").asText(), /Office Address\s+The permit office is on Main Street\./);
}
});
+test("approximately placed blocks do not run together, nested or not", () => {
+ const pages = join(dir, "nested.json"), out = join(dir, "nested.pdf");
+ const html = "Fees
" +
+ "See the form for tag.
- Permit
- A paper.
" +
+ `${"x".repeat(150)}
`; // wider than the page
+ writeFileSync(pages, JSON.stringify({ lang: "en", title: "Fees", pages: [{ sourcePage: 1, html }] }));
+ const r = spawnSync(process.execPath, [cli, "tag", "--pdf", fixture("scan-300dpi.pdf"), "--pages", pages, "--out", out], { encoding: "utf8", env: { PATH: "" } });
+ assert.equal(r.status, 0, r.stderr);
+ const lines = new mupdf.PDFDocument(readFileSync(out)).loadPage(0).toStructuredText("").asText().split("\n").filter(Boolean);
+ assert.deepEqual(lines, ["Fees", "• One link", "• Two", "Year", "2024", "See the form for tag.", "Permit A paper.", "x".repeat(150)]);
+});
+
test("a failed verification exits 2 and writes no PDF", () => {
// A stream ending inside a string swallows the overlay (see tag.test.ts).
const doc = new mupdf.PDFDocument(readFileSync(fixture("text-simple.pdf")));
diff --git a/test/pdfua.test.ts b/test/pdfua.test.ts
index 1c3538a..4fb652c 100644
--- a/test/pdfua.test.ts
+++ b/test/pdfua.test.ts
@@ -3,12 +3,12 @@
import { test } from "node:test";
import assert from "node:assert/strict";
import { spawnSync } from "node:child_process";
-import { mkdtempSync, writeFileSync } from "node:fs";
+import { mkdtempSync, readFileSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { tesseractInstalled } from "../src/ocr/tesseract.ts";
import { checkPdfUa } from "../src/verify/pdfua.ts";
-import { tagFixture } from "./helpers.ts";
+import { fixture, tagFixture } from "./helpers.ts";
const noVera = !process.env.IRIS_REQUIRE_VERAPDF && !!spawnSync("verapdf", ["--version"]).error && "veraPDF is not installed";
const noOcr = !tesseractInstalled() && "Tesseract is not installed";
@@ -21,6 +21,16 @@ const CORPUS: [string, boolean][] = [
];
const SCANS = new Set(["scan-300dpi", "scan-skewed", "mixed"]);
+test("a scan tagged without Tesseract claims PDF/UA-1 and passes veraPDF", { skip: noVera }, () => {
+ const path = join(dir, "scan-approx.pdf");
+ const cli = new URL("../src/cli.ts", import.meta.url).pathname;
+ const r = spawnSync(process.execPath, [cli, "tag", "--pdf", fixture("scan-300dpi.pdf"), "--pages", fixture("scan-300dpi.pages.json"), "--out", path], { encoding: "utf8", env: { PATH: "" } });
+ assert.equal(r.status, 0, r.stderr);
+ assert.match(readFileSync(path, "latin1"), /1);
+ const { passed, message } = checkPdfUa(path);
+ assert.equal(passed, true, message);
+});
+
for (const [name, conforms] of CORPUS) {
test(`${name}: ${conforms ? "claims PDF/UA-1 and passes veraPDF" : "does not claim PDF/UA-1"}`, { skip: noVera || (SCANS.has(name) && noOcr) }, () => {
const { out, doc } = tagFixture(name);