From d590acdf9932baa3112e8cf7a0a66ae5cb76c55e Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 10:16:41 -0400 Subject: [PATCH 1/5] fix(links): a link target lengthened to its printed URL is not a drop (#503) In production, 9 of the 10 hrefs behind links_dropped_rate were one document's PDF link targets cut at a line wrap, which the copy editor replaced with the full printed URL. droppedHrefs now leaves those out, and the round logs them as editor_links_completed. Co-Authored-By: Claude Opus 5.5 --- docs/API.md | 12 ++++++++++-- src/pipeline/links.ts | 22 +++++++++++++++++++++- src/pipeline/review.ts | 4 +++- test/editor-sections.test.ts | 15 +++++++++++++++ test/pdf-links.test.ts | 19 +++++++++++++++++++ 5 files changed, 68 insertions(+), 4 deletions(-) diff --git a/docs/API.md b/docs/API.md index 1fc61eaa..b10f392b 100644 --- a/docs/API.md +++ b/docs/API.md @@ -1062,8 +1062,8 @@ The events worth grepping for have a section each below, and the index is a link the index when you have a `type` off a log line and want to know what it means; read a section when you want to know what the field it names is for and what it costs. -**The index is the whole log.** `src/` emits **124** event types and every one of them has a section -below — **116** sections, because a few cover two or three events that are only read together. So a +**The index is the whole log.** `src/` emits **125** event types and every one of them has a section +below — **117** sections, because a few cover two or three events that are only read together. So a `type` you cannot find here is not one the index skipped: it is a misread line, or a name `src/` no longer emits. @@ -1137,6 +1137,7 @@ emits fails it too. | [`editor_images_refused`](#editor_images_refused) | The payload was refused as too large, so it was re-sent **without** images | | [`editor_fidelity_observed`](#editor_fidelity_observed) | The Copy Editor reports a disagreement **nobody asked it about** | | [`editor_links_dropped`](#editor_links_dropped) | An `href` present before that round's correction was missing after it | +| [`editor_links_completed`](#editor_links_completed) | The Copy Editor lengthened an `href` to the URL its link prints | | [`internal_links`](#internal_links) | The delivered document has an in-document reference that lands nowhere | | [`delivered_markup`](#delivered_markup) | The delivered document's own structure disagrees with itself | | [`delivered_structure`](#delivered_structure) | Four structural defects **no rule in the gate reports** | @@ -3090,6 +3091,13 @@ An `href` present before that round's correction was missing after it (`iteratio link's target came from the source **file**, not from a page image, so a dropped one cannot be recovered by looking again — logged rather than repaired, and counted into `links_dropped_rate`. +### `editor_links_completed` + +The Copy Editor replaced an `href` with a longer URL that starts with it and is the link's printed +text (`iteration`, `hrefs`, the old URLs). This happens when a PDF's link target is cut where the +printed URL wraps onto a second line. It is a repair, so it is not in `editor_links_dropped` or +`links_dropped_rate`. + ### `internal_links` The delivered document contains an in-document reference that lands nowhere (`refs` fragment links diff --git a/src/pipeline/links.ts b/src/pipeline/links.ts index bd2c7136..a052bf22 100644 --- a/src/pipeline/links.ts +++ b/src/pipeline/links.ts @@ -162,9 +162,29 @@ export function missingLinkProblem(link: PdfLink): string { // anchors.ts renames colliding ids as pages are joined, and the editor renumbers // footnotes when it fixes their structure — so including them would report ordinary // work as loss and bury the case that matters. +// +// A URL the rewrite lengthened to the one its link prints is not counted: see `completedHrefs`. export function droppedHrefs(before: string, after: string): string[] { const kept = hrefsIn(after); - return [...hrefsIn(before)].filter((h) => isAbsolute(h) && !kept.has(h)).sort(); + const completed = new Set(completedHrefs(before, after)); + return [...hrefsIn(before)].filter((h) => isAbsolute(h) && !kept.has(h) && !completed.has(h)).sort(); +} + +// Absolute URLs a rewrite replaced with a longer one that starts with it and is the link's own +// printed text (#503). A URL that wraps onto a second line in a PDF can carry a link target cut at +// the wrap, and the editor, which sees the page, writes the whole printed URL. The link then goes +// where the page says, so it is a repair and not a loss. +export function completedHrefs(before: string, after: string): string[] { + const kept = hrefsIn(after); + const printed: string[] = []; + for (const m of after.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)')[^>]*>([\s\S]*?)<\/a>/gi)) { + const href = normalizeHref(m[1] ?? m[2] ?? ""); + const text = normalizeHref(m[3]!.replace(/<[^>]*>/g, "").replace(/\s+/g, "")); + if (href === text) printed.push(href); + } + return [...hrefsIn(before)] + .filter((h) => isAbsolute(h) && !kept.has(h) && printed.some((p) => p.length > h.length && p.startsWith(h))) + .sort(); } // Every in-document reference in the delivered document, and whether it lands (#234). diff --git a/src/pipeline/review.ts b/src/pipeline/review.ts index 896a00c6..b6996e87 100644 --- a/src/pipeline/review.ts +++ b/src/pipeline/review.ts @@ -37,7 +37,7 @@ import { import { flatten } from "./flatten.ts"; import { examplesForPrompt } from "./memory.ts"; import { knownPages, pageIndex, type IndexedPage } from "./pageindex.ts"; -import { droppedHrefs } from "./links.ts"; +import { completedHrefs, droppedHrefs } from "./links.ts"; import { sameWordedHeadingNote, sameWordedHeadingRuns } from "./headings.ts"; export interface ReviewIssue { @@ -3361,6 +3361,8 @@ export async function runReview( droppedLinks += dropped.length; ctx.log.event("editor_links_dropped", { iteration: iterations, hrefs: dropped }); } + const completed = completedHrefs(before, body); + if (completed.length) ctx.log.event("editor_links_completed", { iteration: iterations, hrefs: completed }); // See BODY_MARKERS: the only place a marker's DISAPPEARANCE is recorded. An arrival is also // recorded on the page path, by `markers_added` on `page_corrected` (#373) — additions only, // because that corrector is handed the image and resolving an illegible passage is its job. The diff --git a/test/editor-sections.test.ts b/test/editor-sections.test.ts index d48a13c7..f1c6beed 100644 --- a/test/editor-sections.test.ts +++ b/test/editor-sections.test.ts @@ -620,6 +620,21 @@ test("the round is measured like any other before the loop ends on it", async () }); }); +test("a link target the round lengthened to its printed URL is logged as completed, not dropped (#503)", async () => { + await withTemp(async (dir) => { + const first = `

https://example.com/forms/annual-report.pdf the rest of it

`; + const fixed = `

https://example.com/forms/annual-report.pdf the rest of it

`; + const { ctx, rec } = ctxWith(dir, { + sectionAnswer: (s) => (s.index === 1 ? s.html.replace(first, fixed) : s.html), + }); + const result = await review(ctx, `${first}\n\n${LONG}`); + const completed = rec.events.find((e) => e.type === "editor_links_completed"); + assert.deepEqual(completed?.data, { iteration: 1, hrefs: ["https://example.com/forms/annual"] }); + assert.equal(rec.events.some((e) => e.type === "editor_links_dropped"), false); + assert.equal(result.droppedLinks, 0); + }); +}); + test("a round answered piece by piece is not a round that converged", async () => { await withTemp(async (dir) => { // `review_converged` claims the editor read the whole document, decided it was better left diff --git a/test/pdf-links.test.ts b/test/pdf-links.test.ts index fe954d2e..134d8607 100644 --- a/test/pdf-links.test.ts +++ b/test/pdf-links.test.ts @@ -15,6 +15,7 @@ import { type PdfLink, } from "../src/util/pdf.ts"; import { + completedHrefs, droppedHrefs, missingLinkProblem, missingLinks, @@ -597,6 +598,24 @@ test("a rewrite that loses a link is detectable; one that only renames anchors i assert.deepEqual(droppedHrefs(before, `

the annual report

`), []); }); +test("a target cut at a line wrap, lengthened to the printed URL, is a repair and not a drop (#503)", () => { + // The PDF's link target stops where the printed URL wraps. The editor writes the whole printed URL. + const before = `

https://example.org/forms/annual- +report_2026.pdf

`; + const repaired = `

https://example.org/forms/annual-report_2026.pdf

`; + assert.deepEqual(droppedHrefs(before, repaired), []); + assert.deepEqual(completedHrefs(before, repaired), ["https://example.org/forms/annual"]); + // Longer, but not what the link prints: still a drop. + const other = `

https://example.org/forms/annual-report_2026.pdf

`; + assert.deepEqual(droppedHrefs(before, other), ["https://example.org/forms/annual"]); + assert.deepEqual(completedHrefs(before, other), []); + // The printed URL, but not starting with the old target: still a drop. + const elsewhere = `

https://example.net/x

`; + assert.deepEqual(droppedHrefs(before, elsewhere), ["https://example.org/forms/annual"]); + // Unwrapped to plain text: still a drop. + assert.deepEqual(droppedHrefs(before, `

https://example.org/forms/annual-report_2026.pdf

`), ["https://example.org/forms/annual"]); +}); + // --------------------------------------------------------------------------- // In-document references (#234) // --------------------------------------------------------------------------- From d6b7db47b5e415afe23f81e9469cd5cb91689ed9 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 10:30:06 -0400 Subject: [PATCH 2/5] fix(links): a completing URL must be new, and completes one link (#503) Round 1: a lost link to a site's root was passed off as completed by any full URL on that site the document already linked. The longer URL must now be new in the round, and completes only the longest lost URL it starts with. The event logs {from, to}, unquoted hrefs count, and the links_dropped_rate definition says what it leaves out. Co-Authored-By: Claude Opus 5.5 --- docs/API.md | 5 +++-- src/pipeline/links.ts | 28 ++++++++++++++++++---------- src/pipeline/review.ts | 2 +- test/editor-sections.test.ts | 5 ++++- test/pdf-links.test.ts | 12 +++++++++++- 5 files changed, 37 insertions(+), 15 deletions(-) diff --git a/docs/API.md b/docs/API.md index b10f392b..8a57e09b 100644 --- a/docs/API.md +++ b/docs/API.md @@ -319,7 +319,8 @@ curl -s -H "Authorization: Bearer $IRIS_QUALITY_TOKEN" "$BASE/quality?days=30" `truncated` beside `editor_truncated_rate` and the output ceiling. One threshold over both cannot be set honestly, which is why the weekly report's is still on the mixture and says so. * `links_dropped_rate` — share of documents where an `href` present before the copy editor was - missing after it. + missing after it. A URL the editor lengthened to the link's printed URL is not counted (see + [`editor_links_completed`](#editor_links_completed)). * `links_unresolved_rate` — share of documents that shipped with an in-document reference that lands nowhere: an `href="#"`, or a fragment naming an `id` the delivered document does not contain. Counted per document; the per-reference numbers, and *which* ids failed, are on the @@ -3094,7 +3095,7 @@ recovered by looking again — logged rather than repaired, and counted into `li ### `editor_links_completed` The Copy Editor replaced an `href` with a longer URL that starts with it and is the link's printed -text (`iteration`, `hrefs`, the old URLs). This happens when a PDF's link target is cut where the +text (`iteration`, and `links`, each `{from, to}`). The longer URL must be new in that round. This happens when a PDF's link target is cut where the printed URL wraps onto a second line. It is a repair, so it is not in `editor_links_dropped` or `links_dropped_rate`. diff --git a/src/pipeline/links.ts b/src/pipeline/links.ts index a052bf22..30c02b2c 100644 --- a/src/pipeline/links.ts +++ b/src/pipeline/links.ts @@ -166,7 +166,7 @@ export function missingLinkProblem(link: PdfLink): string { // A URL the rewrite lengthened to the one its link prints is not counted: see `completedHrefs`. export function droppedHrefs(before: string, after: string): string[] { const kept = hrefsIn(after); - const completed = new Set(completedHrefs(before, after)); + const completed = new Set(completedHrefs(before, after).map((c) => c.from)); return [...hrefsIn(before)].filter((h) => isAbsolute(h) && !kept.has(h) && !completed.has(h)).sort(); } @@ -174,17 +174,25 @@ export function droppedHrefs(before: string, after: string): string[] { // printed text (#503). A URL that wraps onto a second line in a PDF can carry a link target cut at // the wrap, and the editor, which sees the page, writes the whole printed URL. The link then goes // where the page says, so it is a repair and not a loss. -export function completedHrefs(before: string, after: string): string[] { +// +// The longer URL must be new in this round, and each one completes only the longest lost URL it +// starts with. Otherwise a lost link to a site's root would count as completed by any full URL on +// that site the document already linked. +export function completedHrefs(before: string, after: string): { from: string; to: string }[] { + const had = hrefsIn(before); const kept = hrefsIn(after); - const printed: string[] = []; - for (const m of after.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)')[^>]*>([\s\S]*?)<\/a>/gi)) { - const href = normalizeHref(m[1] ?? m[2] ?? ""); - const text = normalizeHref(m[3]!.replace(/<[^>]*>/g, "").replace(/\s+/g, "")); - if (href === text) printed.push(href); + const lost = [...had].filter((h) => isAbsolute(h) && !kept.has(h)); + const completed = new Map(); + for (const m of after.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))[^>]*>([\s\S]*?)<\/a>/gi)) { + const to = normalizeHref(m[1] ?? m[2] ?? m[3] ?? ""); + const text = normalizeHref(m[4]!.replace(/<[^>]*>/g, "").replace(/\s+/g, "")); + if (to !== text || had.has(to)) continue; + const from = lost + .filter((h) => to.length > h.length && to.startsWith(h) && !completed.has(h)) + .sort((a, b) => b.length - a.length)[0]; + if (from) completed.set(from, to); } - return [...hrefsIn(before)] - .filter((h) => isAbsolute(h) && !kept.has(h) && printed.some((p) => p.length > h.length && p.startsWith(h))) - .sort(); + return [...completed].map(([from, to]) => ({ from, to })).sort((a, b) => (a.from < b.from ? -1 : 1)); } // Every in-document reference in the delivered document, and whether it lands (#234). diff --git a/src/pipeline/review.ts b/src/pipeline/review.ts index b6996e87..e14277b4 100644 --- a/src/pipeline/review.ts +++ b/src/pipeline/review.ts @@ -3362,7 +3362,7 @@ export async function runReview( ctx.log.event("editor_links_dropped", { iteration: iterations, hrefs: dropped }); } const completed = completedHrefs(before, body); - if (completed.length) ctx.log.event("editor_links_completed", { iteration: iterations, hrefs: completed }); + if (completed.length) ctx.log.event("editor_links_completed", { iteration: iterations, links: completed }); // See BODY_MARKERS: the only place a marker's DISAPPEARANCE is recorded. An arrival is also // recorded on the page path, by `markers_added` on `page_corrected` (#373) — additions only, // because that corrector is handed the image and resolving an illegible passage is its job. The diff --git a/test/editor-sections.test.ts b/test/editor-sections.test.ts index f1c6beed..9ac32e10 100644 --- a/test/editor-sections.test.ts +++ b/test/editor-sections.test.ts @@ -629,7 +629,10 @@ test("a link target the round lengthened to its printed URL is logged as complet }); const result = await review(ctx, `${first}\n\n${LONG}`); const completed = rec.events.find((e) => e.type === "editor_links_completed"); - assert.deepEqual(completed?.data, { iteration: 1, hrefs: ["https://example.com/forms/annual"] }); + assert.deepEqual(completed?.data, { + iteration: 1, + links: [{ from: "https://example.com/forms/annual", to: "https://example.com/forms/annual-report.pdf" }], + }); assert.equal(rec.events.some((e) => e.type === "editor_links_dropped"), false); assert.equal(result.droppedLinks, 0); }); diff --git a/test/pdf-links.test.ts b/test/pdf-links.test.ts index 134d8607..3cf91247 100644 --- a/test/pdf-links.test.ts +++ b/test/pdf-links.test.ts @@ -604,7 +604,11 @@ test("a target cut at a line wrap, lengthened to the printed URL, is a repair an report_2026.pdf

`; const repaired = `

https://example.org/forms/annual-report_2026.pdf

`; assert.deepEqual(droppedHrefs(before, repaired), []); - assert.deepEqual(completedHrefs(before, repaired), ["https://example.org/forms/annual"]); + assert.deepEqual(completedHrefs(before, repaired), [ + { from: "https://example.org/forms/annual", to: "https://example.org/forms/annual-report_2026.pdf" }, + ]); + // Unquoted, as a model sometimes writes it. + assert.deepEqual(droppedHrefs(before, repaired.replace(/href="([^"]*)"/, "href=$1")), []); // Longer, but not what the link prints: still a drop. const other = `

https://example.org/forms/annual-report_2026.pdf

`; assert.deepEqual(droppedHrefs(before, other), ["https://example.org/forms/annual"]); @@ -612,6 +616,12 @@ report_2026.pdf

`; // The printed URL, but not starting with the old target: still a drop. const elsewhere = `

https://example.net/x

`; assert.deepEqual(droppedHrefs(before, elsewhere), ["https://example.org/forms/annual"]); + // A lost link to the site's root is not "completed" by a full URL the document already linked. + const report = `https://example.org/forms/annual-report.pdf`; + assert.deepEqual(droppedHrefs(`

Home ${report}

`, `

Home ${report}

`), ["https://example.org"]); + // A new URL completes only the longest lost URL it starts with. + const both = `

Home https://example.org/forms/annual-report_2026.pdf

`; + assert.deepEqual(droppedHrefs(both, `

Home ${repaired}

`), ["https://example.org"]); // Unwrapped to plain text: still a drop. assert.deepEqual(droppedHrefs(before, `

https://example.org/forms/annual-report_2026.pdf

`), ["https://example.org/forms/annual"]); }); From 1078e22fb1f3cbbb77e52ec90febd9dc31031d11 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 10:44:47 -0400 Subject: [PATCH 3/5] fix(links): tie a completed URL to the link that printed it (#503) Round 2: a string prefix still let a lost link count as completed by an unrelated new link on the same host, and one URL could complete two. The lost URL's own link must now have printed the longer URL, and each longer URL completes one link. Prod's 9 repairs all meet this. The tag strip loops until stable (CodeQL js/incomplete-multi-character-sanitization). Co-Authored-By: Claude Opus 5.5 --- docs/API.md | 2 +- src/pipeline/links.ts | 34 ++++++++++++++++++++++------------ test/pdf-links.test.ts | 19 ++++++++++++++++++- 3 files changed, 41 insertions(+), 14 deletions(-) diff --git a/docs/API.md b/docs/API.md index 8a57e09b..4f595503 100644 --- a/docs/API.md +++ b/docs/API.md @@ -3095,7 +3095,7 @@ recovered by looking again — logged rather than repaired, and counted into `li ### `editor_links_completed` The Copy Editor replaced an `href` with a longer URL that starts with it and is the link's printed -text (`iteration`, and `links`, each `{from, to}`). The longer URL must be new in that round. This happens when a PDF's link target is cut where the +text (`iteration`, and `links`, each `{from, to}`). The old link must already have printed the longer URL, and the longer URL must be new in that round. This happens when a PDF's link target is cut where the printed URL wraps onto a second line. It is a repair, so it is not in `editor_links_dropped` or `links_dropped_rate`. diff --git a/src/pipeline/links.ts b/src/pipeline/links.ts index 30c02b2c..953aacbf 100644 --- a/src/pipeline/links.ts +++ b/src/pipeline/links.ts @@ -175,26 +175,36 @@ export function droppedHrefs(before: string, after: string): string[] { // the wrap, and the editor, which sees the page, writes the whole printed URL. The link then goes // where the page says, so it is a repair and not a loss. // -// The longer URL must be new in this round, and each one completes only the longest lost URL it -// starts with. Otherwise a lost link to a site's root would count as completed by any full URL on -// that site the document already linked. +// The lost URL's own link must already have printed the longer URL, and the longer URL must be new +// in this round. A string prefix alone would let a lost link to a site's root count as completed +// by any full URL on that site. export function completedHrefs(before: string, after: string): { from: string; to: string }[] { const had = hrefsIn(before); const kept = hrefsIn(after); - const lost = [...had].filter((h) => isAbsolute(h) && !kept.has(h)); const completed = new Map(); - for (const m of after.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))[^>]*>([\s\S]*?)<\/a>/gi)) { - const to = normalizeHref(m[1] ?? m[2] ?? m[3] ?? ""); - const text = normalizeHref(m[4]!.replace(/<[^>]*>/g, "").replace(/\s+/g, "")); - if (to !== text || had.has(to)) continue; - const from = lost - .filter((h) => to.length > h.length && to.startsWith(h) && !completed.has(h)) - .sort((a, b) => b.length - a.length)[0]; - if (from) completed.set(from, to); + const used = new Set(); + const selfLinked = new Set(anchorsIn(after).filter((b) => b.href === b.text).map((b) => b.href)); + // Longest first, so a printed URL completes the longest lost URL it starts with. + for (const a of anchorsIn(before).sort((x, y) => y.href.length - x.href.length)) { + const { href: from, text: to } = a; + if (!isAbsolute(from) || kept.has(from) || completed.has(from) || used.has(to)) continue; + if (to.length <= from.length || !to.startsWith(from) || had.has(to)) continue; + if (!selfLinked.has(to)) continue; + completed.set(from, to); + used.add(to); } return [...completed].map(([from, to]) => ({ from, to })).sort((a, b) => (a.from < b.from ? -1 : 1)); } +// Each `` and its printed text with tags and whitespace removed, both normalized. +function anchorsIn(html: string): { href: string; text: string }[] { + return [...html.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))[^>]*>([\s\S]*?)<\/a>/gi)].map((m) => { + let text = m[4]!; + for (let prev = ""; prev !== text; ) [prev, text] = [text, text.replace(/<[^>]*>/g, "")]; + return { href: normalizeHref(m[1] ?? m[2] ?? m[3] ?? ""), text: normalizeHref(text.replace(/\s+/g, "")) }; + }); +} + // Every in-document reference in the delivered document, and whether it lands (#234). // // This is the question no component asked. `missingLinks` asks whether the source file's diff --git a/test/pdf-links.test.ts b/test/pdf-links.test.ts index 3cf91247..c769db5e 100644 --- a/test/pdf-links.test.ts +++ b/test/pdf-links.test.ts @@ -619,9 +619,26 @@ report_2026.pdf

`; // A lost link to the site's root is not "completed" by a full URL the document already linked. const report = `https://example.org/forms/annual-report.pdf`; assert.deepEqual(droppedHrefs(`

Home ${report}

`, `

Home ${report}

`), ["https://example.org"]); - // A new URL completes only the longest lost URL it starts with. + // Nor by a URL the round newly linked, unless the lost link itself printed it. + const newly = `

Home https://example.org/forms/annual-report.pdf

`; + assert.deepEqual(droppedHrefs(newly, `

Home ${report}

`), ["https://example.org"]); + // A lost root and a repaired target in one round: only the repair is left out, even with the + // repaired URL linked twice. const both = `

Home https://example.org/forms/annual-report_2026.pdf

`; assert.deepEqual(droppedHrefs(both, `

Home ${repaired}

`), ["https://example.org"]); + const rootPrints = `

https://example.org/forms/annual-report_2026.pdf ${before}

`; + assert.deepEqual(completedHrefs(rootPrints, `

${repaired} ${repaired}

`), [ + { from: "https://example.org/forms/annual", to: "https://example.org/forms/annual-report_2026.pdf" }, + ]); + // The longer URL was already linked before the round: still a drop. + const full = "https://example.org/forms/annual-report_2026.pdf"; + assert.deepEqual(droppedHrefs(`${before} ${repaired}`, `

${full}

${repaired}`), ["https://example.org/forms/annual"]); + // The lost link printed the URL, but its target does not start it: still a drop. + assert.deepEqual(droppedHrefs(`

${full}

`, repaired), ["https://example.net/x"]); + // The new link does not print its own URL: still a drop. + assert.deepEqual(droppedHrefs(before, `

the annual report

`), ["https://example.org/forms/annual"]); + // Inner tags in the printed text are ignored. + assert.deepEqual(droppedHrefs(before, repaired.replace(/>(https[^<]*)$1<")), []); // Unwrapped to plain text: still a drop. assert.deepEqual(droppedHrefs(before, `

https://example.org/forms/annual-report_2026.pdf

`), ["https://example.org/forms/annual"]); }); From 52a3ce8e86595760ae64fe63d379c2a630acdf40 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 10:56:49 -0400 Subject: [PATCH 4/5] fix(links): strip stray angle brackets in link text; shorten the event doc (#503) CodeQL did not read the fixed-point loop as complete sanitization, so the text now drops any < or > left after the tag strip. Co-Authored-By: Claude Opus 5.5 --- docs/API.md | 4 ++-- src/pipeline/links.ts | 5 ++--- 2 files changed, 4 insertions(+), 5 deletions(-) diff --git a/docs/API.md b/docs/API.md index 4f595503..174693a5 100644 --- a/docs/API.md +++ b/docs/API.md @@ -3094,8 +3094,8 @@ recovered by looking again — logged rather than repaired, and counted into `li ### `editor_links_completed` -The Copy Editor replaced an `href` with a longer URL that starts with it and is the link's printed -text (`iteration`, and `links`, each `{from, to}`). The old link must already have printed the longer URL, and the longer URL must be new in that round. This happens when a PDF's link target is cut where the +The Copy Editor replaced an `href` with a longer URL that starts with it and is the text its own +link printed (`iteration`, and `links`, each `{from, to}`). The longer URL must be new in that round. This happens when a PDF's link target is cut where the printed URL wraps onto a second line. It is a repair, so it is not in `editor_links_dropped` or `links_dropped_rate`. diff --git a/src/pipeline/links.ts b/src/pipeline/links.ts index 953aacbf..fc06a0b1 100644 --- a/src/pipeline/links.ts +++ b/src/pipeline/links.ts @@ -199,9 +199,8 @@ export function completedHrefs(before: string, after: string): { from: string; t // Each `` and its printed text with tags and whitespace removed, both normalized. function anchorsIn(html: string): { href: string; text: string }[] { return [...html.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))[^>]*>([\s\S]*?)<\/a>/gi)].map((m) => { - let text = m[4]!; - for (let prev = ""; prev !== text; ) [prev, text] = [text, text.replace(/<[^>]*>/g, "")]; - return { href: normalizeHref(m[1] ?? m[2] ?? m[3] ?? ""), text: normalizeHref(text.replace(/\s+/g, "")) }; + const text = m[4]!.replace(/<[^>]*>/g, "").replace(/[<>\s]+/g, ""); + return { href: normalizeHref(m[1] ?? m[2] ?? m[3] ?? ""), text: normalizeHref(text) }; }); } From 7f46a6e507fe622cf6b5b845704d0820fcc63186 Mon Sep 17 00:00:00 2001 From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:09:11 -0400 Subject: [PATCH 5/5] fix(links): back to the single tag strip (#503) The [<>] clause had no test and handled only an unmatched bracket. One global pass of <[^>]*> leaves no complete tag, as in #497, and the text only feeds a comparison. Co-Authored-By: Claude Opus 5.5 --- src/pipeline/links.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/pipeline/links.ts b/src/pipeline/links.ts index fc06a0b1..a4d2ea56 100644 --- a/src/pipeline/links.ts +++ b/src/pipeline/links.ts @@ -199,7 +199,7 @@ export function completedHrefs(before: string, after: string): { from: string; t // Each `` and its printed text with tags and whitespace removed, both normalized. function anchorsIn(html: string): { href: string; text: string }[] { return [...html.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))[^>]*>([\s\S]*?)<\/a>/gi)].map((m) => { - const text = m[4]!.replace(/<[^>]*>/g, "").replace(/[<>\s]+/g, ""); + const text = m[4]!.replace(/<[^>]*>/g, "").replace(/\s+/g, ""); return { href: normalizeHref(m[1] ?? m[2] ?? m[3] ?? ""), text: normalizeHref(text) }; }); }