]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)')[^>]*>([\s\S]*?)<\/a>/gi)) {
+ const href = normalizeHref(m[1] ?? m[2] ?? "");
+ const text = normalizeHref(m[3]!.replace(/<[^>]*>/g, "").replace(/\s+/g, ""));
+ if (href === text) printed.push(href);
+ }
+ return [...hrefsIn(before)]
+ .filter((h) => isAbsolute(h) && !kept.has(h) && printed.some((p) => p.length > h.length && p.startsWith(h)))
+ .sort();
}
// Every in-document reference in the delivered document, and whether it lands (#234).
diff --git a/src/pipeline/review.ts b/src/pipeline/review.ts
index 896a00c6..b6996e87 100644
--- a/src/pipeline/review.ts
+++ b/src/pipeline/review.ts
@@ -37,7 +37,7 @@ import {
import { flatten } from "./flatten.ts";
import { examplesForPrompt } from "./memory.ts";
import { knownPages, pageIndex, type IndexedPage } from "./pageindex.ts";
-import { droppedHrefs } from "./links.ts";
+import { completedHrefs, droppedHrefs } from "./links.ts";
import { sameWordedHeadingNote, sameWordedHeadingRuns } from "./headings.ts";
export interface ReviewIssue {
@@ -3361,6 +3361,8 @@ export async function runReview(
droppedLinks += dropped.length;
ctx.log.event("editor_links_dropped", { iteration: iterations, hrefs: dropped });
}
+ const completed = completedHrefs(before, body);
+ if (completed.length) ctx.log.event("editor_links_completed", { iteration: iterations, hrefs: completed });
// See BODY_MARKERS: the only place a marker's DISAPPEARANCE is recorded. An arrival is also
// recorded on the page path, by `markers_added` on `page_corrected` (#373) — additions only,
// because that corrector is handed the image and resolving an illegible passage is its job. The
diff --git a/test/editor-sections.test.ts b/test/editor-sections.test.ts
index d48a13c7..f1c6beed 100644
--- a/test/editor-sections.test.ts
+++ b/test/editor-sections.test.ts
@@ -620,6 +620,21 @@ test("the round is measured like any other before the loop ends on it", async ()
});
});
+test("a link target the round lengthened to its printed URL is logged as completed, not dropped (#503)", async () => {
+ await withTemp(async (dir) => {
+ const first = `https://example.com/forms/annual-report.pdf the rest of it
`;
+ const fixed = `https://example.com/forms/annual-report.pdf the rest of it
`;
+ const { ctx, rec } = ctxWith(dir, {
+ sectionAnswer: (s) => (s.index === 1 ? s.html.replace(first, fixed) : s.html),
+ });
+ const result = await review(ctx, `${first}\n\n${LONG}`);
+ const completed = rec.events.find((e) => e.type === "editor_links_completed");
+ assert.deepEqual(completed?.data, { iteration: 1, hrefs: ["https://example.com/forms/annual"] });
+ assert.equal(rec.events.some((e) => e.type === "editor_links_dropped"), false);
+ assert.equal(result.droppedLinks, 0);
+ });
+});
+
test("a round answered piece by piece is not a round that converged", async () => {
await withTemp(async (dir) => {
// `review_converged` claims the editor read the whole document, decided it was better left
diff --git a/test/pdf-links.test.ts b/test/pdf-links.test.ts
index fe954d2e..134d8607 100644
--- a/test/pdf-links.test.ts
+++ b/test/pdf-links.test.ts
@@ -15,6 +15,7 @@ import {
type PdfLink,
} from "../src/util/pdf.ts";
import {
+ completedHrefs,
droppedHrefs,
missingLinkProblem,
missingLinks,
@@ -597,6 +598,24 @@ test("a rewrite that loses a link is detectable; one that only renames anchors i
assert.deepEqual(droppedHrefs(before, `the annual report
`), []);
});
+test("a target cut at a line wrap, lengthened to the printed URL, is a repair and not a drop (#503)", () => {
+ // The PDF's link target stops where the printed URL wraps. The editor writes the whole printed URL.
+ const before = `https://example.org/forms/annual-
+report_2026.pdf
`;
+ const repaired = `https://example.org/forms/annual-report_2026.pdf
`;
+ assert.deepEqual(droppedHrefs(before, repaired), []);
+ assert.deepEqual(completedHrefs(before, repaired), ["https://example.org/forms/annual"]);
+ // Longer, but not what the link prints: still a drop.
+ const other = `https://example.org/forms/annual-report_2026.pdf
`;
+ assert.deepEqual(droppedHrefs(before, other), ["https://example.org/forms/annual"]);
+ assert.deepEqual(completedHrefs(before, other), []);
+ // The printed URL, but not starting with the old target: still a drop.
+ const elsewhere = `https://example.net/x
`;
+ assert.deepEqual(droppedHrefs(before, elsewhere), ["https://example.org/forms/annual"]);
+ // Unwrapped to plain text: still a drop.
+ assert.deepEqual(droppedHrefs(before, `https://example.org/forms/annual-report_2026.pdf
`), ["https://example.org/forms/annual"]);
+});
+
// ---------------------------------------------------------------------------
// In-document references (#234)
// ---------------------------------------------------------------------------
From d6b7db47b5e415afe23f81e9469cd5cb91689ed9 Mon Sep 17 00:00:00 2001
From: Blake Bertuccelli-Booth <46652+bbertucc@users.noreply.github.com>
Date: Sun, 4 Oct 2026 10:30:06 -0400
Subject: [PATCH 2/5] fix(links): a completing URL must be new, and completes
one link (#503)
Round 1: a lost link to a site's root was passed off as completed by
any full URL on that site the document already linked. The longer URL
must now be new in the round, and completes only the longest lost URL
it starts with. The event logs {from, to}, unquoted hrefs count, and
the links_dropped_rate definition says what it leaves out.
Co-Authored-By: Claude Opus 5.5
---
docs/API.md | 5 +++--
src/pipeline/links.ts | 28 ++++++++++++++++++----------
src/pipeline/review.ts | 2 +-
test/editor-sections.test.ts | 5 ++++-
test/pdf-links.test.ts | 12 +++++++++++-
5 files changed, 37 insertions(+), 15 deletions(-)
diff --git a/docs/API.md b/docs/API.md
index b10f392b..8a57e09b 100644
--- a/docs/API.md
+++ b/docs/API.md
@@ -319,7 +319,8 @@ curl -s -H "Authorization: Bearer $IRIS_QUALITY_TOKEN" "$BASE/quality?days=30"
`truncated` beside `editor_truncated_rate` and the output ceiling. One threshold over both
cannot be set honestly, which is why the weekly report's is still on the mixture and says so.
* `links_dropped_rate` — share of documents where an `href` present before the copy editor was
- missing after it.
+ missing after it. A URL the editor lengthened to the link's printed URL is not counted (see
+ [`editor_links_completed`](#editor_links_completed)).
* `links_unresolved_rate` — share of documents that shipped with an in-document reference that
lands nowhere: an `href="#"`, or a fragment naming an `id` the delivered document does not
contain. Counted per document; the per-reference numbers, and *which* ids failed, are on the
@@ -3094,7 +3095,7 @@ recovered by looking again — logged rather than repaired, and counted into `li
### `editor_links_completed`
The Copy Editor replaced an `href` with a longer URL that starts with it and is the link's printed
-text (`iteration`, `hrefs`, the old URLs). This happens when a PDF's link target is cut where the
+text (`iteration`, and `links`, each `{from, to}`). The longer URL must be new in that round. This happens when a PDF's link target is cut where the
printed URL wraps onto a second line. It is a repair, so it is not in `editor_links_dropped` or
`links_dropped_rate`.
diff --git a/src/pipeline/links.ts b/src/pipeline/links.ts
index a052bf22..30c02b2c 100644
--- a/src/pipeline/links.ts
+++ b/src/pipeline/links.ts
@@ -166,7 +166,7 @@ export function missingLinkProblem(link: PdfLink): string {
// A URL the rewrite lengthened to the one its link prints is not counted: see `completedHrefs`.
export function droppedHrefs(before: string, after: string): string[] {
const kept = hrefsIn(after);
- const completed = new Set(completedHrefs(before, after));
+ const completed = new Set(completedHrefs(before, after).map((c) => c.from));
return [...hrefsIn(before)].filter((h) => isAbsolute(h) && !kept.has(h) && !completed.has(h)).sort();
}
@@ -174,17 +174,25 @@ export function droppedHrefs(before: string, after: string): string[] {
// printed text (#503). A URL that wraps onto a second line in a PDF can carry a link target cut at
// the wrap, and the editor, which sees the page, writes the whole printed URL. The link then goes
// where the page says, so it is a repair and not a loss.
-export function completedHrefs(before: string, after: string): string[] {
+//
+// The longer URL must be new in this round, and each one completes only the longest lost URL it
+// starts with. Otherwise a lost link to a site's root would count as completed by any full URL on
+// that site the document already linked.
+export function completedHrefs(before: string, after: string): { from: string; to: string }[] {
+ const had = hrefsIn(before);
const kept = hrefsIn(after);
- const printed: string[] = [];
- for (const m of after.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)')[^>]*>([\s\S]*?)<\/a>/gi)) {
- const href = normalizeHref(m[1] ?? m[2] ?? "");
- const text = normalizeHref(m[3]!.replace(/<[^>]*>/g, "").replace(/\s+/g, ""));
- if (href === text) printed.push(href);
+ const lost = [...had].filter((h) => isAbsolute(h) && !kept.has(h));
+ const completed = new Map();
+ for (const m of after.matchAll(/]*?\bhref\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))[^>]*>([\s\S]*?)<\/a>/gi)) {
+ const to = normalizeHref(m[1] ?? m[2] ?? m[3] ?? "");
+ const text = normalizeHref(m[4]!.replace(/<[^>]*>/g, "").replace(/\s+/g, ""));
+ if (to !== text || had.has(to)) continue;
+ const from = lost
+ .filter((h) => to.length > h.length && to.startsWith(h) && !completed.has(h))
+ .sort((a, b) => b.length - a.length)[0];
+ if (from) completed.set(from, to);
}
- return [...hrefsIn(before)]
- .filter((h) => isAbsolute(h) && !kept.has(h) && printed.some((p) => p.length > h.length && p.startsWith(h)))
- .sort();
+ return [...completed].map(([from, to]) => ({ from, to })).sort((a, b) => (a.from < b.from ? -1 : 1));
}
// Every in-document reference in the delivered document, and whether it lands (#234).
diff --git a/src/pipeline/review.ts b/src/pipeline/review.ts
index b6996e87..e14277b4 100644
--- a/src/pipeline/review.ts
+++ b/src/pipeline/review.ts
@@ -3362,7 +3362,7 @@ export async function runReview(
ctx.log.event("editor_links_dropped", { iteration: iterations, hrefs: dropped });
}
const completed = completedHrefs(before, body);
- if (completed.length) ctx.log.event("editor_links_completed", { iteration: iterations, hrefs: completed });
+ if (completed.length) ctx.log.event("editor_links_completed", { iteration: iterations, links: completed });
// See BODY_MARKERS: the only place a marker's DISAPPEARANCE is recorded. An arrival is also
// recorded on the page path, by `markers_added` on `page_corrected` (#373) — additions only,
// because that corrector is handed the image and resolving an illegible passage is its job. The
diff --git a/test/editor-sections.test.ts b/test/editor-sections.test.ts
index f1c6beed..9ac32e10 100644
--- a/test/editor-sections.test.ts
+++ b/test/editor-sections.test.ts
@@ -629,7 +629,10 @@ test("a link target the round lengthened to its printed URL is logged as complet
});
const result = await review(ctx, `${first}\n\n${LONG}`);
const completed = rec.events.find((e) => e.type === "editor_links_completed");
- assert.deepEqual(completed?.data, { iteration: 1, hrefs: ["https://example.com/forms/annual"] });
+ assert.deepEqual(completed?.data, {
+ iteration: 1,
+ links: [{ from: "https://example.com/forms/annual", to: "https://example.com/forms/annual-report.pdf" }],
+ });
assert.equal(rec.events.some((e) => e.type === "editor_links_dropped"), false);
assert.equal(result.droppedLinks, 0);
});
diff --git a/test/pdf-links.test.ts b/test/pdf-links.test.ts
index 134d8607..3cf91247 100644
--- a/test/pdf-links.test.ts
+++ b/test/pdf-links.test.ts
@@ -604,7 +604,11 @@ test("a target cut at a line wrap, lengthened to the printed URL, is a repair an
report_2026.pdf