From 4ec75b4fc68a22bc607404e87fb59a87624b8d05 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 28 Jul 2026 16:01:09 +0000 Subject: [PATCH] fix(leads): a link ends where its URL characters end MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The live draft read "...at https://threatcrush.com/get-whitepaper—download takes a minute", and the guard extracted the URL as "https://threatcrush.com/get-whitepaper—download" — matched it against the campaign, found nothing, and rejected the draft for linking somewhere invented. The campaign had declared that URL. The em-dash was prose. Matching "anything up to whitespace" is what did it. Non-ASCII cannot appear in a URL without being percent-encoded, so the dash was never part of the link; the pattern now matches only what RFC 3986 permits, and trailing sentence punctuation is stripped rather than only a full stop or a comma. The draft is also repaired rather than merely re-checked. A URL welded to an em-dash is one some mail clients will autolink including the dash, which is a 404 on the single thing the email asked the recipient to click — so a space is inserted, and the tidied body is what gets sent. Checking a repaired draft and then sending the broken original would have been worse than not repairing it. Co-Authored-By: Claude Opus 5 (1M context) --- lib/outreach/cold.ts | 41 ++++++++++++++++++++++++++-- lib/outreach/pipeline.ts | 11 ++++++-- tests/prod-regressions.test.ts | 50 +++++++++++++++++++++++++++++++++- 3 files changed, 97 insertions(+), 5 deletions(-) diff --git a/lib/outreach/cold.ts b/lib/outreach/cold.ts index c74cedde..37866e76 100644 --- a/lib/outreach/cold.ts +++ b/lib/outreach/cold.ts @@ -414,6 +414,44 @@ export function unsupportedClaims(body: string, facts: ProspectFacts): string[] * facts, links to places the campaign never mentioned, and claimed prior * contact. Tone and phrasing are the prompt's job. */ +/** + * A URL as it appears inside prose. + * + * Restricted to the characters RFC 3986 actually permits. Matching "anything + * up to whitespace" swallows whatever punctuation the writer put immediately + * after the link: a draft reading "...at https://x.test/paper—it takes two + * minutes" yielded the URL "https://x.test/paper—it", which matched nothing + * the campaign declared and got the draft rejected for linking somewhere + * invented. Non-ASCII cannot appear in a URL unencoded, so the dash is the + * end of the link by definition. + */ +export const URL_IN_TEXT = /https?:\/\/[A-Za-z0-9\-._~:/?#[\]@!$&'()*+,;=%]+/g; + +/** Punctuation that ends a sentence rather than belonging to the link. */ +const TRAILING_PUNCT = /[.,;:!?'")\]]+$/; + +/** The URLs a piece of text actually links to. */ +export function urlsIn(text: string): string[] { + return [...text.matchAll(URL_IN_TEXT)].map((m) => m[0].replace(TRAILING_PUNCT, "")); +} + +/** + * Put a space between a link and whatever is jammed against it. + * + * Belt and braces for the same draft. Even with the check corrected, a URL + * butted directly against an em-dash is a link some mail clients will + * autolink *including* the dash — so the recipient gets a 404 on the one + * thing the email asked them to click. + */ +export function separateUrlPunctuation(body: string): string { + return body.replace(URL_IN_TEXT, (url, offset: number, whole: string) => { + const next = whole[offset + url.length]; + // Only when something non-ASCII is touching the end of the link. A space, + // a full stop or the end of the string are all already fine. + return next && next.charCodeAt(0) > 127 ? `${url} ` : url; + }); +} + export function unsupportedCustomClaims(body: string, declaredText: string[]): string[] { const problems: string[] = []; const lower = body.toLowerCase(); @@ -438,8 +476,7 @@ export function unsupportedCustomClaims(body: string, declaredText: string[]): s // Links must be traceable too: a made-up portfolio URL // is both a false claim and a broken promise. - for (const m of body.matchAll(/https?:\/\/[^\s)>\]]+/gi)) { - const url = m[0].replace(/[.,]$/, ""); + for (const url of urlsIn(body)) { if (!declared.includes(url.toLowerCase())) { problems.push(`links to ${url}, which the campaign never mentions`); } diff --git a/lib/outreach/pipeline.ts b/lib/outreach/pipeline.ts index 5157d6a8..dd6cd8cb 100644 --- a/lib/outreach/pipeline.ts +++ b/lib/outreach/pipeline.ts @@ -39,6 +39,7 @@ import { unsupportedClaims, roleAddressGuesses, unsupportedCustomClaims, + separateUrlPunctuation, type ContactCandidate, type OutreachStep, type ProspectFacts, @@ -841,7 +842,11 @@ async function draftCustomEmail(input: { // draft to them for inventing it — the campaign, quite correctly, had never // mentioned a four. Naming who you are writing to is not a claim about // them; it is the minimum a personalised email does. - const problems = unsupportedCustomClaims(output.body, [ + // Applied before the check, and kept, so the recipient gets the tidied + // version rather than a link with punctuation welded to its end. + const body = separateUrlPunctuation(output.body); + + const problems = unsupportedCustomClaims(body, [ ...facts, input.pitch.intro, input.pitch.ask ?? "", @@ -857,7 +862,9 @@ async function draftCustomEmail(input: { return { ok: true, subject, - body: output.body.trim(), + // The tidied body, not the raw one — checking a repaired draft and then + // sending the broken original would defeat the repair entirely. + body: body.trim(), evidenceUsed: output.evidence_used ?? [], }; } diff --git a/tests/prod-regressions.test.ts b/tests/prod-regressions.test.ts index 49861379..0479b244 100644 --- a/tests/prod-regressions.test.ts +++ b/tests/prod-regressions.test.ts @@ -1,6 +1,6 @@ import { describe, it, expect } from "vitest"; import { readFileSync } from "node:fs"; -import { unsupportedCustomClaims } from "@/lib/outreach/cold"; +import { separateUrlPunctuation, unsupportedCustomClaims, urlsIn } from "@/lib/outreach/cold"; // Three failures that only appeared once campaigns ran against real // businesses. Each is pinned by the case that produced it. @@ -68,3 +68,51 @@ describe("people are output worth paying for", () => { expect(runner).toMatch(/if \(billing\.charged\(\) && !producedSomething\)/); }); }); + +describe("a link ends where the URL characters end", () => { + const guard = unsupportedCustomClaims; + const declared = ["download it at https://threatcrush.com/get-whitepaper"]; + + it("does not swallow the punctuation jammed against it", () => { + // The live draft read "...get-whitepaper—download..." and the guard read + // the em-dash and the next word as part of the URL, then rejected the + // draft for linking somewhere the campaign never mentioned. + expect(urlsIn("grab it at https://threatcrush.com/get-whitepaper—download now")).toEqual([ + "https://threatcrush.com/get-whitepaper", + ]); + }); + + it("accepts that draft instead of rejecting it", () => { + const body = "Grab it at https://threatcrush.com/get-whitepaper—download takes a minute."; + expect(guard(body, declared)).toEqual([]); + }); + + it("still strips ordinary sentence punctuation", () => { + expect(urlsIn("see https://x.test/a.")).toEqual(["https://x.test/a"]); + expect(urlsIn("see (https://x.test/a), then")).toEqual(["https://x.test/a"]); + }); + + it("keeps punctuation that is genuinely part of the path", () => { + expect(urlsIn("see https://x.test/a_b-c~d/e?f=1&g=2#h then")).toEqual([ + "https://x.test/a_b-c~d/e?f=1&g=2#h", + ]); + }); + + it("separates the dash so no mail client can autolink it", () => { + // Even with the check corrected, a URL welded to an em-dash is a link + // some clients will autolink including the dash — a 404 on the one thing + // the email asked the recipient to click. + expect(separateUrlPunctuation("at https://x.test/p—download now")).toBe( + "at https://x.test/p —download now", + ); + }); + + it("leaves a well-formed sentence untouched", () => { + const clean = "Grab it at https://x.test/p. It takes a minute."; + expect(separateUrlPunctuation(clean)).toBe(clean); + }); + + it("still catches a genuinely invented link", () => { + expect(guard("see https://not-declared.test/x", declared).join(" ")).toMatch(/links to/); + }); +});