diff --git a/lib/lx/autoGuestPost.ts b/lib/lx/autoGuestPost.ts index e2238f8..d8c1058 100644 --- a/lib/lx/autoGuestPost.ts +++ b/lib/lx/autoGuestPost.ts @@ -21,6 +21,7 @@ import type { SupabaseClient } from "@supabase/supabase-js"; import { findGuestPostOpportunities } from "./guestPostMatcher"; +import { subjectFromTopicFeeds } from "./feedTopics"; /** * One in every this many published posts is a guest post. @@ -157,8 +158,37 @@ export async function planGuestPost( }; } - // Every partner is either cooling off or has had all its crossed topics - // used. That is a healthy network doing its job, not an error — the caller - // publishes to the author's own blog instead. + // Every partner is either cooling off or has had all its crossed topics used. + // + // The crossings are finite — they are combinations of two fixed keyword + // lists — so a partner written for a few times exhausts them and the slot + // falls back to the author's own blog for ever after. Before giving up, take + // a subject from what the small web is actually publishing: a real post from + // an RSS Amplifier topic feed, picked at random, which the generator then + // writes a full article about. Nothing is copied; what the feed contributes + // is a subject somebody genuinely cared about this week rather than one + // assembled from two keyword lists. + for (const opportunity of opportunities) { + if (cooling.has(opportunity.partner_site_id)) continue; + + const found = await subjectFromTopicFeeds( + // The partner's own suggested topics are the best guide to what their + // readers came for, and they are already crossed against ours. + opportunity.suggested_topics ?? [], + ); + if (!found) continue; + + const key = `${opportunity.partner_site_id}::${found.subject.trim().toLowerCase()}`; + if (taken.has(key)) continue; + + return { + targetSiteId: opportunity.partner_site_id, + targetDomain: opportunity.partner_domain, + topic: found.subject, + }; + } + + // Nothing anywhere. A healthy network doing its job rather than an error — + // the caller publishes to the author's own blog instead. return null; } diff --git a/lib/lx/feedTopics.ts b/lib/lx/feedTopics.ts new file mode 100644 index 0000000..80215bb --- /dev/null +++ b/lib/lx/feedTopics.ts @@ -0,0 +1,176 @@ +// Guest-post subjects taken from what the small web is actually writing about. +// +// The crossed-seed topics that `guestPostMatcher` produces are combinations of +// two sites' own keywords — reliable, and finite. Once a partner has been +// written for a few times the crossings run out and the slot goes back to the +// author's own blog. This is the other source: a real, recently published post +// from an RSS Amplifier topic feed, used as the subject for a full article. +// +// The article is still entirely ours — the generator writes it from scratch on +// the subject. Nothing is copied. What the feed contributes is a subject that +// somebody in the niche genuinely cared about this week, rather than one +// assembled from two keyword lists. +// +// Three things this is careful about, and the first is the one that would be +// most embarrassing. + +/** Where the directory lives. */ +const RSSAMPLIFIER = "https://rssamplifier.com"; + +/** + * How long to wait for a topic feed. + * + * This runs inside the publishing cron, which is walking every active site. A + * slow directory must cost one site its guest post, not the whole sweep, so the + * budget is small and every failure path returns null. + */ +const TIMEOUT_MS = 3000; + +/** Feeds tried before giving up on finding a usable subject. */ +const MAX_FEEDS = 3; + +/** A title shorter than this is not a subject — it is a label. */ +const MIN_TITLE_LEN = 24; + +/** And one longer than this is a paragraph that will not survive a prompt. */ +const MAX_TITLE_LEN = 160; + +/** + * A keyword as it appears in a topic URL. + * + * Matches the directory's own slugging — lowercase, non-alphanumerics to + * hyphens, collapsed. A keyword that slugs to nothing is skipped rather than + * requested, since `/topics/.rss` is not a feed. + */ +export function topicSlug(keyword: string): string { + return (keyword ?? "") + .toLowerCase() + .replace(/[^a-z0-9]+/g, "-") + .replace(/^-+|-+$/g, ""); +} + +/** + * Titles of real posts in a topic feed. + * + * Parsed with a regex rather than an XML library on purpose: this is one known + * document shape from one known publisher, the only field wanted is the title, + * and a parse failure here must degrade to "no subject today" rather than + * throw inside a cron. Dragging an XML dependency into the worker to read one + * element would be the worse trade. + * + * @param xml an RSS document + * @returns post titles, channel title and sponsored items removed + */ +export function itemTitles(xml: string): string[] { + const out: string[] = []; + + // Item blocks only — this is what keeps the channel's own (the name + // of the topic) out of the candidate list. + const items = xml.match(/<item\b[\s\S]*?<\/item>/gi) ?? []; + + for (const item of items) { + // **Sponsored items are ours.** The directory's feeds now carry CrawlProof + // ad fills as syndication items, so without this the cron could pick one of + // our own advertisements and commission a guest post about it — an ad, + // laundered into editorial, published on a partner's blog under our name. + // Both the category and the title suffix are checked because either alone + // is a single point of failure for something that must never happen. + if (/<category>\s*Sponsored\s*<\/category>/i.test(item)) continue; + + const raw = item.match(/<title>([\s\S]*?)<\/title>/i)?.[1]; + if (!raw) continue; + + const title = decodeXml(raw).trim(); + if (/\(sponsored\)\s*$/i.test(title)) continue; + + if (title.length < MIN_TITLE_LEN || title.length > MAX_TITLE_LEN) continue; + out.push(title); + } + + return out; +} + +/** + * Undo the escaping a feed document applies, and nothing else. + * + * The five predefined entities plus numeric references, because titles arrive + * carrying `→` and `'` from fifty thousand different publishers and + * a subject line reading "Don't" would be written into the article. + */ +function decodeXml(s: string): string { + return s + .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, "$1") + .replace(/&#(\d+);/g, (_m, d) => String.fromCodePoint(Number(d))) + .replace(/&#x([0-9a-f]+);/gi, (_m, h) => String.fromCodePoint(parseInt(h, 16))) + .replace(/</g, "<") + .replace(/>/g, ">") + .replace(/"/g, '"') + .replace(/'/g, "'") + .replace(/&/g, "&"); +} + +/** + * Pick a subject at random from what the directory is carrying. + * + * Random rather than ranked, deliberately. Ranking these would mean deciding + * that one publisher's headline is a better subject than another's on evidence + * we do not have — and the failure it would cause is worse than the one it + * would prevent: a stable ranking over a slow-moving feed writes about the same + * thing repeatedly, which is exactly what this source exists to avoid. + * + * @param keywords subject words to look for, tried in random order + * @param fetchImpl injected by the tests + * @returns a subject, or null to let the caller fall back + */ +export async function subjectFromTopicFeeds( + keywords: string[], + fetchImpl: typeof fetch = fetch, +): Promise<{ subject: string; topic: string } | null> { + const slugs = shuffle( + Array.from(new Set((keywords ?? []).map(topicSlug).filter(Boolean))), + ).slice(0, MAX_FEEDS); + + for (const slug of slugs) { + const titles = await titlesFor(slug, fetchImpl); + if (titles.length === 0) continue; + + const subject = titles[Math.floor(Math.random() * titles.length)]; + return { subject, topic: slug.replace(/-/g, " ") }; + } + + return null; +} + +/** + * @param slug + * @param fetchImpl + * @returns usable titles, or [] for any failure at all + */ +async function titlesFor(slug: string, fetchImpl: typeof fetch): Promise<string[]> { + try { + const res = await fetchImpl(`${RSSAMPLIFIER}/topics/${encodeURIComponent(slug)}.rss`, { + signal: AbortSignal.timeout(TIMEOUT_MS), + headers: { accept: "application/rss+xml, application/xml;q=0.9" }, + }); + if (!res.ok) return []; + return itemTitles(await res.text()); + } catch { + // A missing topic, a slow directory, a malformed document — all the same + // answer. The caller has an ordinary post to publish instead. + return []; + } +} + +/** + * @template T + * @param xs + * @returns a shuffled copy + */ +function shuffle<T>(xs: T[]): T[] { + const out = [...xs]; + for (let i = out.length - 1; i > 0; i -= 1) { + const j = Math.floor(Math.random() * (i + 1)); + [out[i], out[j]] = [out[j], out[i]]; + } + return out; +} diff --git a/tests/contract/feed-topics.test.ts b/tests/contract/feed-topics.test.ts new file mode 100644 index 0000000..3c03eea --- /dev/null +++ b/tests/contract/feed-topics.test.ts @@ -0,0 +1,155 @@ +import { describe, it, expect } from "vitest"; + +import { + itemTitles, + subjectFromTopicFeeds, + topicSlug, +} from "@/lib/lx/feedTopics"; + +const rss = (items: string) => `<?xml version="1.0"?> +<rss version="2.0"><channel> + <title>marketing — RSS Amplifier + ${items} +`; + +const item = (title: string, extra = "") => + `${title}${extra}`; + +describe("reading subjects out of a topic feed", () => { + it("never offers one of our own ads as a subject", () => { + // The failure this exists to prevent, and it is not hypothetical: the + // directory's feeds now carry CrawlProof ad fills as syndication items. + // Without this filter the cron could commission a guest post *about an + // advertisement* and publish it on a partner's blog under our name — an ad + // laundered into editorial. Both signals are checked because either alone + // is a single point of failure for something that must never happen. + const xml = rss( + [ + item( + "Home of the Agentic Pull Request (Sponsored)", + "Sponsored", + ), + item("Another Paid Placement Dressed As A Post", "Sponsored"), + item("A Sponsored Title With No Category Element (Sponsored)"), + item("Why I Said Yes to a $1,000-an-Hour Coach"), + ].join(""), + ); + + expect(itemTitles(xml)).toEqual(["Why I Said Yes to a $1,000-an-Hour Coach"]); + }); + + it("leaves the channel's own title out of the candidates", () => { + // The topic name is a label, not a subject. It sits outside , which + // is why the parse is scoped to item blocks. + const titles = itemTitles(rss(item("A perfectly ordinary post title here"))); + expect(titles).not.toContain("marketing — RSS Amplifier"); + expect(titles).toEqual(["A perfectly ordinary post title here"]); + }); + + it("drops titles too short to be a subject and too long to be a prompt", () => { + const xml = rss( + [ + item("Weeknotes"), + item("Ok"), + item("x".repeat(400)), + item("A title of a perfectly reasonable length for an article"), + ].join(""), + ); + expect(itemTitles(xml)).toEqual([ + "A title of a perfectly reasonable length for an article", + ]); + }); + + it("decodes what the feed escaped, so the subject reads as written", () => { + // Titles arrive from fifty thousand publishers carrying entities. A subject + // line reading "Don't" would be written into the article verbatim. + const xml = rss( + item("Don't Ship It & Hope — A Checklist For Releases"), + ); + expect(itemTitles(xml)[0]).toBe( + "Don't Ship It & Hope — A Checklist For Releases", + ); + }); + + it("reads a CDATA title", () => { + const xml = rss(item("")); + expect(itemTitles(xml)).toEqual(["A title wrapped in CDATA for safety"]); + }); + + it("returns nothing for a document that is not a feed", () => { + expect(itemTitles("not a feed at all")).toEqual([]); + expect(itemTitles("")).toEqual([]); + }); +}); + +describe("topicSlug", () => { + it("matches the directory's own slugging", () => { + expect(topicSlug("Machine Learning")).toBe("machine-learning"); + expect(topicSlug(" AI & robotics ")).toBe("ai-robotics"); + expect(topicSlug("C++")).toBe("c"); + }); + + it("slugs unusable input to empty rather than to a bad URL", () => { + // `/topics/.rss` is not a feed; the caller skips these instead of asking. + expect(topicSlug("!!!")).toBe(""); + expect(topicSlug("")).toBe(""); + }); +}); + +describe("picking a subject", () => { + const ok = (body: string) => + ({ ok: true, text: async () => body }) as unknown as Response; + + it("returns a real post title from the first feed that has one", async () => { + const fetchImpl = (async () => + ok(rss(item("How We Cut Our Build Time By Ninety Percent")))) as typeof fetch; + + const found = await subjectFromTopicFeeds(["build systems"], fetchImpl); + expect(found?.subject).toBe("How We Cut Our Build Time By Ninety Percent"); + }); + + it("moves on when a feed is empty or missing", async () => { + const seen: string[] = []; + const fetchImpl = (async (url: string) => { + seen.push(String(url)); + if (seen.length === 1) return { ok: false } as Response; + if (seen.length === 2) return ok(rss("")); + return ok(rss(item("The third feed finally has something usable"))); + }) as unknown as typeof fetch; + + const found = await subjectFromTopicFeeds(["a", "b", "c"], fetchImpl); + expect(found?.subject).toBe("The third feed finally has something usable"); + }); + + it("returns null rather than throwing when the directory is unreachable", async () => { + // This runs inside the publishing cron while it walks every active site. A + // slow directory must cost one site its guest post, not the sweep. + const fetchImpl = (async () => { + throw new Error("network down"); + }) as unknown as typeof fetch; + + expect(await subjectFromTopicFeeds(["anything"], fetchImpl)).toBeNull(); + }); + + it("asks for nothing when there are no usable keywords", async () => { + let called = false; + const fetchImpl = (async () => { + called = true; + return ok(rss("")); + }) as unknown as typeof fetch; + + expect(await subjectFromTopicFeeds(["!!!", ""], fetchImpl)).toBeNull(); + expect(called).toBe(false); + }); + + it("does not ask the same feed twice for a duplicated keyword", async () => { + const asked: string[] = []; + const fetchImpl = (async (url: string) => { + asked.push(String(url)); + return ok(rss("")); + }) as unknown as typeof fetch; + + await subjectFromTopicFeeds(["Design", "design", " DESIGN "], fetchImpl); + expect(asked).toHaveLength(1); + }); +});