diff --git a/lib/lx/autoGuestPost.ts b/lib/lx/autoGuestPost.ts
index e2238f8..d8c1058 100644
--- a/lib/lx/autoGuestPost.ts
+++ b/lib/lx/autoGuestPost.ts
@@ -21,6 +21,7 @@
import type { SupabaseClient } from "@supabase/supabase-js";
import { findGuestPostOpportunities } from "./guestPostMatcher";
+import { subjectFromTopicFeeds } from "./feedTopics";
/**
* One in every this many published posts is a guest post.
@@ -157,8 +158,37 @@ export async function planGuestPost(
};
}
- // Every partner is either cooling off or has had all its crossed topics
- // used. That is a healthy network doing its job, not an error — the caller
- // publishes to the author's own blog instead.
+ // Every partner is either cooling off or has had all its crossed topics used.
+ //
+ // The crossings are finite — they are combinations of two fixed keyword
+ // lists — so a partner written for a few times exhausts them and the slot
+ // falls back to the author's own blog for ever after. Before giving up, take
+ // a subject from what the small web is actually publishing: a real post from
+ // an RSS Amplifier topic feed, picked at random, which the generator then
+ // writes a full article about. Nothing is copied; what the feed contributes
+ // is a subject somebody genuinely cared about this week rather than one
+ // assembled from two keyword lists.
+ for (const opportunity of opportunities) {
+ if (cooling.has(opportunity.partner_site_id)) continue;
+
+ const found = await subjectFromTopicFeeds(
+ // The partner's own suggested topics are the best guide to what their
+ // readers came for, and they are already crossed against ours.
+ opportunity.suggested_topics ?? [],
+ );
+ if (!found) continue;
+
+ const key = `${opportunity.partner_site_id}::${found.subject.trim().toLowerCase()}`;
+ if (taken.has(key)) continue;
+
+ return {
+ targetSiteId: opportunity.partner_site_id,
+ targetDomain: opportunity.partner_domain,
+ topic: found.subject,
+ };
+ }
+
+ // Nothing anywhere. A healthy network doing its job rather than an error —
+ // the caller publishes to the author's own blog instead.
return null;
}
diff --git a/lib/lx/feedTopics.ts b/lib/lx/feedTopics.ts
new file mode 100644
index 0000000..80215bb
--- /dev/null
+++ b/lib/lx/feedTopics.ts
@@ -0,0 +1,176 @@
+// Guest-post subjects taken from what the small web is actually writing about.
+//
+// The crossed-seed topics that `guestPostMatcher` produces are combinations of
+// two sites' own keywords — reliable, and finite. Once a partner has been
+// written for a few times the crossings run out and the slot goes back to the
+// author's own blog. This is the other source: a real, recently published post
+// from an RSS Amplifier topic feed, used as the subject for a full article.
+//
+// The article is still entirely ours — the generator writes it from scratch on
+// the subject. Nothing is copied. What the feed contributes is a subject that
+// somebody in the niche genuinely cared about this week, rather than one
+// assembled from two keyword lists.
+//
+// Three things this is careful about, and the first is the one that would be
+// most embarrassing.
+
+/** Where the directory lives. */
+const RSSAMPLIFIER = "https://rssamplifier.com";
+
+/**
+ * How long to wait for a topic feed.
+ *
+ * This runs inside the publishing cron, which is walking every active site. A
+ * slow directory must cost one site its guest post, not the whole sweep, so the
+ * budget is small and every failure path returns null.
+ */
+const TIMEOUT_MS = 3000;
+
+/** Feeds tried before giving up on finding a usable subject. */
+const MAX_FEEDS = 3;
+
+/** A title shorter than this is not a subject — it is a label. */
+const MIN_TITLE_LEN = 24;
+
+/** And one longer than this is a paragraph that will not survive a prompt. */
+const MAX_TITLE_LEN = 160;
+
+/**
+ * A keyword as it appears in a topic URL.
+ *
+ * Matches the directory's own slugging — lowercase, non-alphanumerics to
+ * hyphens, collapsed. A keyword that slugs to nothing is skipped rather than
+ * requested, since `/topics/.rss` is not a feed.
+ */
+export function topicSlug(keyword: string): string {
+ return (keyword ?? "")
+ .toLowerCase()
+ .replace(/[^a-z0-9]+/g, "-")
+ .replace(/^-+|-+$/g, "");
+}
+
+/**
+ * Titles of real posts in a topic feed.
+ *
+ * Parsed with a regex rather than an XML library on purpose: this is one known
+ * document shape from one known publisher, the only field wanted is the title,
+ * and a parse failure here must degrade to "no subject today" rather than
+ * throw inside a cron. Dragging an XML dependency into the worker to read one
+ * element would be the worse trade.
+ *
+ * @param xml an RSS document
+ * @returns post titles, channel title and sponsored items removed
+ */
+export function itemTitles(xml: string): string[] {
+ const out: string[] = [];
+
+ // Item blocks only — this is what keeps the channel's own
(the name
+ // of the topic) out of the candidate list.
+ const items = xml.match(/- /gi) ?? [];
+
+ for (const item of items) {
+ // **Sponsored items are ours.** The directory's feeds now carry CrawlProof
+ // ad fills as syndication items, so without this the cron could pick one of
+ // our own advertisements and commission a guest post about it — an ad,
+ // laundered into editorial, published on a partner's blog under our name.
+ // Both the category and the title suffix are checked because either alone
+ // is a single point of failure for something that must never happen.
+ if (/\s*Sponsored\s*<\/category>/i.test(item)) continue;
+
+ const raw = item.match(/([\s\S]*?)<\/title>/i)?.[1];
+ if (!raw) continue;
+
+ const title = decodeXml(raw).trim();
+ if (/\(sponsored\)\s*$/i.test(title)) continue;
+
+ if (title.length < MIN_TITLE_LEN || title.length > MAX_TITLE_LEN) continue;
+ out.push(title);
+ }
+
+ return out;
+}
+
+/**
+ * Undo the escaping a feed document applies, and nothing else.
+ *
+ * The five predefined entities plus numeric references, because titles arrive
+ * carrying `→` and `'` from fifty thousand different publishers and
+ * a subject line reading "Don't" would be written into the article.
+ */
+function decodeXml(s: string): string {
+ return s
+ .replace(//g, "$1")
+ .replace(/(\d+);/g, (_m, d) => String.fromCodePoint(Number(d)))
+ .replace(/([0-9a-f]+);/gi, (_m, h) => String.fromCodePoint(parseInt(h, 16)))
+ .replace(/</g, "<")
+ .replace(/>/g, ">")
+ .replace(/"/g, '"')
+ .replace(/'/g, "'")
+ .replace(/&/g, "&");
+}
+
+/**
+ * Pick a subject at random from what the directory is carrying.
+ *
+ * Random rather than ranked, deliberately. Ranking these would mean deciding
+ * that one publisher's headline is a better subject than another's on evidence
+ * we do not have — and the failure it would cause is worse than the one it
+ * would prevent: a stable ranking over a slow-moving feed writes about the same
+ * thing repeatedly, which is exactly what this source exists to avoid.
+ *
+ * @param keywords subject words to look for, tried in random order
+ * @param fetchImpl injected by the tests
+ * @returns a subject, or null to let the caller fall back
+ */
+export async function subjectFromTopicFeeds(
+ keywords: string[],
+ fetchImpl: typeof fetch = fetch,
+): Promise<{ subject: string; topic: string } | null> {
+ const slugs = shuffle(
+ Array.from(new Set((keywords ?? []).map(topicSlug).filter(Boolean))),
+ ).slice(0, MAX_FEEDS);
+
+ for (const slug of slugs) {
+ const titles = await titlesFor(slug, fetchImpl);
+ if (titles.length === 0) continue;
+
+ const subject = titles[Math.floor(Math.random() * titles.length)];
+ return { subject, topic: slug.replace(/-/g, " ") };
+ }
+
+ return null;
+}
+
+/**
+ * @param slug
+ * @param fetchImpl
+ * @returns usable titles, or [] for any failure at all
+ */
+async function titlesFor(slug: string, fetchImpl: typeof fetch): Promise {
+ try {
+ const res = await fetchImpl(`${RSSAMPLIFIER}/topics/${encodeURIComponent(slug)}.rss`, {
+ signal: AbortSignal.timeout(TIMEOUT_MS),
+ headers: { accept: "application/rss+xml, application/xml;q=0.9" },
+ });
+ if (!res.ok) return [];
+ return itemTitles(await res.text());
+ } catch {
+ // A missing topic, a slow directory, a malformed document — all the same
+ // answer. The caller has an ordinary post to publish instead.
+ return [];
+ }
+}
+
+/**
+ * @template T
+ * @param xs
+ * @returns a shuffled copy
+ */
+function shuffle(xs: T[]): T[] {
+ const out = [...xs];
+ for (let i = out.length - 1; i > 0; i -= 1) {
+ const j = Math.floor(Math.random() * (i + 1));
+ [out[i], out[j]] = [out[j], out[i]];
+ }
+ return out;
+}
diff --git a/tests/contract/feed-topics.test.ts b/tests/contract/feed-topics.test.ts
new file mode 100644
index 0000000..3c03eea
--- /dev/null
+++ b/tests/contract/feed-topics.test.ts
@@ -0,0 +1,155 @@
+import { describe, it, expect } from "vitest";
+
+import {
+ itemTitles,
+ subjectFromTopicFeeds,
+ topicSlug,
+} from "@/lib/lx/feedTopics";
+
+const rss = (items: string) => `
+
+ marketing — RSS Amplifier
+ ${items}
+`;
+
+const item = (title: string, extra = "") =>
+ `
- ${title}${extra}
`;
+
+describe("reading subjects out of a topic feed", () => {
+ it("never offers one of our own ads as a subject", () => {
+ // The failure this exists to prevent, and it is not hypothetical: the
+ // directory's feeds now carry CrawlProof ad fills as syndication items.
+ // Without this filter the cron could commission a guest post *about an
+ // advertisement* and publish it on a partner's blog under our name — an ad
+ // laundered into editorial. Both signals are checked because either alone
+ // is a single point of failure for something that must never happen.
+ const xml = rss(
+ [
+ item(
+ "Home of the Agentic Pull Request (Sponsored)",
+ "Sponsored",
+ ),
+ item("Another Paid Placement Dressed As A Post", "Sponsored"),
+ item("A Sponsored Title With No Category Element (Sponsored)"),
+ item("Why I Said Yes to a $1,000-an-Hour Coach"),
+ ].join(""),
+ );
+
+ expect(itemTitles(xml)).toEqual(["Why I Said Yes to a $1,000-an-Hour Coach"]);
+ });
+
+ it("leaves the channel's own title out of the candidates", () => {
+ // The topic name is a label, not a subject. It sits outside - , which
+ // is why the parse is scoped to item blocks.
+ const titles = itemTitles(rss(item("A perfectly ordinary post title here")));
+ expect(titles).not.toContain("marketing — RSS Amplifier");
+ expect(titles).toEqual(["A perfectly ordinary post title here"]);
+ });
+
+ it("drops titles too short to be a subject and too long to be a prompt", () => {
+ const xml = rss(
+ [
+ item("Weeknotes"),
+ item("Ok"),
+ item("x".repeat(400)),
+ item("A title of a perfectly reasonable length for an article"),
+ ].join(""),
+ );
+ expect(itemTitles(xml)).toEqual([
+ "A title of a perfectly reasonable length for an article",
+ ]);
+ });
+
+ it("decodes what the feed escaped, so the subject reads as written", () => {
+ // Titles arrive from fifty thousand publishers carrying entities. A subject
+ // line reading "Don't" would be written into the article verbatim.
+ const xml = rss(
+ item("Don't Ship It & Hope — A Checklist For Releases"),
+ );
+ expect(itemTitles(xml)[0]).toBe(
+ "Don't Ship It & Hope — A Checklist For Releases",
+ );
+ });
+
+ it("reads a CDATA title", () => {
+ const xml = rss(item(""));
+ expect(itemTitles(xml)).toEqual(["A title wrapped in CDATA for safety"]);
+ });
+
+ it("returns nothing for a document that is not a feed", () => {
+ expect(itemTitles("not a feed at all")).toEqual([]);
+ expect(itemTitles("")).toEqual([]);
+ });
+});
+
+describe("topicSlug", () => {
+ it("matches the directory's own slugging", () => {
+ expect(topicSlug("Machine Learning")).toBe("machine-learning");
+ expect(topicSlug(" AI & robotics ")).toBe("ai-robotics");
+ expect(topicSlug("C++")).toBe("c");
+ });
+
+ it("slugs unusable input to empty rather than to a bad URL", () => {
+ // `/topics/.rss` is not a feed; the caller skips these instead of asking.
+ expect(topicSlug("!!!")).toBe("");
+ expect(topicSlug("")).toBe("");
+ });
+});
+
+describe("picking a subject", () => {
+ const ok = (body: string) =>
+ ({ ok: true, text: async () => body }) as unknown as Response;
+
+ it("returns a real post title from the first feed that has one", async () => {
+ const fetchImpl = (async () =>
+ ok(rss(item("How We Cut Our Build Time By Ninety Percent")))) as typeof fetch;
+
+ const found = await subjectFromTopicFeeds(["build systems"], fetchImpl);
+ expect(found?.subject).toBe("How We Cut Our Build Time By Ninety Percent");
+ });
+
+ it("moves on when a feed is empty or missing", async () => {
+ const seen: string[] = [];
+ const fetchImpl = (async (url: string) => {
+ seen.push(String(url));
+ if (seen.length === 1) return { ok: false } as Response;
+ if (seen.length === 2) return ok(rss(""));
+ return ok(rss(item("The third feed finally has something usable")));
+ }) as unknown as typeof fetch;
+
+ const found = await subjectFromTopicFeeds(["a", "b", "c"], fetchImpl);
+ expect(found?.subject).toBe("The third feed finally has something usable");
+ });
+
+ it("returns null rather than throwing when the directory is unreachable", async () => {
+ // This runs inside the publishing cron while it walks every active site. A
+ // slow directory must cost one site its guest post, not the sweep.
+ const fetchImpl = (async () => {
+ throw new Error("network down");
+ }) as unknown as typeof fetch;
+
+ expect(await subjectFromTopicFeeds(["anything"], fetchImpl)).toBeNull();
+ });
+
+ it("asks for nothing when there are no usable keywords", async () => {
+ let called = false;
+ const fetchImpl = (async () => {
+ called = true;
+ return ok(rss(""));
+ }) as unknown as typeof fetch;
+
+ expect(await subjectFromTopicFeeds(["!!!", ""], fetchImpl)).toBeNull();
+ expect(called).toBe(false);
+ });
+
+ it("does not ask the same feed twice for a duplicated keyword", async () => {
+ const asked: string[] = [];
+ const fetchImpl = (async (url: string) => {
+ asked.push(String(url));
+ return ok(rss(""));
+ }) as unknown as typeof fetch;
+
+ await subjectFromTopicFeeds(["Design", "design", " DESIGN "], fetchImpl);
+ expect(asked).toHaveLength(1);
+ });
+});