Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
36 changes: 33 additions & 3 deletions lib/lx/autoGuestPost.ts
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@

import type { SupabaseClient } from "@supabase/supabase-js";
import { findGuestPostOpportunities } from "./guestPostMatcher";
import { subjectFromTopicFeeds } from "./feedTopics";

/**
* One in every this many published posts is a guest post.
Expand Down Expand Up @@ -157,8 +158,37 @@ export async function planGuestPost(
};
}

// Every partner is either cooling off or has had all its crossed topics
// used. That is a healthy network doing its job, not an error — the caller
// publishes to the author's own blog instead.
// Every partner is either cooling off or has had all its crossed topics used.
//
// The crossings are finite — they are combinations of two fixed keyword
// lists — so a partner written for a few times exhausts them and the slot
// falls back to the author's own blog for ever after. Before giving up, take
// a subject from what the small web is actually publishing: a real post from
// an RSS Amplifier topic feed, picked at random, which the generator then
// writes a full article about. Nothing is copied; what the feed contributes
// is a subject somebody genuinely cared about this week rather than one
// assembled from two keyword lists.
for (const opportunity of opportunities) {
if (cooling.has(opportunity.partner_site_id)) continue;

const found = await subjectFromTopicFeeds(
// The partner's own suggested topics are the best guide to what their
// readers came for, and they are already crossed against ours.
opportunity.suggested_topics ?? [],
);
if (!found) continue;

const key = `${opportunity.partner_site_id}::${found.subject.trim().toLowerCase()}`;
if (taken.has(key)) continue;

return {
targetSiteId: opportunity.partner_site_id,
targetDomain: opportunity.partner_domain,
topic: found.subject,
};
}

// Nothing anywhere. A healthy network doing its job rather than an error —
// the caller publishes to the author's own blog instead.
return null;
}
176 changes: 176 additions & 0 deletions lib/lx/feedTopics.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,176 @@
// Guest-post subjects taken from what the small web is actually writing about.
//
// The crossed-seed topics that `guestPostMatcher` produces are combinations of
// two sites' own keywords — reliable, and finite. Once a partner has been
// written for a few times the crossings run out and the slot goes back to the
// author's own blog. This is the other source: a real, recently published post
// from an RSS Amplifier topic feed, used as the subject for a full article.
//
// The article is still entirely ours — the generator writes it from scratch on
// the subject. Nothing is copied. What the feed contributes is a subject that
// somebody in the niche genuinely cared about this week, rather than one
// assembled from two keyword lists.
//
// Three things this is careful about, and the first is the one that would be
// most embarrassing.

/** Where the directory lives. */
const RSSAMPLIFIER = "https://rssamplifier.com";

/**
* How long to wait for a topic feed.
*
* This runs inside the publishing cron, which is walking every active site. A
* slow directory must cost one site its guest post, not the whole sweep, so the
* budget is small and every failure path returns null.
*/
const TIMEOUT_MS = 3000;

/** Feeds tried before giving up on finding a usable subject. */
const MAX_FEEDS = 3;

/** A title shorter than this is not a subject — it is a label. */
const MIN_TITLE_LEN = 24;

/** And one longer than this is a paragraph that will not survive a prompt. */
const MAX_TITLE_LEN = 160;

/**
* A keyword as it appears in a topic URL.
*
* Matches the directory's own slugging — lowercase, non-alphanumerics to
* hyphens, collapsed. A keyword that slugs to nothing is skipped rather than
* requested, since `/topics/.rss` is not a feed.
*/
export function topicSlug(keyword: string): string {
return (keyword ?? "")
.toLowerCase()
.replace(/[^a-z0-9]+/g, "-")
.replace(/^-+|-+$/g, "");
}

/**
* Titles of real posts in a topic feed.
*
* Parsed with a regex rather than an XML library on purpose: this is one known
* document shape from one known publisher, the only field wanted is the title,
* and a parse failure here must degrade to "no subject today" rather than
* throw inside a cron. Dragging an XML dependency into the worker to read one
* element would be the worse trade.
*
* @param xml an RSS document
* @returns post titles, channel title and sponsored items removed
*/
export function itemTitles(xml: string): string[] {
const out: string[] = [];

// Item blocks only — this is what keeps the channel's own <title> (the name
// of the topic) out of the candidate list.
const items = xml.match(/<item\b[\s\S]*?<\/item>/gi) ?? [];

for (const item of items) {
// **Sponsored items are ours.** The directory's feeds now carry CrawlProof
// ad fills as syndication items, so without this the cron could pick one of
// our own advertisements and commission a guest post about it — an ad,
// laundered into editorial, published on a partner's blog under our name.
// Both the category and the title suffix are checked because either alone
// is a single point of failure for something that must never happen.
if (/<category>\s*Sponsored\s*<\/category>/i.test(item)) continue;

const raw = item.match(/<title>([\s\S]*?)<\/title>/i)?.[1];
if (!raw) continue;

const title = decodeXml(raw).trim();
if (/\(sponsored\)\s*$/i.test(title)) continue;

if (title.length < MIN_TITLE_LEN || title.length > MAX_TITLE_LEN) continue;
out.push(title);
}

return out;
}

/**
* Undo the escaping a feed document applies, and nothing else.
*
* The five predefined entities plus numeric references, because titles arrive
* carrying `&#8594;` and `&apos;` from fifty thousand different publishers and
* a subject line reading "Don&apos;t" would be written into the article.
*/
function decodeXml(s: string): string {
return s
.replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, "$1")
.replace(/&#(\d+);/g, (_m, d) => String.fromCodePoint(Number(d)))
.replace(/&#x([0-9a-f]+);/gi, (_m, h) => String.fromCodePoint(parseInt(h, 16)))
.replace(/&lt;/g, "<")
.replace(/&gt;/g, ">")
.replace(/&quot;/g, '"')
.replace(/&apos;/g, "'")
.replace(/&amp;/g, "&");
}

/**
* Pick a subject at random from what the directory is carrying.
*
* Random rather than ranked, deliberately. Ranking these would mean deciding
* that one publisher's headline is a better subject than another's on evidence
* we do not have — and the failure it would cause is worse than the one it
* would prevent: a stable ranking over a slow-moving feed writes about the same
* thing repeatedly, which is exactly what this source exists to avoid.
*
* @param keywords subject words to look for, tried in random order
* @param fetchImpl injected by the tests
* @returns a subject, or null to let the caller fall back
*/
export async function subjectFromTopicFeeds(
keywords: string[],
fetchImpl: typeof fetch = fetch,
): Promise<{ subject: string; topic: string } | null> {
const slugs = shuffle(
Array.from(new Set((keywords ?? []).map(topicSlug).filter(Boolean))),
).slice(0, MAX_FEEDS);

for (const slug of slugs) {
const titles = await titlesFor(slug, fetchImpl);
if (titles.length === 0) continue;

const subject = titles[Math.floor(Math.random() * titles.length)];
return { subject, topic: slug.replace(/-/g, " ") };
}

return null;
}

/**
* @param slug
* @param fetchImpl
* @returns usable titles, or [] for any failure at all
*/
async function titlesFor(slug: string, fetchImpl: typeof fetch): Promise<string[]> {
try {
const res = await fetchImpl(`${RSSAMPLIFIER}/topics/${encodeURIComponent(slug)}.rss`, {
signal: AbortSignal.timeout(TIMEOUT_MS),
headers: { accept: "application/rss+xml, application/xml;q=0.9" },
});
if (!res.ok) return [];
return itemTitles(await res.text());
} catch {
// A missing topic, a slow directory, a malformed document — all the same
// answer. The caller has an ordinary post to publish instead.
return [];
}
}

/**
* @template T
* @param xs
* @returns a shuffled copy
*/
function shuffle<T>(xs: T[]): T[] {
const out = [...xs];
for (let i = out.length - 1; i > 0; i -= 1) {
const j = Math.floor(Math.random() * (i + 1));
[out[i], out[j]] = [out[j], out[i]];
}
return out;
}
155 changes: 155 additions & 0 deletions tests/contract/feed-topics.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,155 @@
import { describe, it, expect } from "vitest";

import {
itemTitles,
subjectFromTopicFeeds,
topicSlug,
} from "@/lib/lx/feedTopics";

const rss = (items: string) => `<?xml version="1.0"?>
<rss version="2.0"><channel>
<title>marketing — RSS Amplifier</title>
${items}
</channel></rss>`;

const item = (title: string, extra = "") =>
`<item><title>${title}</title>${extra}</item>`;

describe("reading subjects out of a topic feed", () => {
it("never offers one of our own ads as a subject", () => {
// The failure this exists to prevent, and it is not hypothetical: the
// directory's feeds now carry CrawlProof ad fills as syndication items.
// Without this filter the cron could commission a guest post *about an
// advertisement* and publish it on a partner's blog under our name — an ad
// laundered into editorial. Both signals are checked because either alone
// is a single point of failure for something that must never happen.
const xml = rss(
[
item(
"Home of the Agentic Pull Request (Sponsored)",
"<category>Sponsored</category>",
),
item("Another Paid Placement Dressed As A Post", "<category>Sponsored</category>"),
item("A Sponsored Title With No Category Element (Sponsored)"),
item("Why I Said Yes to a $1,000-an-Hour Coach"),
].join(""),
);

expect(itemTitles(xml)).toEqual(["Why I Said Yes to a $1,000-an-Hour Coach"]);
});

it("leaves the channel's own title out of the candidates", () => {
// The topic name is a label, not a subject. It sits outside <item>, which
// is why the parse is scoped to item blocks.
const titles = itemTitles(rss(item("A perfectly ordinary post title here")));
expect(titles).not.toContain("marketing — RSS Amplifier");
expect(titles).toEqual(["A perfectly ordinary post title here"]);
});

it("drops titles too short to be a subject and too long to be a prompt", () => {
const xml = rss(
[
item("Weeknotes"),
item("Ok"),
item("x".repeat(400)),
item("A title of a perfectly reasonable length for an article"),
].join(""),
);
expect(itemTitles(xml)).toEqual([
"A title of a perfectly reasonable length for an article",
]);
});

it("decodes what the feed escaped, so the subject reads as written", () => {
// Titles arrive from fifty thousand publishers carrying entities. A subject
// line reading "Don&apos;t" would be written into the article verbatim.
const xml = rss(
item("Don&apos;t Ship It &amp; Hope &#8212; A Checklist For Releases"),
);
expect(itemTitles(xml)[0]).toBe(
"Don't Ship It & Hope — A Checklist For Releases",
);
});

it("reads a CDATA title", () => {
const xml = rss(item("<![CDATA[A title wrapped in CDATA for safety]]>"));
expect(itemTitles(xml)).toEqual(["A title wrapped in CDATA for safety"]);
});

it("returns nothing for a document that is not a feed", () => {
expect(itemTitles("<html><body>not a feed at all</body></html>")).toEqual([]);
expect(itemTitles("")).toEqual([]);
});
});

describe("topicSlug", () => {
it("matches the directory's own slugging", () => {
expect(topicSlug("Machine Learning")).toBe("machine-learning");
expect(topicSlug(" AI & robotics ")).toBe("ai-robotics");
expect(topicSlug("C++")).toBe("c");
});

it("slugs unusable input to empty rather than to a bad URL", () => {
// `/topics/.rss` is not a feed; the caller skips these instead of asking.
expect(topicSlug("!!!")).toBe("");
expect(topicSlug("")).toBe("");
});
});

describe("picking a subject", () => {
const ok = (body: string) =>
({ ok: true, text: async () => body }) as unknown as Response;

it("returns a real post title from the first feed that has one", async () => {
const fetchImpl = (async () =>
ok(rss(item("How We Cut Our Build Time By Ninety Percent")))) as typeof fetch;

const found = await subjectFromTopicFeeds(["build systems"], fetchImpl);
expect(found?.subject).toBe("How We Cut Our Build Time By Ninety Percent");
});

it("moves on when a feed is empty or missing", async () => {
const seen: string[] = [];
const fetchImpl = (async (url: string) => {
seen.push(String(url));
if (seen.length === 1) return { ok: false } as Response;
if (seen.length === 2) return ok(rss(""));
return ok(rss(item("The third feed finally has something usable")));
}) as unknown as typeof fetch;

const found = await subjectFromTopicFeeds(["a", "b", "c"], fetchImpl);
expect(found?.subject).toBe("The third feed finally has something usable");
});

it("returns null rather than throwing when the directory is unreachable", async () => {
// This runs inside the publishing cron while it walks every active site. A
// slow directory must cost one site its guest post, not the sweep.
const fetchImpl = (async () => {
throw new Error("network down");
}) as unknown as typeof fetch;

expect(await subjectFromTopicFeeds(["anything"], fetchImpl)).toBeNull();
});

it("asks for nothing when there are no usable keywords", async () => {
let called = false;
const fetchImpl = (async () => {
called = true;
return ok(rss(""));
}) as unknown as typeof fetch;

expect(await subjectFromTopicFeeds(["!!!", ""], fetchImpl)).toBeNull();
expect(called).toBe(false);
});

it("does not ask the same feed twice for a duplicated keyword", async () => {
const asked: string[] = [];
const fetchImpl = (async (url: string) => {
asked.push(String(url));
return ok(rss(""));
}) as unknown as typeof fetch;

await subjectFromTopicFeeds(["Design", "design", " DESIGN "], fetchImpl);
expect(asked).toHaveLength(1);
});
});
Loading