diff --git a/app/(app)/dashboard/autoblog/crawler/page.tsx b/app/(app)/dashboard/autoblog/crawler/page.tsx
new file mode 100644
index 00000000..f056fa4f
--- /dev/null
+++ b/app/(app)/dashboard/autoblog/crawler/page.tsx
@@ -0,0 +1,168 @@
+// Live view of the feed crawler, modelled on rssamplifier.com/crawlstats.
+//
+// The autoblog cites real posts from RSS Amplifier topic feeds, and until the
+// crawl was daemonized that source was invisible: a topic going missing looked
+// identical to a topic nobody had configured. This is where that difference
+// becomes visible.
+//
+// Deliberately read-only and unconfigurable. The feed list is derived from
+// every active site's master keywords on each sweep, so there is nothing here
+// to add or remove — a subject added to a blog appears within a tick. An
+// "add a feed" button would imply a curation step that does not exist, and
+// would create a queue of requests waiting for a human, which is the thing
+// the automation is meant to remove.
+
+import { serviceClient } from "@/lib/supabase/service";
+import { ago, loadCrawlerStats } from "@/lib/lx/feedCrawlStats";
+
+// Always fresh: a status page served from a cache reports the cache's health,
+// not the crawler's, and would go on saying "healthy" through an outage.
+export const dynamic = "force-dynamic";
+
+function Stat({
+ label,
+ value,
+ note,
+}: {
+ label: string;
+ value: string;
+ note?: string;
+}) {
+ return (
+
+ Live view of the daemon that reads the RSS Amplifier topic feeds the
+ autoblog cites from. The feed list is derived from every active
+ site's master keywords — nothing here is curated by hand. The same
+ numbers are available as{" "}
+
+ JSON
+
+ .
+
+
+ {stats.stalled ? (
+
+ Stalled. {stats.dueNow.toLocaleString()} feed(s) are
+ due and nothing has been read successfully since{" "}
+ {ago(stats.lastSuccessAt, now)}. Either the worker is not running or
+ the directory is unreachable — this page cannot tell which, and the
+ two have different fixes.
+
+ ) : null}
+
+
+
+
+
+
+
+
+
+ Last successful fetch {ago(stats.lastSuccessAt, now)} ·{" "}
+ {stats.neverFetched.toLocaleString()} never fetched · generated{" "}
+ {stats.generatedAt}
+
+
+
Feeds
+
+ One row per subject any active blog covers. A feed is re-read at most
+ every 6 hours, least-recently-fetched first, 25 per sweep — so the work
+ per tick is fixed however many subjects the platform grows to.
+
+
+ {sources.length === 0 ? (
+
+ No feeds yet. The first sweep derives them from the active sites'
+ master keywords; if this stays empty, no site has master keywords set.
+
+ );
+}
diff --git a/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx b/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx
index e21e7329..5ab4bc9f 100644
--- a/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx
+++ b/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx
@@ -23,8 +23,10 @@ type Existing = {
target_audiences: string[];
description: string;
seed_keywords: string[];
+ master_keywords: string[];
modifiers: string[];
preserve_keywords: boolean;
+ ads_enabled: boolean;
keywords: string[];
seo_title: string | null;
seo_description: string | null;
@@ -109,6 +111,14 @@ export function SetupForm({
(initial?.target_audiences ?? []).join(", "),
);
const [description, setDescription] = useState(initial?.description ?? "");
+ const [masterKeywords, setMasterKeywords] = useState(
+ // Falls back to the seed list for any site the backfill has not reached,
+ // so the field is never blank on a project that already has subjects.
+ (initial?.master_keywords?.length
+ ? initial.master_keywords
+ : (initial?.seed_keywords ?? [])
+ ).join(", "),
+ );
const [seedKeywords, setSeedKeywords] = useState(
(initial?.seed_keywords ?? []).join(", "),
);
@@ -118,6 +128,11 @@ export function SetupForm({
const [preserveKeywords, setPreserveKeywords] = useState(
initial?.preserve_keywords ?? false,
);
+ // Default ON, matching the column default. A new project is in the network
+ // unless its owner takes the tick out.
+ const [adsEnabled, setAdsEnabled] = useState(
+ initial?.ads_enabled ?? true,
+ );
const [keywords, setKeywords] = useState(
(initial?.keywords ?? []).join("\n"),
);
@@ -425,6 +440,13 @@ export function SetupForm({
// Each line is a CSV row ",". For seed-building and
// dedupe we only care about the keyword half (everything before the
// first comma).
+ function masterKeywordsAsArray(): string[] {
+ return masterKeywords
+ .split(",")
+ .map((s) => s.trim())
+ .filter(Boolean);
+ }
+
function keywordsAsArray(): string[] {
return keywords
.split("\n")
@@ -569,9 +591,11 @@ export function SetupForm({
niche,
targetAudiences: audiences,
description,
+ masterKeywords,
seedKeywords,
modifiers,
preserveKeywords,
+ adsEnabled,
keywords,
seoTitle,
seoDescription,
@@ -843,6 +867,34 @@ export function SetupForm({
onChange={(e) => setDescription(e.target.value)}
/>
+
+
+ setMasterKeywords(e.target.value)}
+ />
+
+ The subjects this blog covers, and the list it always rebuilds from.
+ The planner splits every research run evenly across{" "}
+ all of them and fills whichever is furthest behind
+ first, so a subject cannot run away with the schedule. Each one is
+ researched crossed with a modifier below — never on its own.{" "}
+ {masterKeywordsAsArray().length > 12 ? (
+
+ {masterKeywordsAsArray().length} entered — only the first 12 are
+ kept.
+
+ ) : (
+ <>{masterKeywordsAsArray().length} subject(s).>
+ )}
+
+
+
+
+
+
+
diff --git a/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx b/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx
index fa990710..b5201701 100644
--- a/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx
+++ b/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx
@@ -7,7 +7,7 @@ import { SetupForm } from "./form";
export const metadata = { title: "Autoblog · Setup" };
const SITE_COLUMNS =
- "id, domain, blog_root_url, sitemap_url, niche, target_audiences, description, seed_keywords, modifiers, preserve_keywords, keywords, seo_title, seo_description, tone, competitors, webhook_url, webhook_secret, daily_article_count, publish_days, publish_hour, internal_links_per_article, backlinks_enabled, external_links_per_article, banner_style, status";
+ "id, domain, blog_root_url, sitemap_url, niche, target_audiences, description, seed_keywords, master_keywords, modifiers, preserve_keywords, ads_enabled, keywords, seo_title, seo_description, tone, competitors, webhook_url, webhook_secret, daily_article_count, publish_days, publish_hour, internal_links_per_article, backlinks_enabled, external_links_per_article, banner_style, status";
export default async function AutoblogSetupPage({
params,
diff --git a/app/actions/linkExchange.ts b/app/actions/linkExchange.ts
index 514ff120..1d109971 100644
--- a/app/actions/linkExchange.ts
+++ b/app/actions/linkExchange.ts
@@ -289,6 +289,9 @@ export type SiteInput = {
niche: string;
targetAudiences: string;
description: string;
+ // Comma-separated 3-12 subjects this blog covers. The durable list the
+ // keyword planner allocates evenly across; see lib/lx/topicPlan.
+ masterKeywords: string;
// Comma-separated 1-3 word head terms for DataForSEO expansion.
seedKeywords: string;
// Comma-separated tail terms ("payments", "merchant account") that the
@@ -297,6 +300,9 @@ export type SiteInput = {
// When true, Refetch flows skip overwriting the keywords text[]. Used
// after a hand-curated build via seeds × modifiers.
preserveKeywords: boolean;
+ // Network opt-in: ad units and partner/directory links in published
+ // articles. Default true — see the migration note on lx_site.ads_enabled.
+ adsEnabled: boolean;
// Comma-separated. parseList() into a text[].
keywords: string;
seoTitle: string;
@@ -360,6 +366,12 @@ export async function createOrUpdateSite(
const seedKeywords = parseList(input.seedKeywords, 50, 40);
const modifiers = parseList(input.modifiers, 20, 40);
const preserveKeywords = !!input.preserveKeywords;
+ // The DB constraint caps this at 12; parse to the same number so an
+ // oversized paste is trimmed here rather than rejected as a write error the
+ // form cannot explain.
+ const masterKeywords = parseList(input.masterKeywords, 12, MAX.keyword);
+ // Absent reads as opted in, matching the column default.
+ const adsEnabled = input.adsEnabled !== false;
// Keywords use parseKeywordRows so each row keeps its `,` hint.
// Allow up to ~80 chars per row to fit "keyword phrase here,12345".
const keywords = parseKeywordRows(input.keywords, MAX.keywords, MAX.keyword + 20);
@@ -433,6 +445,8 @@ export async function createOrUpdateSite(
target_audiences: audiences,
description,
seed_keywords: seedKeywords,
+ master_keywords: masterKeywords,
+ ads_enabled: adsEnabled,
modifiers,
preserve_keywords: preserveKeywords,
keywords,
@@ -489,6 +503,8 @@ export async function createOrUpdateSite(
target_audiences: audiences,
description,
seed_keywords: seedKeywords,
+ master_keywords: masterKeywords,
+ ads_enabled: adsEnabled,
modifiers,
preserve_keywords: preserveKeywords,
keywords,
@@ -534,6 +550,8 @@ export async function createOrUpdateSite(
target_audiences: audiences,
description,
seed_keywords: seedKeywords,
+ master_keywords: masterKeywords,
+ ads_enabled: adsEnabled,
modifiers,
preserve_keywords: preserveKeywords,
keywords,
diff --git a/app/api/lx/feed-crawl/status/route.ts b/app/api/lx/feed-crawl/status/route.ts
new file mode 100644
index 00000000..d53cbdba
--- /dev/null
+++ b/app/api/lx/feed-crawl/status/route.ts
@@ -0,0 +1,20 @@
+import { NextResponse } from "next/server";
+import { serviceClient } from "@/lib/supabase/service";
+import { loadCrawlerStats } from "@/lib/lx/feedCrawlStats";
+
+export const runtime = "nodejs";
+// A cached status response reports the cache's health rather than the
+// crawler's, and would keep answering "fine" straight through an outage.
+export const dynamic = "force-dynamic";
+
+/**
+ * Counts only.
+ *
+ * The per-feed rows stay on the dashboard page behind a session: this is the
+ * shape a monitor polls, and it has no reason to carry the error strings some
+ * upstream server wrote.
+ */
+export async function GET() {
+ const { stats } = await loadCrawlerStats(serviceClient());
+ return NextResponse.json(stats, { headers: { "cache-control": "no-store" } });
+}
diff --git a/lib/lx/feedCrawl.ts b/lib/lx/feedCrawl.ts
new file mode 100644
index 00000000..f20c4ae5
--- /dev/null
+++ b/lib/lx/feedCrawl.ts
@@ -0,0 +1,285 @@
+// The daemon side of the directory feeds.
+//
+// `postsFromTopicFeeds` used to fetch RSS live, inside article delivery, on a
+// three-second budget. That is a third-party HTTP call on the critical path of
+// the one operation a customer pays for, and it made the source invisible: a
+// topic feed going missing showed up as an empty block and nothing else.
+//
+// This moves the fetching onto the worker's own schedule and leaves a record
+// of every attempt. Delivery then reads rows. The two halves fail
+// independently — a dead directory ages the cache instead of slowing down a
+// publish, and a publish never waits on anybody else's server.
+//
+// The source list is derived on every sweep from the master_keywords of every
+// active site. Nobody maintains it. A blog that adds a subject has that
+// subject's feed crawled on the next tick, which is the behaviour "no human
+// intervention" actually requires — a curated feed list is a queue of requests
+// waiting for somebody.
+
+import type { SupabaseClient } from "@supabase/supabase-js";
+import { itemEntries } from "./feedTopics";
+import { resolveMasters, tokens } from "./topicPlan";
+
+const RSSAMPLIFIER = "https://rssamplifier.com";
+
+/** Per-feed budget. Generous next to delivery's, because nothing waits on it. */
+const FETCH_TIMEOUT_MS = 12_000;
+
+/** Feeds read per sweep. Bounded so one tick cannot run for an hour. */
+const PER_SWEEP = 25;
+
+/** Don't re-read a feed more often than this. */
+export const REFRESH_AFTER_MS = 6 * 60 * 60 * 1000; // 6h
+
+/**
+ * Consecutive failures before a source is parked.
+ *
+ * Parked, not deleted. A deleted row is re-derived on the very next sweep from
+ * the same master keyword and starts failing again immediately — an infinite
+ * retry wearing the costume of a clean table. The row stays so the status page
+ * can say "this topic has no feed, here is why".
+ */
+const GIVE_UP_AFTER = 8;
+
+/** Items kept per source. Older ones are pruned so the table stays bounded. */
+const KEEP_PER_SOURCE = 40;
+
+/**
+ * The directory's slug for a subject.
+ *
+ * Mirrors `topicSlug` in feedTopics but drops the stopword-free tokenisation
+ * through `tokens` first, so a master keyword of "merchant account payments"
+ * slugs to something the directory actually has a topic for rather than a long
+ * phrase that will always 404.
+ */
+export function topicFor(master: string): string | null {
+ const parts = tokens(master);
+ if (parts.length === 0) {
+ // A short subject ("iptv", "weed") has no tokens over the length floor but
+ // is still a real topic — fall back to the raw slug rather than dropping
+ // exactly the short, high-value subjects.
+ const raw = (master ?? "").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "");
+ return raw || null;
+ }
+ return parts[0];
+}
+
+export type FeedSweepResult = {
+ sources: number;
+ fetched: number;
+ succeeded: number;
+ newItems: number;
+ gaveUp: number;
+};
+
+/**
+ * Make sure every subject any active site covers has a source row.
+ *
+ * Insert-only, ignoring conflicts: a topic already present keeps its history,
+ * and two sites sharing a subject share one crawl.
+ */
+export async function syncFeedSources(
+ supabase: SupabaseClient,
+): Promise {
+ const { data: sites } = await supabase
+ .from("lx_site")
+ .select("master_keywords, seed_keywords")
+ .eq("status", "active");
+
+ const topics = new Set();
+ for (const site of (sites ?? []) as Array<{
+ master_keywords: string[] | null;
+ seed_keywords: string[] | null;
+ }>) {
+ for (const master of resolveMasters(site)) {
+ const topic = topicFor(master);
+ if (topic) topics.add(topic);
+ }
+ }
+ if (topics.size === 0) return 0;
+
+ const rows = Array.from(topics).map((topic) => ({
+ topic,
+ url: `${RSSAMPLIFIER}/topics/${encodeURIComponent(topic)}.rss`,
+ }));
+
+ const { error } = await supabase
+ .from("lx_feed_source")
+ .upsert(rows, { onConflict: "topic", ignoreDuplicates: true });
+ if (error) console.warn("[lx] feed source sync:", error.message);
+
+ return topics.size;
+}
+
+/**
+ * Read the feeds that are due.
+ *
+ * Least-recently-fetched first, capped, so the sweep is a fixed amount of work
+ * regardless of how many subjects the platform covers.
+ */
+export async function crawlFeeds(
+ supabase: SupabaseClient,
+ fetchImpl: typeof fetch = fetch,
+): Promise {
+ const sources = await syncFeedSources(supabase);
+ const due = new Date(Date.now() - REFRESH_AFTER_MS).toISOString();
+
+ const { data: batch } = await supabase
+ .from("lx_feed_source")
+ .select("id, topic, url, consecutive_failures")
+ .eq("status", "active")
+ .or(`last_fetch_at.is.null,last_fetch_at.lt.${due}`)
+ .order("last_fetch_at", { ascending: true, nullsFirst: true })
+ .limit(PER_SWEEP);
+
+ const result: FeedSweepResult = {
+ sources,
+ fetched: 0,
+ succeeded: 0,
+ newItems: 0,
+ gaveUp: 0,
+ };
+
+ for (const source of (batch ?? []) as Array<{
+ id: string;
+ topic: string;
+ url: string;
+ consecutive_failures: number;
+ }>) {
+ result.fetched += 1;
+ const now = new Date().toISOString();
+
+ let status: number | null = null;
+ let entries: Array<{ title: string; link: string | null }> = [];
+ let error: string | null = null;
+
+ try {
+ const res = await fetchImpl(source.url, {
+ signal: AbortSignal.timeout(FETCH_TIMEOUT_MS),
+ headers: { accept: "application/rss+xml, application/xml;q=0.9" },
+ });
+ status = res.status;
+ if (res.ok) {
+ entries = itemEntries(await res.text()).filter((e) => e.link);
+ } else {
+ error = `HTTP ${res.status}`;
+ }
+ } catch (err) {
+ error = err instanceof Error ? err.message : String(err);
+ }
+
+ // A 200 carrying no items is not a success. It is what the directory
+ // returns for a topic nobody publishes under, and counting it as one
+ // would let a permanently empty topic sit at the front of the
+ // least-recently-fetched queue for ever, crowding out real feeds.
+ const ok = error === null && entries.length > 0;
+ if (!ok && !error) error = "no usable items";
+
+ if (ok) {
+ result.succeeded += 1;
+ const rows = entries.slice(0, KEEP_PER_SOURCE).map((e) => ({
+ source_id: source.id,
+ title: e.title,
+ link: e.link as string,
+ }));
+ const { data: inserted } = await supabase
+ .from("lx_feed_item")
+ .upsert(rows, { onConflict: "source_id,link", ignoreDuplicates: true })
+ .select("id");
+ result.newItems += inserted?.length ?? 0;
+ await pruneItems(supabase, source.id);
+ }
+
+ const failures = ok ? 0 : source.consecutive_failures + 1;
+ const givingUp = failures >= GIVE_UP_AFTER;
+ if (givingUp) result.gaveUp += 1;
+
+ await supabase
+ .from("lx_feed_source")
+ .update({
+ last_fetch_at: now,
+ ...(ok ? { last_success_at: now, item_count: entries.length } : {}),
+ last_status: status,
+ last_error: ok ? null : error,
+ consecutive_failures: failures,
+ status: givingUp ? "given_up" : "active",
+ updated_at: now,
+ })
+ .eq("id", source.id);
+ }
+
+ return result;
+}
+
+/** Keep the newest KEEP_PER_SOURCE items and drop the rest. */
+async function pruneItems(supabase: SupabaseClient, sourceId: string): Promise {
+ const { data: keep } = await supabase
+ .from("lx_feed_item")
+ .select("id")
+ .eq("source_id", sourceId)
+ .order("first_seen_at", { ascending: false })
+ .limit(KEEP_PER_SOURCE);
+ const ids = (keep ?? []).map((r: { id: string }) => r.id);
+ if (ids.length < KEEP_PER_SOURCE) return;
+ await supabase
+ .from("lx_feed_item")
+ .delete()
+ .eq("source_id", sourceId)
+ .not("id", "in", `(${ids.join(",")})`);
+}
+
+/**
+ * Cached posts for a set of subjects.
+ *
+ * The read side of the crawl, used by article delivery. Returns nothing rather
+ * than falling back to a live fetch: a cache miss here means the sweep has not
+ * reached that topic yet, and the correct behaviour is an article without a
+ * citation block, not a publish that blocks on somebody else's server. The
+ * next article gets the block.
+ */
+export async function cachedFeedPosts(
+ supabase: SupabaseClient,
+ masters: string[],
+ limit = 3,
+): Promise> {
+ const topics = Array.from(
+ new Set(masters.map(topicFor).filter((t): t is string => !!t)),
+ );
+ if (topics.length === 0) return [];
+
+ const { data } = await supabase
+ .from("lx_feed_item")
+ .select("title, link, source:lx_feed_source!inner(topic)")
+ .in("source.topic", topics)
+ .order("first_seen_at", { ascending: false })
+ .limit(limit * 12);
+
+ type Row = { title: string; link: string; source: { topic: string } | { topic: string }[] };
+ const byTopic = new Map>();
+ for (const row of (data ?? []) as Row[]) {
+ const source = Array.isArray(row.source) ? row.source[0] : row.source;
+ const topic = source?.topic;
+ if (!topic) continue;
+ const bucket = byTopic.get(topic) ?? [];
+ bucket.push({ title: row.title, link: row.link, topic });
+ byTopic.set(topic, bucket);
+ }
+
+ // One per topic before a second from any, so three citations show three
+ // subjects rather than three posts from whichever feed is busiest.
+ const out: Array<{ title: string; link: string; topic: string }> = [];
+ const queues = Array.from(byTopic.values());
+ let live = true;
+ while (live && out.length < limit) {
+ live = false;
+ for (const queue of queues) {
+ if (out.length >= limit) break;
+ const next = queue.shift();
+ if (next) {
+ out.push(next);
+ live = true;
+ }
+ }
+ }
+ return out;
+}
diff --git a/lib/lx/feedCrawlStats.ts b/lib/lx/feedCrawlStats.ts
new file mode 100644
index 00000000..745bbcad
--- /dev/null
+++ b/lib/lx/feedCrawlStats.ts
@@ -0,0 +1,119 @@
+// The numbers behind the crawler status page.
+//
+// Kept out of the page so the HTML view and the JSON endpoint cannot drift —
+// a status page and a machine-readable feed of the same status that disagree
+// are worse than having only one of them.
+
+import type { SupabaseClient } from "@supabase/supabase-js";
+import { REFRESH_AFTER_MS } from "./feedCrawl";
+
+export type FeedSourceRow = {
+ id: string;
+ topic: string;
+ url: string;
+ status: string;
+ last_fetch_at: string | null;
+ last_success_at: string | null;
+ last_status: number | null;
+ last_error: string | null;
+ item_count: number;
+ consecutive_failures: number;
+};
+
+export type CrawlerStats = {
+ sources: number;
+ active: number;
+ gaveUp: number;
+ dueNow: number;
+ neverFetched: number;
+ erroring: number;
+ items: number;
+ newItems24h: number;
+ lastSuccessAt: string | null;
+ /**
+ * The daemon looks stopped.
+ *
+ * Sources are due and nothing has succeeded in a while. Stated as a
+ * suspicion rather than a fact because this cannot distinguish "the worker
+ * is down" from "the directory is down" — and the operator's next step
+ * differs, so claiming to know which would send them the wrong way.
+ */
+ stalled: boolean;
+ generatedAt: string;
+};
+
+/** Nothing succeeded in this long, with work waiting, reads as stalled. */
+const STALL_AFTER_MS = 2 * 60 * 60 * 1000; // 2h — four missed 30-min ticks
+
+export async function loadCrawlerStats(
+ supabase: SupabaseClient,
+): Promise<{ stats: CrawlerStats; sources: FeedSourceRow[] }> {
+ const now = Date.now();
+ const dueBefore = new Date(now - REFRESH_AFTER_MS).toISOString();
+ const since24h = new Date(now - 24 * 60 * 60 * 1000).toISOString();
+
+ const [{ data: sources }, itemCount, new24h] = await Promise.all([
+ supabase
+ .from("lx_feed_source")
+ .select(
+ "id, topic, url, status, last_fetch_at, last_success_at, last_status, last_error, item_count, consecutive_failures",
+ )
+ .order("last_fetch_at", { ascending: true, nullsFirst: true })
+ .limit(500),
+ supabase
+ .from("lx_feed_item")
+ .select("id", { count: "exact", head: true })
+ .then((r) => r.count ?? 0),
+ supabase
+ .from("lx_feed_item")
+ .select("id", { count: "exact", head: true })
+ .gte("first_seen_at", since24h)
+ .then((r) => r.count ?? 0),
+ ]);
+
+ const rows = (sources ?? []) as FeedSourceRow[];
+ const active = rows.filter((r) => r.status === "active");
+ const dueNow = active.filter(
+ (r) => !r.last_fetch_at || r.last_fetch_at < dueBefore,
+ ).length;
+
+ const lastSuccessAt = rows
+ .map((r) => r.last_success_at)
+ .filter((v): v is string => !!v)
+ .sort()
+ .at(-1) ?? null;
+
+ const stalled =
+ dueNow > 0 &&
+ (!lastSuccessAt || now - Date.parse(lastSuccessAt) > STALL_AFTER_MS);
+
+ return {
+ stats: {
+ sources: rows.length,
+ active: active.length,
+ gaveUp: rows.filter((r) => r.status === "given_up").length,
+ dueNow,
+ neverFetched: rows.filter((r) => !r.last_fetch_at).length,
+ erroring: rows.filter((r) => r.consecutive_failures > 0).length,
+ items: itemCount,
+ newItems24h: new24h,
+ lastSuccessAt,
+ stalled,
+ generatedAt: new Date(now).toISOString(),
+ },
+ sources: rows,
+ };
+}
+
+/** "15m ago", "3h ago", "never" — the only formatting this page needs. */
+export function ago(iso: string | null, now = Date.now()): string {
+ if (!iso) return "never";
+ const ms = now - Date.parse(iso);
+ if (!Number.isFinite(ms) || ms < 0) return "just now";
+ const mins = Math.floor(ms / 60000);
+ if (mins < 1) return "just now";
+ if (mins < 60) return `${mins}m ago`;
+ const hours = Math.floor(mins / 60);
+ if (hours < 24) return `${hours}h ago`;
+ return `${Math.floor(hours / 24)}d ago`;
+}
diff --git a/lib/lx/feedTopics.ts b/lib/lx/feedTopics.ts
index 80215bb9..6d67d574 100644
--- a/lib/lx/feedTopics.ts
+++ b/lib/lx/feedTopics.ts
@@ -62,7 +62,26 @@ export function topicSlug(keyword: string): string {
* @returns post titles, channel title and sponsored items removed
*/
export function itemTitles(xml: string): string[] {
- const out: string[] = [];
+ return itemEntries(xml).map((e) => e.title);
+}
+
+/** A usable post from a topic feed. */
+export type FeedEntry = { title: string; link: string | null };
+
+/**
+ * Posts in a topic feed, with their links.
+ *
+ * The link is what a *citation* needs; `itemTitles` only ever needed the
+ * subject, and is now a projection of this. Keeping one parser means the
+ * sponsored-item exclusion below cannot be enforced on one path and forgotten
+ * on the other — and the path that carries links out to a published page is
+ * precisely the one where forgetting it would be worst.
+ *
+ * @param xml an RSS document
+ * @returns entries, channel-level and sponsored items removed
+ */
+export function itemEntries(xml: string): FeedEntry[] {
+ const out: FeedEntry[] = [];
// Item blocks only — this is what keeps the channel's own (the name
// of the topic) out of the candidate list.
@@ -84,12 +103,75 @@ export function itemTitles(xml: string): string[] {
if (/\(sponsored\)\s*$/i.test(title)) continue;
if (title.length < MIN_TITLE_LEN || title.length > MAX_TITLE_LEN) continue;
- out.push(title);
+
+ const href = decodeXml(item.match(/([\s\S]*?)<\/link>/i)?.[1] ?? "").trim();
+ // Only absolute http(s) links survive. A feed carrying a relative link, a
+ // javascript: URL or a bare guid must not be able to put either into an
+ // anchor on a customer's published page.
+ const link = /^https?:\/\/\S+$/i.test(href) ? href : null;
+
+ out.push({ title, link });
}
return out;
}
+/**
+ * Real posts from the directory on the subjects given, with links.
+ *
+ * Unlike `subjectFromTopicFeeds`, which wants one subject to write about, this
+ * wants several posts to cite — so it reads across every requested topic
+ * rather than stopping at the first that answers, and returns only entries
+ * that carry a usable link.
+ *
+ * @param keywords subject words, tried in random order
+ * @param limit most entries to return
+ * @param fetchImpl injected by the tests
+ */
+export async function postsFromTopicFeeds(
+ keywords: string[],
+ limit = 3,
+ fetchImpl: typeof fetch = fetch,
+): Promise> {
+ const slugs = shuffle(
+ Array.from(new Set((keywords ?? []).map(topicSlug).filter(Boolean))),
+ ).slice(0, MAX_FEEDS);
+
+ const out: Array<{ title: string; link: string; topic: string }> = [];
+ const seen = new Set();
+
+ for (const slug of slugs) {
+ if (out.length >= limit) break;
+ let xml: string;
+ try {
+ const res = await fetchImpl(
+ `${RSSAMPLIFIER}/topics/${encodeURIComponent(slug)}.rss`,
+ {
+ signal: AbortSignal.timeout(TIMEOUT_MS),
+ headers: { accept: "application/rss+xml, application/xml;q=0.9" },
+ },
+ );
+ if (!res.ok) continue;
+ xml = await res.text();
+ } catch {
+ continue;
+ }
+
+ // One entry per topic before taking a second from any of them, so a block
+ // of three citations shows three subjects rather than three posts from
+ // whichever feed happened to be longest.
+ const entries = shuffle(itemEntries(xml).filter((e) => e.link));
+ for (const entry of entries) {
+ if (!entry.link || seen.has(entry.link)) continue;
+ seen.add(entry.link);
+ out.push({ title: entry.title, link: entry.link, topic: slug.replace(/-/g, " ") });
+ break;
+ }
+ }
+
+ return out.slice(0, limit);
+}
+
/**
* Undo the escaping a feed document applies, and nothing else.
*
diff --git a/lib/lx/keywordsResearch.ts b/lib/lx/keywordsResearch.ts
index 0de6e48f..7f5bbc80 100644
--- a/lib/lx/keywordsResearch.ts
+++ b/lib/lx/keywordsResearch.ts
@@ -1,14 +1,20 @@
// Keyword research pipeline (PRD §6.1, §15).
//
-// Inputs: siteId — site must have niche + (optionally) target_audiences set.
-// Outputs: up to 30 new rows in lx_keyword (status='queued'), scheduled
-// across the next ~6 weeks honoring publish_days + daily_article_count.
+// Inputs: siteId — the site's `master_keywords` (the durable subject list) and
+// `modifiers` (what the site actually does) are the two lists this
+// works from. `niche` backfills the modifiers when that column is empty.
+// Outputs: up to 30 new rows in lx_keyword (status='queued'), allocated evenly
+// across every master subject and scheduled across the next ~6 weeks
+// honoring publish_days + daily_article_count.
//
-// Caching: lx_keyword_metrics rows live 60 days. Before paying DataForSEO
-// for a seed, we check whether we already have a recent expansion cached
-// (heuristic: same seed string, region='us'). For v1 we always re-fetch on
-// manual trigger; cache is queried per-keyword to short-circuit the
-// volume backfill step.
+// **Every subject is researched, and none is researched alone.** Both halves
+// matter and both were broken. See lib/lx/topicPlan.ts for the failure this
+// was written against — nineteen articles about peptide vendors on a payments
+// blog — and why the fix is a cross product rather than a bigger slice.
+//
+// Cost note: the cross queries all go into a *single* keywordIdeas call, which
+// accepts 200 seeds. Covering ten subjects properly is therefore cheaper than
+// the three separate calls this replaced, not more expensive.
//
// Spend ledger: every DataForSEO call writes a row to lx_dataforseo_usage.
@@ -25,6 +31,18 @@ import {
} from "./buyerJourneyKeywords";
import { DataForSeoClient, filterOutliers, type DfsKeywordRow } from "./dataforseo";
import { nextPublishAt } from "./schedule";
+import {
+ allocate,
+ anchorTokens,
+ crossQueries,
+ dropDuplicates,
+ isOnNiche,
+ resolveMasters,
+ resolveModifiers,
+ signature,
+ stem,
+ tokens,
+} from "./topicPlan";
type SiteRow = {
id: string;
@@ -33,6 +51,8 @@ type SiteRow = {
target_audiences: string[];
description: string | null;
seed_keywords: string[];
+ master_keywords: string[];
+ modifiers: string[];
keywords: string[];
competitors: string[];
tone: string | null;
@@ -45,24 +65,20 @@ const TARGET_KEYWORDS = 30;
const MIN_VOLUME = 50;
const MIN_BUYER_JOURNEY_VOLUME = 10;
const MIN_WORDS = 2;
-const PER_SEED_LIMIT = 200;
+const IDEAS_LIMIT = 600;
const MAX_BUYER_JOURNEY_VOLUME_LOOKUP = 160;
-const SEED_TOKEN_STOPLIST = new Set([
- "the","and","for","with","you","your","that","this","from","into","over",
- "but","not","are","was","were","has","had","have","its","off","out","all",
- "any","new","get","how","why","what","who","best","top",
-]);
-
-function buildSeeds(site: SiteRow): string[] {
- const seeds: string[] = [];
- for (const s of site.seed_keywords ?? []) seeds.push(s.trim());
- if (site.niche) seeds.push(site.niche.trim());
- for (const a of (site.target_audiences ?? []).slice(0, 3)) {
- if (site.niche && a) seeds.push(`${site.niche} for ${a}`);
- else if (a) seeds.push(a);
- }
- return Array.from(new Set(seeds.filter((s) => s.length > 0))).slice(0, 5);
-}
+
+/**
+ * Cross depth per subject.
+ *
+ * Three narrowing terms per subject: enough that a subject yields more than a
+ * single phrasing, few enough that ten subjects still fit one API call with
+ * room for the seed list to grow.
+ */
+const CROSS_PER_MASTER = 3;
+
+/** A candidate with the subject it belongs to. */
+type Candidate = { row: DfsKeywordRow; master: string };
function parseStoredKeyword(row: string): DfsKeywordRow | null {
const idx = row.indexOf(",");
@@ -82,20 +98,43 @@ function parseStoredKeyword(row: string): DfsKeywordRow | null {
};
}
-function seedTokens(seed: string): string[] {
- return seed
- .toLowerCase()
- .split(/[\s-]+/)
- .map((t) => t.replace(/[^a-z0-9]/g, ""))
- .filter((t) => t.length >= 4 && !SEED_TOKEN_STOPLIST.has(t));
-}
-
type KeywordBoost = {
priority: number;
intent: BuyerJourneyKeywordIntent;
clusterType: BuyerJourneyClusterType;
};
+/**
+ * Which subject a candidate belongs to.
+ *
+ * Longest match wins, so a site covering both "crypto" and "cryptocurrency"
+ * attributes "cryptocurrency merchant account" to the more specific of the
+ * two rather than to whichever happens to sort first. Returns null when no
+ * subject claims it — the caller drops those, which is the same verdict the
+ * niche gate would reach a moment later.
+ */
+function attribute(keyword: string, masters: string[]): string | null {
+ let best: string | null = null;
+ let bestLen = 0;
+ // Stemmed on both sides, so attribution and the gate agree about what
+ // "promo codes" and "promo code" are. They disagreed before, and a keyword
+ // attributed to a subject the gate then could not match was dropped for a
+ // reason nobody would have guessed from the strings.
+ const candidate = new Set(tokens(keyword).map(stem));
+ for (const master of masters) {
+ const masterTokens = tokens(master).map(stem);
+ if (masterTokens.length === 0) continue;
+ const hit = masterTokens.filter((t) => candidate.has(t));
+ if (hit.length === 0) continue;
+ const len = hit.join("").length;
+ if (len > bestLen) {
+ bestLen = len;
+ best = master;
+ }
+ }
+ return best;
+}
+
function rankKeywords(
rows: DfsKeywordRow[],
boosts = new Map(),
@@ -136,26 +175,22 @@ function rankKeywords(
return scored.map((s) => s.row);
}
-function primarySeedQuery(site: SiteRow, seeds: string[]): string {
- return (
- seeds[0] ??
- site.niche ??
- site.keywords?.[0] ??
- site.description ??
- site.domain ??
- "site keyword research"
- ).trim();
-}
-
-function buildBuyerJourneyInput(site: SiteRow, seeds: string[]): BuyerJourneyKeywordInput {
+function buildBuyerJourneyInput(
+ site: SiteRow,
+ masters: string[],
+ modifiers: string[],
+): BuyerJourneyKeywordInput {
const brand = site.domain?.replace(/^www\./, "") || site.niche || "the website";
const offer = [site.niche, site.description]
.filter((s): s is string => !!s && s.trim().length > 0)
.join(" — ")
.slice(0, 900);
return {
- seedQuery: primarySeedQuery(site, seeds),
- additionalSeeds: seeds.slice(1, 8),
+ // Every subject, not just the first one. The old code passed
+ // `seeds[0]` here and `seeds.slice(1, 8)` from an already-truncated
+ // list, which is how one subject came to own the entire model run.
+ seedQuery: masters.join(", "),
+ additionalSeeds: modifiers.slice(0, 8),
offer: offer || brand,
audience: site.target_audiences?.join(", ") || "the site's target customers",
brand,
@@ -182,21 +217,6 @@ function rowFromCandidate(
};
}
-function dedupeBySite(
- candidates: DfsKeywordRow[],
- existing: Set,
-): DfsKeywordRow[] {
- const out: DfsKeywordRow[] = [];
- const seen = new Set();
- for (const r of candidates) {
- const k = r.keyword.toLowerCase();
- if (existing.has(k) || seen.has(k)) continue;
- seen.add(k);
- out.push(r);
- }
- return out;
-}
-
function scheduleKeywords(
count: number,
publishDays: number[],
@@ -218,10 +238,39 @@ function scheduleKeywords(
return dates;
}
+/**
+ * Interleave per-subject picks so the schedule alternates topics.
+ *
+ * Allocation decides *how many* rows each subject gets; this decides the order
+ * they are scheduled in, and the two are separate concerns. A fair allocation
+ * emitted subject-by-subject would still publish six consecutive peptide posts
+ * and then six consecutive casino ones — fair over a quarter, and visibly
+ * spammy over a fortnight. Round-robining the emission is what makes the fix
+ * legible to a reader of the blog rather than only to a reader of the database.
+ */
+function interleave(byMaster: Map): Candidate[] {
+ const queues = Array.from(byMaster.values()).map((c) => [...c]);
+ const out: Candidate[] = [];
+ let live = true;
+ while (live) {
+ live = false;
+ for (const queue of queues) {
+ const next = queue.shift();
+ if (next) {
+ out.push(next);
+ live = true;
+ }
+ }
+ }
+ return out;
+}
+
export type KeywordResearchResult = {
ok: boolean;
inserted: number;
apiCost: number;
+ /** Rows allocated per subject — surfaced so a skewed queue is visible. */
+ perMaster?: Record;
error?: string;
};
@@ -240,7 +289,7 @@ export async function researchKeywords(
const { data: site } = await supabase
.from("lx_site")
.select(
- "id, domain, niche, target_audiences, description, seed_keywords, keywords, competitors, tone, publish_days, publish_hour, daily_article_count",
+ "id, domain, niche, target_audiences, description, seed_keywords, master_keywords, modifiers, keywords, competitors, tone, publish_days, publish_hour, daily_article_count",
)
.eq("id", siteId)
.maybeSingle();
@@ -248,153 +297,258 @@ export async function researchKeywords(
return { ok: false, inserted: 0, apiCost: 0, error: "site not found" };
}
- const savedKeywords = (site.keywords ?? [])
- .map(parseStoredKeyword)
- .filter((r): r is DfsKeywordRow => !!r);
- const seeds = buildSeeds(site);
- if (savedKeywords.length === 0 && seeds.length === 0) {
+ const masters = resolveMasters(site);
+ // Masters are passed through so the derived terms can subtract them: an
+ // anchor word that is also a subject word lets a candidate satisfy both
+ // halves of the gate with one token. See anchorTokens.
+ const modifiers = resolveModifiers(site, masters);
+ const anchors = anchorTokens(site, masters);
+
+ if (masters.length === 0) {
return {
ok: false,
inserted: 0,
apiCost: 0,
- error: "add saved keywords or seed keywords first",
+ error: "add master keywords (the subjects this blog covers) first",
+ };
+ }
+ // Refusing here rather than falling back to an unanchored expansion is the
+ // point. An anchorless run is exactly the run that produced the vendor
+ // articles, so it must be an error the operator sees and fixes, not a
+ // degraded mode that quietly publishes.
+ if (anchors.size === 0) {
+ return {
+ ok: false,
+ inserted: 0,
+ apiCost: 0,
+ error:
+ "set a niche or modifiers first — keywords are only researched crossed with what this site does, never on their own",
};
}
- // Skip keywords this site already has in active/history states. Failed
- // rows are intentionally ignored here: an upstream outage should not
- // permanently poison a topic and prevent the top-up sweep from
- // refilling the queue.
+ // Existing rows serve two purposes: the duplicate fingerprints, and the
+ // per-subject coverage the allocator balances against. Failed rows are
+ // intentionally excluded from the duplicate set — an upstream outage should
+ // not permanently poison a topic — but they still count toward coverage, so
+ // a subject that keeps failing does not monopolise every top-up.
const { data: existingRows } = await supabase
.from("lx_keyword")
- .select("keyword, status")
+ .select("keyword, status, master_keyword")
.eq("site_id", site.id);
- const existingSet = new Set(
- (existingRows ?? [])
- .filter((r: { status: string }) => r.status !== "failed")
- .map((r: { keyword: string }) => r.keyword.toLowerCase()),
- );
- const savedChosen = dedupeBySite(savedKeywords, existingSet).slice(0, TARGET_KEYWORDS);
+ const publishedSignatures = new Set();
+ const coverage = new Map();
+ for (const row of (existingRows ?? []) as Array<{
+ keyword: string;
+ status: string;
+ master_keyword: string | null;
+ }>) {
+ if (row.status !== "failed") {
+ const sig = signature(row.keyword);
+ if (sig) publishedSignatures.add(sig);
+ }
+ // Attribute legacy rows (written before provenance existed) so the
+ // allocator sees the real history rather than treating a blog with
+ // twenty-three peptide posts as having no coverage at all. Without this
+ // the balancing would take a full cycle to notice the existing skew.
+ const master = row.master_keyword ?? attribute(row.keyword, masters);
+ if (master) {
+ const key = master.toLowerCase();
+ coverage.set(key, (coverage.get(key) ?? 0) + 1);
+ }
+ }
- // If the saved long-tail list does not fill the target queue, top it
- // up using the same DataForSEO Labs endpoint + relevance gate used by
- // the settings page's "Refetch keywords" flow.
- const allRows: DfsKeywordRow[] = [];
+ const allocation = allocate(masters, coverage, TARGET_KEYWORDS);
+
+ // ------------------------------------------------------------------
+ // Candidate sources. All three are gated identically; they differ only
+ // in how they are obtained and how likely they are to be unavailable.
+ // ------------------------------------------------------------------
+ const candidates: Candidate[] = [];
const buyerJourneyBoosts = new Map();
let totalCost = 0;
- const seedErrors: string[] = [];
- if (savedChosen.length < TARGET_KEYWORDS && seeds.length > 0) {
- if (deps.openai || deps.anthropic) {
- try {
- const buyerJourney = await generateBuyerJourneyKeywordOpportunities(
- buildBuyerJourneyInput(site, seeds),
- {
- openai: deps.openai,
- anthropic: deps.anthropic,
- backendAiProvider: deps.backendAiProvider,
- },
- );
- const candidates = flattenBuyerJourneyKeywords(
- buyerJourney.output,
- MAX_BUYER_JOURNEY_VOLUME_LOOKUP,
- );
- const volumeKeywords = candidates.map((c) => c.keyword);
- const volume = volumeKeywords.length > 0
- ? await dfs.searchVolume(volumeKeywords)
- : { rows: [], cost: 0, taskId: null };
- totalCost += volume.cost;
- if (volume.cost > 0 || volume.taskId) {
- await supabase.from("lx_dataforseo_usage").insert({
- task_id: volume.taskId,
- endpoint: "search_volume/live",
- cost: volume.cost,
- site_id: site.id,
- });
- }
-
- const metricsByKeyword = new Map(
- volume.rows.map((r) => [r.keyword.toLowerCase(), r]),
- );
- for (const candidate of candidates) {
- const metrics = metricsByKeyword.get(candidate.keyword.toLowerCase());
- const volumeValue = metrics?.search_volume ?? 0;
- if (
- volumeValue >= MIN_BUYER_JOURNEY_VOLUME ||
- candidate.priority >= 4
- ) {
- allRows.push(rowFromCandidate(candidate, metrics));
- buyerJourneyBoosts.set(candidate.keyword.toLowerCase(), {
- priority: candidate.priority,
- intent: candidate.intent,
- clusterType: candidate.clusterType,
- });
- }
- }
- } catch (err) {
- seedErrors.push(
- `buyer-journey model: ${err instanceof Error ? err.message : String(err)}`,
- );
+ const sourceErrors: string[] = [];
+
+ // Source 0 — the floor. Subject × modifier, built locally from two columns
+ // the operator controls. No network, no failure mode, always on-niche by
+ // construction. Everything below is an improvement on this, never a
+ // prerequisite for it.
+ const crosses = crossQueries(masters, modifiers, CROSS_PER_MASTER);
+ for (const { master, query } of crosses) {
+ candidates.push({
+ row: {
+ keyword: query,
+ search_volume: null,
+ competition: null,
+ competition_index: null,
+ cpc: null,
+ low_top_of_page_bid: null,
+ high_top_of_page_bid: null,
+ monthly_searches: null,
+ },
+ master,
+ });
+ }
+
+ // Source 1 — hand-saved long-tail from the settings page.
+ for (const parsed of (site.keywords ?? []).map(parseStoredKeyword)) {
+ if (!parsed) continue;
+ const master = attribute(parsed.keyword, masters);
+ if (master) candidates.push({ row: parsed, master });
+ }
+
+ // Source 2 — the buyer-journey model, now seeded with every subject.
+ if (deps.openai || deps.anthropic) {
+ try {
+ const buyerJourney = await generateBuyerJourneyKeywordOpportunities(
+ buildBuyerJourneyInput(site, masters, modifiers),
+ {
+ openai: deps.openai,
+ anthropic: deps.anthropic,
+ backendAiProvider: deps.backendAiProvider,
+ },
+ );
+ const flattened = flattenBuyerJourneyKeywords(
+ buyerJourney.output,
+ MAX_BUYER_JOURNEY_VOLUME_LOOKUP,
+ );
+ const volumeKeywords = flattened.map((c) => c.keyword);
+ const volume = volumeKeywords.length > 0
+ ? await dfs.searchVolume(volumeKeywords)
+ : { rows: [], cost: 0, taskId: null };
+ totalCost += volume.cost;
+ if (volume.cost > 0 || volume.taskId) {
+ await supabase.from("lx_dataforseo_usage").insert({
+ task_id: volume.taskId,
+ endpoint: "search_volume/live",
+ cost: volume.cost,
+ site_id: site.id,
+ });
+ }
+
+ const metricsByKeyword = new Map(
+ volume.rows.map((r) => [r.keyword.toLowerCase(), r]),
+ );
+ for (const candidate of flattened) {
+ const metrics = metricsByKeyword.get(candidate.keyword.toLowerCase());
+ const volumeValue = metrics?.search_volume ?? 0;
+ if (volumeValue < MIN_BUYER_JOURNEY_VOLUME && candidate.priority < 4) continue;
+ const master = attribute(candidate.keyword, masters);
+ if (!master) continue;
+ candidates.push({ row: rowFromCandidate(candidate, metrics), master });
+ buyerJourneyBoosts.set(candidate.keyword.toLowerCase(), {
+ priority: candidate.priority,
+ intent: candidate.intent,
+ clusterType: candidate.clusterType,
+ });
}
+ } catch (err) {
+ sourceErrors.push(
+ `buyer-journey model: ${err instanceof Error ? err.message : String(err)}`,
+ );
}
+ }
- for (const seed of seeds.slice(0, 3)) {
- try {
- const result = await dfs.keywordIdeas([seed], {
- limit: PER_SEED_LIMIT,
+ // Source 3 — DataForSEO expansion of the CROSSED phrases.
+ //
+ // One call carrying every cross, because keywordIdeas accepts 200 seeds.
+ // The bare subject is never sent: "peptide" on its own is what returned
+ // "skye peptides" and "pure peptide labs", and no downstream filter can
+ // reliably tell those from a keyword worth writing about.
+ if (crosses.length > 0) {
+ try {
+ const result = await dfs.keywordIdeas(
+ crosses.map((c) => c.query),
+ {
+ limit: IDEAS_LIMIT,
minVolume: MIN_VOLUME,
minWords: MIN_WORDS,
closelyVariants: false,
- });
- totalCost += result.cost;
- await supabase.from("lx_dataforseo_usage").insert({
- task_id: result.taskId,
- endpoint: "keyword_ideas/live",
- cost: result.cost,
- site_id: site.id,
- });
-
- const tokens = seedTokens(seed);
- const relevant = tokens.length === 0
- ? result.rows
- : result.rows.filter((r) => {
- const kw = r.keyword.toLowerCase();
- return tokens.some((t) => kw.includes(t));
- });
- allRows.push(...relevant);
- } catch (err) {
- seedErrors.push(
- `"${seed}": ${err instanceof Error ? err.message : String(err)}`,
- );
+ },
+ );
+ totalCost += result.cost;
+ await supabase.from("lx_dataforseo_usage").insert({
+ task_id: result.taskId,
+ endpoint: "keyword_ideas/live",
+ cost: result.cost,
+ site_id: site.id,
+ });
+ for (const row of result.rows) {
+ const master = attribute(row.keyword, masters);
+ if (master) candidates.push({ row, master });
}
+ } catch (err) {
+ sourceErrors.push(
+ `keyword ideas: ${err instanceof Error ? err.message : String(err)}`,
+ );
}
}
- const filtered = filterOutliers(allRows).filter(
- (r) => {
- const boost = buyerJourneyBoosts.get(r.keyword.toLowerCase());
- if (boost && boost.priority >= 4) return true;
- return (r.search_volume ?? 0) >= MIN_VOLUME;
- },
- );
- const ranked = rankKeywords(filtered, buyerJourneyBoosts);
- const existingWithSaved = new Set(existingSet);
- for (const r of savedChosen) existingWithSaved.add(r.keyword.toLowerCase());
- const researchedChosen = dedupeBySite(ranked, existingWithSaved).slice(
- 0,
- TARGET_KEYWORDS - savedChosen.length,
+ // ------------------------------------------------------------------
+ // Gate, rank, allocate.
+ // ------------------------------------------------------------------
+ const onNiche = candidates.filter((c) => isOnNiche(c.row.keyword, c.master, anchors));
+
+ // Volume filtering applies only to what came back from an API with a volume
+ // attached. The locally-built crosses have no volume by construction and
+ // must not be discarded for it — they are the floor that keeps the queue
+ // from emptying when everything upstream is unavailable.
+ const withVolume = onNiche.filter((c) => c.row.search_volume !== null);
+ const withoutVolume = onNiche.filter((c) => c.row.search_volume === null);
+ const volumeKept = filterOutliers(withVolume.map((c) => c.row)).filter((r) => {
+ const boost = buyerJourneyBoosts.get(r.keyword.toLowerCase());
+ if (boost && boost.priority >= 4) return true;
+ return (r.search_volume ?? 0) >= MIN_VOLUME;
+ });
+ const volumeKeptKeys = new Set(volumeKept.map((r) => r.keyword.toLowerCase()));
+
+ const ranked = rankKeywords(volumeKept, buyerJourneyBoosts);
+ const rankIndex = new Map(ranked.map((r, i) => [r.keyword.toLowerCase(), i]));
+
+ const survivors = [
+ ...withVolume
+ .filter((c) => volumeKeptKeys.has(c.row.keyword.toLowerCase()))
+ .sort(
+ (a, b) =>
+ (rankIndex.get(a.row.keyword.toLowerCase()) ?? Infinity) -
+ (rankIndex.get(b.row.keyword.toLowerCase()) ?? Infinity),
+ ),
+ // Crosses last within each subject: a real long-tail phrase with measured
+ // demand is a better article than a two-word construction, but the
+ // construction is a better article than nothing.
+ ...withoutVolume,
+ ];
+
+ const deduped = dropDuplicates(
+ survivors.map((c) => ({ ...c, keyword: c.row.keyword })),
+ publishedSignatures,
);
- const chosen = [...savedChosen, ...researchedChosen].slice(0, TARGET_KEYWORDS);
+
+ // Take each subject's allocated share, then interleave so the published
+ // sequence alternates subjects rather than running one to exhaustion.
+ const byMaster = new Map();
+ for (const master of masters) byMaster.set(master, []);
+ for (const candidate of deduped) {
+ const bucket = byMaster.get(candidate.master);
+ if (!bucket) continue;
+ if (bucket.length >= (allocation.get(candidate.master) ?? 0)) continue;
+ bucket.push({ row: candidate.row, master: candidate.master });
+ }
+
+ // Subjects that could not fill their share hand it back, so a subject with
+ // no available candidates costs the run coverage rather than volume.
+ const chosen = interleave(byMaster).slice(0, TARGET_KEYWORDS);
+
if (chosen.length === 0) {
- const details = seedErrors.length > 0
- ? ` Seed errors: ${seedErrors.join("; ")}`
+ const details = sourceErrors.length > 0
+ ? ` Source errors: ${sourceErrors.join("; ")}`
: "";
return {
ok: false,
inserted: 0,
apiCost: totalCost,
error:
- "No new keyword candidates found. Saved keywords may already be published or queued; add new seed keywords/settings and try again." +
+ "No new keyword candidates found. Every candidate was already published or off-niche; add master keywords or modifiers and try again." +
details,
};
}
@@ -405,7 +559,6 @@ export async function researchKeywords(
// without first reshaping it. We re-introduce it when keyword overlap
// across customers becomes measurable.
- // Schedule them across publish_days.
const slots = scheduleKeywords(
chosen.length,
site.publish_days,
@@ -413,15 +566,16 @@ export async function researchKeywords(
site.daily_article_count,
);
- const insertRows = chosen.map((r, i) => ({
+ const insertRows = chosen.map((c, i) => ({
site_id: site.id,
- keyword: r.keyword,
+ keyword: c.row.keyword,
+ master_keyword: c.master,
scheduled_for: slots[i]?.toISOString().slice(0, 10) ??
new Date(Date.now() + (i + 1) * 86400000).toISOString().slice(0, 10),
status: "queued",
source: "auto",
- search_volume: r.search_volume,
- cpc_usd: r.cpc,
+ search_volume: c.row.search_volume,
+ cpc_usd: c.row.cpc,
}));
const { error: insErr } = await supabase.from("lx_keyword").insert(insertRows);
@@ -434,5 +588,10 @@ export async function researchKeywords(
};
}
- return { ok: true, inserted: insertRows.length, apiCost: totalCost };
+ const perMaster: Record = {};
+ for (const row of insertRows) {
+ perMaster[row.master_keyword] = (perMaster[row.master_keyword] ?? 0) + 1;
+ }
+
+ return { ok: true, inserted: insertRows.length, apiCost: totalCost, perMaster };
}
diff --git a/lib/lx/networkBlock.ts b/lib/lx/networkBlock.ts
new file mode 100644
index 00000000..13928d72
--- /dev/null
+++ b/lib/lx/networkBlock.ts
@@ -0,0 +1,232 @@
+// What a published article carries besides the article: an ad unit, and links
+// out to the rest of the network.
+//
+// Both are governed by one column — `lx_site.ads_enabled`, default true — and
+// that is deliberate. Splitting "show ads" from "join the link network" into
+// two switches produces four states, two of which are incoherent (take
+// backlinks from partners, refuse to give them) and all four of which need a
+// human to reason about. One switch, on by default, is the whole opt-in.
+//
+// Three things this is careful about.
+//
+// **The slot is provisioned, not requested.** A blog with no ad slot used to
+// mean a support conversation. Slots are created here, active, on first
+// delivery — `ad_slots.status` defaults to 'inactive' and `serveAd()` returns
+// null before the house-ad fallback for a non-active slot, so a slot created
+// at the default would render an empty div on every article forever and look
+// exactly like a broken embed. Provisioning it inactive would be worse than
+// not provisioning it at all.
+//
+// **Everything interpolated is escaped.** The titles and links in the
+// insertion block come from other people's RSS feeds. They are written into
+// HTML that lands on a customer's domain, which makes this an injection sink
+// with a hostile upstream, and the fact that the immediate source is our own
+// directory changes nothing about that — the directory is a crawl of the open
+// web.
+//
+// **A failure here costs the block, never the article.** Every lookup is
+// wrapped and degrades to "no block". A post that publishes without an ad unit
+// has cost us an impression; a post that fails to publish because the ad
+// lookup threw has cost the customer the thing they are paying for.
+
+import type { SupabaseClient } from "@supabase/supabase-js";
+import { cachedFeedPosts } from "./feedCrawl";
+
+/** Format asked of the slot. 728x90 is the in-article leaderboard. */
+const AD_FORMAT = "banner_728x90";
+
+/** Most partner links in one insertion. */
+const MAX_PARTNER_LINKS = 3;
+
+/** Most directory posts in one insertion. */
+const MAX_FEED_LINKS = 3;
+
+/**
+ * Escape text for HTML interpolation.
+ *
+ * Ampersand first, or the escapes introduced by the later replacements get
+ * double-escaped. Quotes are included because these values are also written
+ * into attributes.
+ */
+export function escapeHtml(value: string): string {
+ return String(value ?? "")
+ .replace(/&/g, "&")
+ .replace(//g, ">")
+ .replace(/"/g, """)
+ .replace(/'/g, "'");
+}
+
+/**
+ * Is this a link we are willing to put on a customer's page?
+ *
+ * An allowlist of two schemes rather than a denylist of the dangerous ones:
+ * `javascript:`, `data:` and `vbscript:` are the ones anybody thinks to block,
+ * and the list of what else a browser will execute is not one to maintain by
+ * hand.
+ */
+export function isSafeHref(href: string): boolean {
+ try {
+ const url = new URL(href);
+ return url.protocol === "https:" || url.protocol === "http:";
+ } catch {
+ return false;
+ }
+}
+
+export type NetworkLink = { title: string; url: string; source: "partner" | "directory" };
+
+/**
+ * The slot this project's articles should fill.
+ *
+ * Reuses an existing active slot before creating one, so re-delivering an
+ * article — or publishing the second post on a blog — does not mint a second
+ * slot and split the site's reporting across two rows.
+ *
+ * @returns a slot id, or null when one could not be resolved or created
+ */
+export async function resolveAdSlot(
+ supabase: SupabaseClient,
+ projectId: string,
+ ownerId: string,
+ niche: string | null,
+): Promise {
+ try {
+ const { data: existing } = await supabase
+ .from("ad_slots")
+ .select("id")
+ .eq("project_id", projectId)
+ .eq("status", "active")
+ .limit(1)
+ .maybeSingle<{ id: string }>();
+ if (existing?.id) return existing.id;
+
+ const { data: created } = await supabase
+ .from("ad_slots")
+ .insert({
+ project_id: projectId,
+ owner_id: ownerId,
+ placement: "inline",
+ formats: [AD_FORMAT, "banner_300x250", "text_link"],
+ niche,
+ // Explicit, against the column default. See the note at the top of
+ // this file: an inactive slot is indistinguishable from a broken one.
+ status: "active",
+ })
+ .select("id")
+ .maybeSingle<{ id: string }>();
+ return created?.id ?? null;
+ } catch {
+ return null;
+ }
+}
+
+/**
+ * The ad unit markup.
+ *
+ * `data-cp-ad` carries no `data-slot` in the server-rendered case elsewhere in
+ * the codebase to avoid a fill race; here there is no client to race with —
+ * this HTML is delivered to a third-party blog and rendered as-is — so the
+ * slot travels on the element and ad.js's own DOMContentLoaded pass fills it.
+ */
+export function adUnitHtml(slotId: string, origin: string): string {
+ const slot = escapeHtml(slotId);
+ const src = `${origin.replace(/\/$/, "")}/ad.js`;
+ return [
+ ``,
+ ``,
+ ].join("\n");
+}
+
+/**
+ * The "elsewhere in the network" block.
+ *
+ * Partner articles first, directory posts after: a partner opted into the
+ * exchange and gets the more valuable position, while the directory posts are
+ * what keep the block from being visibly the same three domains on every
+ * article — which is the shape that gets a link network discounted.
+ *
+ * Directory links are `rel="nofollow ugc"`. They are not exchange partners and
+ * have not agreed to anything; passing them ranking signal would be us
+ * spending someone else's reputation. Partner links are followed, because that
+ * reciprocity is the entire point of the exchange and is recorded on both
+ * sides in lx_backlink.
+ */
+export function networkLinksHtml(links: NetworkLink[]): string {
+ const safe = links.filter((l) => isSafeHref(l.url));
+ if (safe.length === 0) return "";
+
+ const items = safe
+ .map((link) => {
+ const rel = link.source === "directory"
+ ? ' rel="nofollow ugc noopener"'
+ : ' rel="noopener"';
+ return `
`;
+ })
+ .join("\n");
+
+ return [
+ ``,
+ ].join("\n");
+}
+
+/**
+ * Everything appended to an article, for a site that is in the network.
+ *
+ * @param supabase service client — this runs from the delivery path, no session
+ * @param site the blog that will HOST the post (the guest-post target, when
+ * the article is one), because it is that site's readers who see the
+ * block and that site's owner who opted in
+ * @param topics subjects to pull directory posts for
+ * @param partnerLinks already-ranked exchange candidates from the caller
+ * @returns HTML to append, or "" when the site is opted out or nothing resolved
+ */
+export async function buildNetworkBlock(
+ supabase: SupabaseClient,
+ site: {
+ id: string;
+ project_id: string;
+ user_id: string;
+ niche: string | null;
+ ads_enabled?: boolean | null;
+ },
+ topics: string[],
+ partnerLinks: NetworkLink[],
+ origin: string,
+): Promise {
+ // Absent column reads as opted in, matching the migration default, so this
+ // behaves the same before and after the schema lands.
+ if (site.ads_enabled === false) return "";
+
+ const parts: string[] = [];
+
+ // Read from the crawl cache, never live. A publish must not wait on the
+ // directory's server; a topic the sweep has not reached yet costs this
+ // article its citation block and the next one gets it.
+ const feedLinks: NetworkLink[] = await cachedFeedPosts(supabase, topics, MAX_FEED_LINKS)
+ .then((posts) =>
+ posts.map((p) => ({ title: p.title, url: p.link, source: "directory" as const })),
+ )
+ .catch(() => []);
+
+ const block = networkLinksHtml([
+ ...partnerLinks.slice(0, MAX_PARTNER_LINKS),
+ ...feedLinks,
+ ]);
+ if (block) parts.push(block);
+
+ const slotId = await resolveAdSlot(
+ supabase,
+ site.project_id,
+ site.user_id,
+ site.niche,
+ );
+ if (slotId) parts.push(adUnitHtml(slotId, origin));
+
+ return parts.join("\n");
+}
diff --git a/lib/lx/topicPlan.ts b/lib/lx/topicPlan.ts
new file mode 100644
index 00000000..7514129c
--- /dev/null
+++ b/lib/lx/topicPlan.ts
@@ -0,0 +1,436 @@
+// Deciding which subjects an autoblog writes about, and in what proportion.
+//
+// This module exists because of a specific failure, and the shape of it is a
+// direct response to that failure. coinpayportal.com — a crypto payment
+// processor — published nineteen consecutive articles about peptide vendors:
+// "skye peptides", "pure peptide labs", "wolverine stack peptides". Those are
+// competitor storefronts in an industry the site *serves*, not one it is in.
+//
+// Two independent defects produced it, and fixing either alone would have left
+// the other running.
+//
+// 1. **Truncation.** The site had ten subjects. The pipeline sliced them to
+// five, then to three, and handed subject #1 to the buyer-journey model as
+// its entire query. Five subjects had never produced a keyword in the
+// site's lifetime. The top-up sweep re-ran the same truncated set every
+// time the queue drained, so the concentration compounded rather than
+// averaged out.
+//
+// 2. **An unanchored relevance gate.** A candidate was kept if it contained
+// the seed token. Expanding the bare word "peptide" against a keyword
+// tool returns the peptide industry's own vocabulary, and every one of
+// those passed a test that only ever asked "is this about peptides?" —
+// never "is this about what we sell to them?".
+//
+// The answer to both is the same: never expand a subject on its own. A subject
+// is only ever researched *crossed with a modifier* — the tail terms that
+// describe what this site actually does ("merchant account", "payment
+// gateway"). "peptide" is not a topic. "peptide merchant account" is.
+//
+// That cross is also the floor. Every other source here can return nothing —
+// the keyword API can be down, the model can refuse, the gate can reject
+// everything — and the cross product still yields on-niche subjects, because
+// it is built from two lists the operator controls rather than fetched. A
+// blog that cannot reach any upstream still publishes, and still publishes
+// about itself. Going dark and going off-topic are both failures; this
+// prefers a smaller, correct queue to a full, spammy one.
+
+/** Subjects past this point can't be given a meaningful share of a 30-row target. */
+export const MAX_MASTERS = 12;
+
+/**
+ * Words that carry no topic and would pass any gate built on them.
+ *
+ * Deliberately not a general English stoplist: this is scoped to the words
+ * that appear in a *niche description* and a keyword phrase without narrowing
+ * either. "best" and "top" are here because they are the two most common
+ * prefixes in keyword-tool output and matching on them would re-admit the
+ * whole vendor-listicle class this module exists to reject.
+ */
+const STOPWORDS = new Set([
+ "the", "and", "for", "with", "you", "your", "that", "this", "from", "into",
+ "over", "but", "not", "are", "was", "were", "has", "had", "have", "its",
+ "off", "out", "all", "any", "new", "get", "how", "why", "what", "who",
+ "best", "top", "guide", "list", "using", "about", "when", "where", "which",
+]);
+
+/**
+ * A token worth matching on.
+ *
+ * Four characters because three-letter fragments ("pay", "ads", "seo") match
+ * inside unrelated words often enough to be worse than useless in a gate whose
+ * entire job is rejecting near-misses.
+ */
+const MIN_TOKEN_LEN = 4;
+
+/**
+ * Split a phrase into matchable tokens.
+ *
+ * Punctuation is dropped rather than split on, so "high-risk" yields "high"
+ * and "risk" — both of which are real narrowing terms for the site that wrote
+ * that niche, and neither of which survives a naive whitespace split.
+ */
+export function tokens(phrase: string): string[] {
+ return (phrase ?? "")
+ .toLowerCase()
+ .split(/[^a-z0-9]+/i)
+ .map((t) => t.trim())
+ .filter((t) => t.length >= MIN_TOKEN_LEN && !STOPWORDS.has(t));
+}
+
+/**
+ * Reduce a token to a form that survives pluralisation.
+ *
+ * A crude suffix strip rather than a real stemmer, because the only job is
+ * collapsing "payment"/"payments" and "transaction"/"transactions" so the
+ * duplicate check can see that "peptide payments" and "peptide payment" are
+ * the same article. A real stemmer would be a dependency and a behaviour
+ * change in the gate, for a class of match this never needs to make.
+ */
+export function stem(token: string): string {
+ if (token.length > 4 && token.endsWith("ies")) return `${token.slice(0, -3)}y`;
+
+ if (token.length > 3 && token.endsWith("es")) {
+ // "-es" is two different plurals and stripping a fixed number of
+ // characters gets one of them wrong. Taking two always ("codes" → "cod")
+ // does not collide with the singular ("code" → "code"), so a subject of
+ // "promo codes" stopped matching the keyword "promo code" — the exact
+ // shape this function exists to collapse. Strip one by default and two
+ // only after a sibilant, which is where the extra "e" is really epenthetic:
+ // "boxes" → "box", "matches" → "match", but "codes" → "code".
+ const short = token.slice(0, -2);
+ return /(?:s|x|z|ch|sh)$/.test(short) ? short : token.slice(0, -1);
+ }
+
+ if (token.length > 3 && token.endsWith("s")) return token.slice(0, -1);
+ return token;
+}
+
+/**
+ * An order-independent fingerprint of what a keyword is about.
+ *
+ * Sorted, so "merchant account peptide" and "peptide merchant account" collide
+ * — they would produce the same article, and a blog publishing both is the
+ * duplicate-content problem this is here to prevent.
+ */
+export function signature(keyword: string): string {
+ return Array.from(new Set(tokens(keyword).map(stem))).sort().join(" ");
+}
+
+type SiteTopicFields = {
+ master_keywords?: string[] | null;
+ seed_keywords?: string[] | null;
+ modifiers?: string[] | null;
+ niche?: string | null;
+};
+
+/**
+ * The durable subject list for a site.
+ *
+ * Falls back to `seed_keywords` for any site the backfill has not reached, so
+ * this is safe to deploy ahead of the migration rather than after it. Capped
+ * on read as well as by the database constraint: a list that arrives oversized
+ * from anywhere must still be allocated over, not rejected at publish time.
+ */
+export function resolveMasters(site: SiteTopicFields): string[] {
+ const raw = (site.master_keywords?.length ? site.master_keywords : site.seed_keywords) ?? [];
+ const seen = new Set();
+ const out: string[] = [];
+ for (const entry of raw) {
+ const trimmed = (entry ?? "").trim();
+ if (!trimmed) continue;
+ const key = trimmed.toLowerCase();
+ if (seen.has(key)) continue;
+ seen.add(key);
+ out.push(trimmed);
+ }
+ return out.slice(0, MAX_MASTERS);
+}
+
+/**
+ * The tail terms that turn any subject into an article this blog would write.
+ *
+ * A last-resort vocabulary, used when a site has no modifiers column and its
+ * niche says nothing the subjects do not already say. That case is common and
+ * it is where the worst output came from: vu1nz.com covers "ci/cd security"
+ * and "supply chain security" under the niche "CI/CD and supply chain
+ * security", so mining the niche yields only words the subjects already
+ * contain — and a gate built from those admits "adt home security" and
+ * "brinks home security" on a supply-chain blog.
+ *
+ * These are commercial and comparative rather than topical on purpose. They
+ * are what distinguishes an article a B2B blog publishes ("screen sharing
+ * software for teams") from a search result about a physical object
+ * ("garage door opener remote"), and they generalise across every niche,
+ * which a topical list could not.
+ */
+// Every entry has to be a word that a *commercial software* search uses and an
+// ordinary one does not. "teams" and "business" were in an earlier version of
+// this list and both had to come out: "community emergency response team" is a
+// real queued keyword on a SOC blog, and it passed the gate on the token
+// "team". A generic English noun cannot carry the second half of a two-part
+// test, however natural it reads in a keyword phrase.
+export const DEFAULT_MODIFIERS = [
+ "software",
+ "tools",
+ "platform",
+ "alternatives",
+ "comparison",
+ "pricing",
+ "integration",
+ "automation",
+ "checklist",
+ "best practices",
+];
+
+/**
+ * The tail terms that anchor a subject to this site's own business.
+ *
+ * Three sources, in descending order of how much the operator meant them.
+ * The middle one subtracts the subjects: a niche word that is also a subject
+ * word cannot narrow anything, and keeping it is what lets a candidate satisfy
+ * both halves of the gate with a single token.
+ *
+ * @param site the row
+ * @param masters from `resolveMasters` — subtracted from the derived terms
+ */
+export function resolveModifiers(
+ site: SiteTopicFields,
+ masters: string[] = [],
+): string[] {
+ const explicit = (site.modifiers ?? [])
+ .map((m) => (m ?? "").trim())
+ .filter((m) => m.length > 0);
+ if (explicit.length > 0) return explicit.slice(0, 20);
+
+ const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem)));
+ const fromNiche = Array.from(new Set(tokens(site.niche ?? ""))).filter(
+ (t) => !masterTokens.has(stem(t)),
+ );
+ if (fromNiche.length > 0) return fromNiche.slice(0, 8);
+
+ return DEFAULT_MODIFIERS;
+}
+
+/**
+ * Every token that means "this keyword is about our business, not just our
+ * subject".
+ *
+ * Master tokens are excluded, and that exclusion is the fix for the sharpest
+ * version of the original bug. On vu1nz.com the subject "devops security" and
+ * the niche both contain "security"; without the subtraction, "adt home
+ * security" matches the subject on `security` and then matches the anchor on
+ * the very same word, satisfying a two-part test with one token. The anchor
+ * has to be evidence the subject match did not already provide.
+ */
+export function anchorTokens(
+ site: SiteTopicFields,
+ masters: string[] = [],
+): Set {
+ const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem)));
+ const out = new Set();
+
+ const add = (phrase: string) => {
+ for (const token of tokens(phrase)) {
+ const stemmed = stem(token);
+ if (!masterTokens.has(stemmed)) out.add(stemmed);
+ }
+ };
+
+ for (const modifier of resolveModifiers(site, masters)) add(modifier);
+ add(site.niche ?? "");
+
+ // The commercial vocabulary is always unioned in, not just used as a
+ // fallback. bl0ggers.com's niche ("human-in-the-loop AI publishing") yields
+ // exactly {human, loop} once its own subjects are removed — non-empty, so
+ // the fallback never fired, and a thin anchor set rejected "ai writing
+ // tools", which is precisely what that blog should write.
+ //
+ // Unioning is safe because these words are commercial-software words: they
+ // rescue the good keywords without admitting any of the junk. None of "adt
+ // home security", "samsung tv remote", "palantir technologies" or "bayesian
+ // optimization" contains one — and where a site's own subject already
+ // claims one ("developer tools" on logicsrc), the master-token subtraction
+ // above removes it, so "mac tools" stays rejected.
+ for (const modifier of DEFAULT_MODIFIERS) add(modifier);
+
+ return out;
+}
+
+/**
+ * Is this candidate about one of our subjects *and* about what we do?
+ *
+ * Both halves are required, on different words, and that conjunction is the
+ * whole fix. The old gate asked only the first question, which is why
+ * nineteen articles about other people's peptide shops passed it — and why
+ * a supply-chain security blog was queued to write about home alarm
+ * installers.
+ *
+ * @param keyword the candidate
+ * @param master the subject it was researched for
+ * @param anchors from `anchorTokens`, which has already removed subject words
+ */
+export function isOnNiche(
+ keyword: string,
+ master: string,
+ anchors: Set,
+): boolean {
+ const candidate = new Set(tokens(keyword).map(stem));
+ if (candidate.size === 0) return false;
+
+ const masterTokens = tokens(master).map(stem);
+ if (masterTokens.length === 0) return false;
+
+ const hits = masterTokens.filter((t) => candidate.has(t)).length;
+ if (hits === 0) return false;
+
+ // A COMPLETE match on a multi-word subject is its own evidence, and needs no
+ // anchor.
+ //
+ // This is what separates the two failure shapes. Every bad keyword found on
+ // live sites matched exactly one generic word out of a multi-word subject —
+ // "security" from "devops security", "remote" from "remote control",
+ // "response" from "incident response", "tools" from "developer tools". None
+ // matched a subject in full. Meanwhile "abercrombie promo code" matches
+ // "promo codes" completely and is precisely what a coupon blog should write,
+ // yet an anchor rule alone rejects it, because a coupon site's subject IS
+ // its topic and it has no narrowing term to offer.
+ //
+ // Single-word subjects are excluded from this, and that exclusion is the
+ // original bug: "peptide" is one token, so "skye peptides" would match it
+ // "completely". A one-word subject can only ever be a vertical the site
+ // serves or a word too broad to stand alone, so it always needs the anchor.
+ if (masterTokens.length > 1 && hits === masterTokens.length) return true;
+
+ // Otherwise: a partial or single-word subject match, which has to be backed
+ // by a word the subject did not supply.
+ //
+ // An anchorless site cannot answer that, and answering "yes" by default
+ // would restore the old behaviour exactly. `resolveModifiers` always yields
+ // something, so an empty set here means a site with no subjects configured.
+ if (anchors.size === 0) return false;
+
+ for (const token of candidate) {
+ if (anchors.has(token)) return true;
+ }
+ return false;
+}
+
+/**
+ * The subjects to research, each already crossed with a narrowing term.
+ *
+ * Ordered subject-major — every subject's first cross comes before any
+ * subject's second — so that a caller which runs out of API budget partway
+ * through has still touched every subject rather than exhausting the first
+ * one. That ordering is the difference between a budget cut costing depth and
+ * costing coverage, and coverage is what was broken.
+ *
+ * A cross whose modifier tokens are already in the subject is skipped:
+ * "crypto" × "crypto payments" would otherwise research "crypto crypto
+ * payments".
+ */
+export function crossQueries(
+ masters: string[],
+ modifiers: string[],
+ perMaster = 3,
+): Array<{ master: string; query: string }> {
+ const out: Array<{ master: string; query: string }> = [];
+ if (masters.length === 0 || modifiers.length === 0) return out;
+
+ const seen = new Set();
+ for (let depth = 0; depth < perMaster; depth += 1) {
+ for (const master of masters) {
+ const masterTokens = new Set(tokens(master).map(stem));
+ // Each subject walks the modifier list from a different offset, so the
+ // narrowing terms are spread across subjects instead of every subject
+ // getting the same first modifier and the tail never being used.
+ const offset = masters.indexOf(master);
+ let taken = 0;
+ for (let i = 0; i < modifiers.length && taken <= depth; i += 1) {
+ const modifier = modifiers[(i + offset) % modifiers.length];
+ if (tokens(modifier).every((t) => masterTokens.has(stem(t)))) continue;
+ if (taken < depth) {
+ taken += 1;
+ continue;
+ }
+ const query = `${master} ${modifier}`.replace(/\s+/g, " ").trim();
+ const key = signature(query);
+ if (key && !seen.has(key)) {
+ seen.add(key);
+ out.push({ master, query });
+ }
+ taken += 1;
+ }
+ }
+ }
+ return out;
+}
+
+/**
+ * How many new keywords each subject should get.
+ *
+ * Fair share of the remaining target, but weighted toward whatever is *behind*
+ * — a subject with no coverage is filled before one that already has twenty
+ * articles. Without that, a queue topped up repeatedly stays in whatever
+ * proportion it started in, and the site that published nineteen peptide posts
+ * would go on publishing them at exactly the rate it always had.
+ *
+ * @param masters the subject list
+ * @param coverage existing keyword count per subject (any status)
+ * @param target how many rows to allocate in total
+ */
+export function allocate(
+ masters: string[],
+ coverage: Map,
+ target: number,
+): Map {
+ const out = new Map();
+ if (masters.length === 0 || target <= 0) return out;
+
+ for (const m of masters) out.set(m, 0);
+
+ // Hand out one row at a time to whichever subject is furthest behind,
+ // counting what has already been handed out in this pass. An O(target ×
+ // masters) loop over at most 30 × 12, which is not worth a heap to avoid and
+ // is far easier to prove correct than an apportionment formula.
+ for (let i = 0; i < target; i += 1) {
+ let pick = masters[0];
+ let lowest = Infinity;
+ for (const master of masters) {
+ const total = (coverage.get(master.toLowerCase()) ?? 0) + (out.get(master) ?? 0);
+ if (total < lowest) {
+ lowest = total;
+ pick = master;
+ }
+ }
+ out.set(pick, (out.get(pick) ?? 0) + 1);
+ }
+ return out;
+}
+
+/**
+ * Drop candidates that would produce an article the blog already has.
+ *
+ * Compares fingerprints rather than strings, so the pluralised and reordered
+ * restatements of an existing post are caught. This is the direct answer to
+ * "spamming blogs with same content": the old dedupe compared lowercased
+ * keywords exactly, which let "peptide payments" and "peptide payment"
+ * both through, and both were published — nine days apart, in May.
+ *
+ * @param candidates in preference order
+ * @param published fingerprints already on the blog, from `signature`
+ */
+export function dropDuplicates(
+ candidates: T[],
+ published: Set,
+): T[] {
+ const out: T[] = [];
+ const seen = new Set(published);
+ for (const candidate of candidates) {
+ const key = signature(candidate.keyword);
+ if (!key || seen.has(key)) continue;
+ seen.add(key);
+ out.push(candidate);
+ }
+ return out;
+}
diff --git a/lib/lx/webhookDeliver.ts b/lib/lx/webhookDeliver.ts
index 49bf3cf8..fb876632 100644
--- a/lib/lx/webhookDeliver.ts
+++ b/lib/lx/webhookDeliver.ts
@@ -14,6 +14,8 @@
import type { SupabaseClient } from "@supabase/supabase-js";
import { buildEvent, sendWebhook, type Post } from "@profullstack/autoblog";
import { env } from "../env";
+import { findExchangeCandidates } from "./exchangeMatcher";
+import { buildNetworkBlock } from "./networkBlock";
type ArticleRow = {
id: string;
@@ -44,6 +46,13 @@ type SiteRow = {
webhook_secret: string | null;
author_name: string | null;
author_url: string | null;
+ // Network opt-in, resolved for the site that will HOST the post. For a guest
+ // post that is the partner, not the author: it is the host's readers who see
+ // the ad unit and the host's owner who agreed to carry it.
+ project_id: string;
+ user_id: string;
+ niche: string | null;
+ ads_enabled: boolean | null;
};
/**
@@ -109,7 +118,7 @@ export type DeliveryResult = {
error?: string;
};
-function articleToPost(article: ArticleRow, site: SiteRow): Post {
+function articleToPost(article: ArticleRow, site: SiteRow, networkHtml = ""): Post {
const blogRoot = site.blog_root_url.replace(/\/$/, "");
const url = `${blogRoot}/${article.slug}`;
@@ -140,7 +149,13 @@ function articleToPost(article: ArticleRow, site: SiteRow): Post {
title: article.title,
slug: article.slug,
excerpt: article.excerpt || article.meta_description || null,
- html: `${jsonLd}\n${article.content_html}`,
+ // The network block goes after the article and outside the JSON-LD, so an
+ // ad unit and a list of other people's links are never described to a
+ // crawler as part of this article's body.
+ html: [jsonLd, article.content_html, networkHtml].filter(Boolean).join("\n"),
+ // Markdown deliberately does NOT carry the block. A receiver rendering the
+ // markdown path would have to trust our HTML through its own sanitiser,
+ // and the ad unit needs a script tag that no markdown renderer will emit.
markdown: article.content_markdown,
status: "published",
published_at: publishedAt,
@@ -187,7 +202,7 @@ export async function deliverArticle(
const { data: site } = await supabase
.from("lx_site")
.select(
- "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url",
+ "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url, project_id, user_id, niche, ads_enabled",
)
.eq("id", deliveryTargetId)
.maybeSingle();
@@ -208,7 +223,42 @@ export async function deliverArticle(
};
}
- const post = articleToPost(claimed, site);
+ // Ads and partner links for the hosting site, if it is in the network.
+ //
+ // Built here rather than at generation time because the host is only known
+ // once the guest-post target has been resolved, and because a redelivery
+ // should carry a current block rather than one frozen weeks ago. The whole
+ // thing is best-effort: `buildNetworkBlock` swallows its own failures, and
+ // this catch covers the rest, because an article that publishes without an
+ // ad has cost an impression while an article that fails to publish has cost
+ // the customer the thing they pay for.
+ const partnerLinks = await findExchangeCandidates(supabase, {
+ selfSiteId: site.id,
+ selfNiche: site.niche,
+ keyword: (claimed.tags ?? []).join(" ") || claimed.title,
+ slots: 3,
+ })
+ .then((r) =>
+ r.candidates.map((c) => ({
+ title: c.title,
+ url: c.url,
+ source: "partner" as const,
+ })),
+ )
+ .catch(() => []);
+
+ const networkHtml = await buildNetworkBlock(
+ supabase,
+ site,
+ claimed.tags ?? [],
+ partnerLinks,
+ env.siteUrl,
+ ).catch((err: unknown) => {
+ console.warn("[lx] network block failed:", err);
+ return "";
+ });
+
+ const post = articleToPost(claimed, site, networkHtml);
// Reuse the saved delivery id on retries so receivers idempotently
// dedupe. SDK uses event.id as the webhook-id header.
const event = buildEvent(post, {
diff --git a/scripts/purge-offniche-keywords.ts b/scripts/purge-offniche-keywords.ts
new file mode 100644
index 00000000..3fc4a32b
--- /dev/null
+++ b/scripts/purge-offniche-keywords.ts
@@ -0,0 +1,128 @@
+// Decide which QUEUED keywords the new gate would never have created.
+//
+// Reads a JSON dump of queued rows and prints the ids to delete, plus a
+// per-site summary. Prints SQL rather than executing it: this deletes
+// scheduled work on live blogs, and the diff between "what the gate rejects"
+// and "what I am about to remove" should be readable by a person before it
+// runs, not inferred from an exit code.
+//
+// Only `queued` rows are ever considered. A published article is a URL that
+// exists on somebody's blog and possibly in an index; removing its keyword row
+// would not unpublish it, it would only lose the record that we wrote it.
+//
+// Usage: npx tsx scripts/purge-offniche-keywords.ts
+
+import { readFileSync } from "node:fs";
+import { anchorTokens, isOnNiche, resolveMasters, stem, tokens } from "../lib/lx/topicPlan";
+
+type Row = {
+ id: string;
+ keyword: string;
+ domain: string;
+ niche: string | null;
+ master_keywords: string[] | null;
+ modifiers: string[] | null;
+};
+
+/** Same longest-match attribution the research pipeline uses. */
+function attribute(keyword: string, masters: string[]): string | null {
+ let best: string | null = null;
+ let bestLen = 0;
+ // Stemmed on both sides, so attribution and the gate agree about what
+ // "promo codes" and "promo code" are. They disagreed before, and a keyword
+ // attributed to a subject the gate then could not match was dropped for a
+ // reason nobody would have guessed from the strings.
+ const candidate = new Set(tokens(keyword).map(stem));
+ for (const master of masters) {
+ const masterTokens = tokens(master).map(stem);
+ if (masterTokens.length === 0) continue;
+ const hit = masterTokens.filter((t) => candidate.has(t));
+ if (hit.length === 0) continue;
+ const len = hit.join("").length;
+ if (len > bestLen) {
+ bestLen = len;
+ best = master;
+ }
+ }
+ return best;
+}
+
+/**
+ * Pull the rows out of whatever wrapper the dump arrived in.
+ *
+ * A plain array, or the MCP tool envelope — which is the JSON *escaped* inside
+ * a JSON string, so the quotes need unescaping before it will parse, and the
+ * rows then sit under a `payload` key.
+ */
+function extractRows(raw: string): Row[] {
+ const slice = raw.slice(raw.indexOf("[{"), raw.lastIndexOf("}]") + 2);
+ const text = slice.includes('\\"') ? slice.replace(/\\"/g, '"') : slice;
+ const parsed = JSON.parse(text);
+ const first = parsed[0];
+ return Array.isArray(first?.payload) ? first.payload : parsed;
+}
+
+const rows: Row[] = extractRows(readFileSync(process.argv[2], "utf8"));
+
+const bySite = new Map();
+
+const unjudgeable = new Set();
+
+for (const row of rows) {
+ const masters = resolveMasters(row);
+ const anchors = anchorTokens(row, masters);
+ const bucket = bySite.get(row.domain) ?? { kept: [], dropped: [], ids: [] };
+
+ // A site with no subjects configured cannot be judged, and "reject
+ // everything" is the wrong reading of "I have no basis for an opinion" —
+ // it would empty a queue whose keywords may be perfectly good, as they are
+ // on khipu-agency. Left alone, and named in the summary so the real fix
+ // (set master keywords) is visible.
+ if (masters.length === 0) {
+ unjudgeable.add(row.domain);
+ bucket.kept.push(row.keyword);
+ bySite.set(row.domain, bucket);
+ continue;
+ }
+
+ const master = attribute(row.keyword, masters);
+ const ok = master !== null && anchors.size > 0 && isOnNiche(row.keyword, master, anchors);
+
+ if (ok) bucket.kept.push(row.keyword);
+ else {
+ bucket.dropped.push(row.keyword);
+ bucket.ids.push(row.id);
+ }
+ bySite.set(row.domain, bucket);
+}
+
+const allIds: string[] = [];
+let totalKept = 0;
+let totalDropped = 0;
+
+for (const [domain, b] of Array.from(bySite).sort()) {
+ totalKept += b.kept.length;
+ totalDropped += b.dropped.length;
+ allIds.push(...b.ids);
+ console.log(
+ `\n=== ${domain}: keep ${b.kept.length}, drop ${b.dropped.length}`,
+ );
+ console.log(` dropping : ${b.dropped.slice(0, 12).join(" | ")}${b.dropped.length > 12 ? " | …" : ""}`);
+ console.log(` keeping : ${b.kept.slice(0, 8).join(" | ")}${b.kept.length > 8 ? " | …" : ""}`);
+}
+
+console.log(`\n--- total: keep ${totalKept}, drop ${totalDropped} of ${rows.length}`);
+if (unjudgeable.size > 0) {
+ console.log(
+ `--- left alone (no master keywords set): ${Array.from(unjudgeable).join(", ")}`,
+ );
+}
+console.log("\n-- SQL:");
+for (let i = 0; i < allIds.length; i += 200) {
+ const chunk = allIds.slice(i, i + 200);
+ console.log(
+ `delete from lx_keyword where status='queued' and id in (${chunk
+ .map((id) => `'${id}'`)
+ .join(",")});`,
+ );
+}
diff --git a/scripts/verify-topic-plan.ts b/scripts/verify-topic-plan.ts
new file mode 100644
index 00000000..4c567971
--- /dev/null
+++ b/scripts/verify-topic-plan.ts
@@ -0,0 +1,96 @@
+// Dry-run the topic planner against live rows and print what it WOULD do.
+//
+// No writes, no API calls. Reads each active site's real subjects, modifiers
+// and existing keyword history, and reports the allocation plus the cross
+// queries. Exists so the fix can be checked against production shapes rather
+// than only against the fixtures in tests/lx/topic-plan.test.ts.
+//
+// Usage: npx tsx scripts/verify-topic-plan.ts (needs .env with the service key)
+
+import { createClient } from "@supabase/supabase-js";
+import {
+ allocate,
+ anchorTokens,
+ crossQueries,
+ isOnNiche,
+ resolveMasters,
+ resolveModifiers,
+} from "../lib/lx/topicPlan";
+
+const supabase = createClient(
+ process.env.NEXT_PUBLIC_SUPABASE_URL!,
+ process.env.SUPABASE_SERVICE_ROLE_KEY!,
+ { auth: { persistSession: false } },
+);
+
+async function main() {
+ const { data: sites } = await supabase
+ .from("lx_site")
+ .select("id, domain, niche, master_keywords, seed_keywords, modifiers")
+ .eq("status", "active")
+ .order("domain");
+
+ for (const site of sites ?? []) {
+ const masters = resolveMasters(site as never);
+ const modifiers = resolveModifiers(site as never);
+ const anchors = anchorTokens(site as never);
+
+ const { data: existing } = await supabase
+ .from("lx_keyword")
+ .select("keyword, master_keyword")
+ .eq("site_id", site.id);
+
+ const coverage = new Map();
+ for (const row of existing ?? []) {
+ const master =
+ row.master_keyword ??
+ masters.find((m) =>
+ row.keyword.toLowerCase().includes(m.toLowerCase()),
+ );
+ if (master) {
+ const k = master.toLowerCase();
+ coverage.set(k, (coverage.get(k) ?? 0) + 1);
+ }
+ }
+
+ const plan = allocate(masters, coverage, 30);
+ const crosses = crossQueries(masters, modifiers, 3);
+ const coveredByCross = new Set(crosses.map((c) => c.master));
+
+ console.log(`\n=== ${site.domain}`);
+ console.log(` niche : ${site.niche ?? "(none)"}`);
+ console.log(` masters : ${masters.length ? masters.join(", ") : "(NONE)"}`);
+ console.log(` modifiers : ${modifiers.length ? modifiers.join(", ") : "(NONE)"}`);
+ console.log(` anchors : ${anchors.size}`);
+ if (anchors.size === 0) {
+ console.log(" !! would ERROR: no niche and no modifiers");
+ continue;
+ }
+ const starved = masters.filter((m) => !coveredByCross.has(m));
+ if (starved.length) {
+ console.log(` !! no cross floor for: ${starved.join(", ")}`);
+ }
+ console.log(
+ ` allocation : ${masters
+ .map((m) => `${m}=${plan.get(m) ?? 0}(have ${coverage.get(m.toLowerCase()) ?? 0})`)
+ .join(" ")}`,
+ );
+ console.log(` sample : ${crosses.slice(0, 6).map((c) => c.query).join(" | ")}`);
+
+ // How the existing published keywords score under the new gate.
+ const kept = (existing ?? []).filter((r) => {
+ const master = masters.find((m) =>
+ r.keyword.toLowerCase().includes(m.toLowerCase()),
+ );
+ return master ? isOnNiche(r.keyword, master, anchors) : false;
+ });
+ console.log(
+ ` gate : ${kept.length}/${(existing ?? []).length} existing keywords would pass`,
+ );
+ }
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql b/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql
new file mode 100644
index 00000000..7f764ed4
--- /dev/null
+++ b/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql
@@ -0,0 +1,82 @@
+-- Master keywords, topic provenance, and the network opt-in.
+--
+-- Three columns, one root cause.
+--
+-- `seed_keywords` had become two lists wearing one name: the durable set of
+-- subjects a blog is *about*, and the working set the research pipeline was
+-- allowed to chew on. Nothing enforced a size on it, so it grew — ten entries
+-- on coinpayportal — and the pipeline quietly truncated it twice on the way
+-- through (`buildSeeds` to five, the DataForSEO loop to three). The seeds past
+-- the cut had never produced a single keyword: thirty-two of them, across nine
+-- sites. Splitting the durable list out under its own name and capping it is
+-- what makes that class of bug impossible to reintroduce, because a list a
+-- human can hold in their head is one the planner can afford to cover *all* of.
+--
+-- `master_keyword` on lx_keyword is the provenance the planner needs to do
+-- that. Fair allocation across topics is not expressible without knowing which
+-- topic each queued row came from, and inferring it after the fact by
+-- substring-matching the keyword back to a seed is exactly the loose matching
+-- that let "skye peptides" through the relevance gate in the first place.
+--
+-- `ads_enabled` is the network opt-in: house ads and partner guest posts in
+-- published articles. Default true, because a network everybody has to be
+-- asked to join is one that stays empty.
+
+alter table public.lx_site
+ add column if not exists master_keywords text[] not null default '{}',
+ add column if not exists ads_enabled boolean not null default true;
+
+comment on column public.lx_site.master_keywords is
+ 'The durable 3-12 subjects this blog covers. The keyword planner allocates its target evenly across ALL of these; nothing truncates it. Crossed with `modifiers` to stay anchored to the site''s own niche.';
+
+comment on column public.lx_site.ads_enabled is
+ 'Opted into the CrawlProof network: house/partner ad units and guest posts are placed in published articles. Default true.';
+
+-- A cap the planner can honour rather than a limit it has to work around.
+-- Twelve is the point past which an even allocation of a 30-keyword target
+-- stops giving each topic enough rows to be worth researching separately.
+do $$
+begin
+ if not exists (
+ select 1 from pg_constraint where conname = 'lx_site_master_keywords_len'
+ ) then
+ alter table public.lx_site
+ add constraint lx_site_master_keywords_len
+ check (coalesce(array_length(master_keywords, 1), 0) <= 12);
+ end if;
+end $$;
+
+-- Which master subject a queued keyword was researched for.
+--
+-- Nullable, and stays nullable: every row written before this migration has no
+-- honest answer, and guessing one would corrupt the very coverage figures the
+-- planner reads. Those rows are simply invisible to the fair-share maths, which
+-- is the correct behaviour — they are already published or queued, so the
+-- topics they belong to need no further filling on their account.
+alter table public.lx_keyword
+ add column if not exists master_keyword text;
+
+comment on column public.lx_keyword.master_keyword is
+ 'The lx_site.master_keywords entry this keyword was researched for. Null on rows predating topic provenance; those are excluded from coverage rather than guessed at.';
+
+create index if not exists lx_keyword_site_master_idx
+ on public.lx_keyword (site_id, master_keyword);
+
+-- Seed the durable list from what each site already had.
+--
+-- First ten only, mirroring the cap, and only where the column is still empty
+-- so a re-run cannot stomp a hand-curated list. This is the one place the old
+-- truncation is preserved on purpose: a site with more than ten seeds had no
+-- coverage of the tail anyway, and the operator should choose which ten matter
+-- rather than have position in an unordered array decide it.
+update public.lx_site
+set master_keywords = (
+ select coalesce(array_agg(s order by ord), '{}')
+ from (
+ select s, ord from unnest(seed_keywords) with ordinality as t(s, ord)
+ where length(trim(s)) > 0
+ limit 10
+ ) picked
+)
+where coalesce(array_length(master_keywords, 1), 0) = 0
+ and coalesce(array_length(seed_keywords, 1), 0) > 0;
diff --git a/supabase/migrations/20260828140000_lx_feed_sources.sql b/supabase/migrations/20260828140000_lx_feed_sources.sql
new file mode 100644
index 00000000..99211548
--- /dev/null
+++ b/supabase/migrations/20260828140000_lx_feed_sources.sql
@@ -0,0 +1,66 @@
+-- The directory feeds the autoblog reads, and what happened last time.
+--
+-- Until now these were fetched live, inside article delivery, on a 3-second
+-- timeout. That put a third-party HTTP call on the critical path of the one
+-- operation a customer is paying for, and it made the whole source invisible:
+-- when a topic feed went missing the block simply came back empty, and there
+-- was nowhere to look. Both problems have the same fix — crawl on a schedule,
+-- cache what came back, and keep the record of the attempt.
+--
+-- The source list is derived, never curated. Topics are the slugged
+-- master_keywords of every active site, so a blog that adds a subject gets its
+-- feed crawled on the next sweep with nobody filing a request. That is the
+-- point: the operator sets subjects, and the feed list follows.
+
+create table if not exists public.lx_feed_source (
+ id uuid primary key default gen_random_uuid(),
+ -- The directory's own topic slug. Unique because two sites covering the
+ -- same subject must share one crawl rather than each causing their own.
+ topic text not null unique,
+ url text not null,
+ -- active: crawl it. given_up: too many consecutive failures, left in place
+ -- so the status page can still explain the absence rather than the row
+ -- silently vanishing and the topic looking like it was never wanted.
+ status text not null default 'active'
+ check (status in ('active', 'given_up')),
+ last_fetch_at timestamptz,
+ last_success_at timestamptz,
+ last_status integer,
+ last_error text,
+ item_count integer not null default 0,
+ consecutive_failures integer not null default 0,
+ created_at timestamptz not null default now(),
+ updated_at timestamptz not null default now()
+);
+
+comment on table public.lx_feed_source is
+ 'RSS Amplifier topic feeds the autoblog cites from. Derived from active sites'' master_keywords by the worker''s feed sweep — not hand-curated.';
+
+-- The sweep picks the least-recently-fetched active source, so this is the
+-- index that decides throughput.
+create index if not exists lx_feed_source_due_idx
+ on public.lx_feed_source (status, last_fetch_at nulls first);
+
+create table if not exists public.lx_feed_item (
+ id uuid primary key default gen_random_uuid(),
+ source_id uuid not null references public.lx_feed_source(id) on delete cascade,
+ title text not null,
+ link text not null,
+ first_seen_at timestamptz not null default now(),
+ -- Link rather than guid: the link is what gets rendered, and it is the only
+ -- field whose uniqueness actually prevents the same anchor appearing twice.
+ unique (source_id, link)
+);
+
+comment on table public.lx_feed_item is
+ 'Cached posts from a topic feed. Read by the article delivery path so a third-party fetch is never on the critical path of publishing.';
+
+create index if not exists lx_feed_item_source_seen_idx
+ on public.lx_feed_item (source_id, first_seen_at desc);
+
+-- Service-role only. Nothing here is user data and no browser reads it
+-- directly; the status page goes through a server component holding the
+-- service client. Enabling RLS with no policy is what makes that explicit
+-- rather than incidental.
+alter table public.lx_feed_source enable row level security;
+alter table public.lx_feed_item enable row level security;
diff --git a/tests/lx/feed-crawl.test.ts b/tests/lx/feed-crawl.test.ts
new file mode 100644
index 00000000..ac5d29a7
--- /dev/null
+++ b/tests/lx/feed-crawl.test.ts
@@ -0,0 +1,50 @@
+import { describe, expect, it } from "vitest";
+import { topicFor } from "@/lib/lx/feedCrawl";
+import { ago } from "@/lib/lx/feedCrawlStats";
+
+describe("topicFor", () => {
+ it("uses the first meaningful token of a multi-word subject", () => {
+ // "merchant account payments" is not a directory topic; "merchant" is.
+ expect(topicFor("merchant account payments")).toBe("merchant");
+ });
+
+ it("keeps a short subject that the token floor would otherwise drop", () => {
+ // These are exactly the high-value subjects — four of the five that had
+ // never produced a keyword were short words. Losing them here would
+ // reintroduce the same blind spot one layer down.
+ expect(topicFor("iptv")).toBe("iptv");
+ expect(topicFor("weed")).toBe("weed");
+ });
+
+ it("slugs punctuation rather than emitting an unfetchable path", () => {
+ expect(topicFor("high-risk")).toBe("high");
+ });
+
+ it("returns null for a subject with nothing in it", () => {
+ expect(topicFor(" ")).toBeNull();
+ expect(topicFor("!!!")).toBeNull();
+ });
+});
+
+describe("ago", () => {
+ const now = Date.parse("2026-08-28T12:00:00.000Z");
+
+ it("reports never for a source that has not been read", () => {
+ expect(ago(null, now)).toBe("never");
+ });
+
+ it.each([
+ ["2026-08-28T11:45:00.000Z", "15m ago"],
+ ["2026-08-28T09:00:00.000Z", "3h ago"],
+ ["2026-08-26T12:00:00.000Z", "2d ago"],
+ ["2026-08-28T11:59:45.000Z", "just now"],
+ ])("formats %j as %j", (iso, expected) => {
+ expect(ago(iso, now)).toBe(expected);
+ });
+
+ it("does not render a negative age when a clock is ahead", () => {
+ // Worker and web can be on different hosts; "-3m ago" reads as a bug in
+ // the page rather than a skewed clock, so it is clamped.
+ expect(ago("2026-08-28T12:03:00.000Z", now)).toBe("just now");
+ });
+});
diff --git a/tests/lx/network-block.test.ts b/tests/lx/network-block.test.ts
new file mode 100644
index 00000000..fad6b1f4
--- /dev/null
+++ b/tests/lx/network-block.test.ts
@@ -0,0 +1,133 @@
+// The insertion block writes other people's text into a customer's page.
+//
+// Most of what follows is injection and link-safety, because that is what this
+// module actually risks. The titles come from RSS feeds crawled off the open
+// web; the fact that they reach us via our own directory does not make them
+// ours, and the page they land on belongs to somebody paying us.
+
+import { describe, expect, it } from "vitest";
+import {
+ adUnitHtml,
+ escapeHtml,
+ isSafeHref,
+ networkLinksHtml,
+ type NetworkLink,
+} from "@/lib/lx/networkBlock";
+import { itemEntries } from "@/lib/lx/feedTopics";
+
+describe("escapeHtml", () => {
+ it("neutralises a script tag in a feed title", () => {
+ expect(escapeHtml('')).toBe(
+ "<script>alert(1)</script>",
+ );
+ });
+
+ it("escapes the ampersand first so nothing is double-escaped", () => {
+ // "<" must not become "<" — which is what a naive ordering does.
+ expect(escapeHtml("<")).toBe("<");
+ expect(escapeHtml("&")).toBe("&");
+ expect(escapeHtml("&<")).toBe("&<");
+ });
+
+ it("escapes both quote characters, since these land in attributes", () => {
+ expect(escapeHtml(`" onload="x`)).toBe("" onload="x");
+ expect(escapeHtml("' onload='x")).toBe("' onload='x");
+ });
+});
+
+describe("isSafeHref", () => {
+ it.each(["javascript:alert(1)", "data:text/html,',
+ url: "https://example.com/x",
+ source: "directory",
+ },
+ ]);
+ expect(html).not.toContain("