diff --git a/app/(app)/dashboard/autoblog/crawler/page.tsx b/app/(app)/dashboard/autoblog/crawler/page.tsx new file mode 100644 index 00000000..f056fa4f --- /dev/null +++ b/app/(app)/dashboard/autoblog/crawler/page.tsx @@ -0,0 +1,168 @@ +// Live view of the feed crawler, modelled on rssamplifier.com/crawlstats. +// +// The autoblog cites real posts from RSS Amplifier topic feeds, and until the +// crawl was daemonized that source was invisible: a topic going missing looked +// identical to a topic nobody had configured. This is where that difference +// becomes visible. +// +// Deliberately read-only and unconfigurable. The feed list is derived from +// every active site's master keywords on each sweep, so there is nothing here +// to add or remove — a subject added to a blog appears within a tick. An +// "add a feed" button would imply a curation step that does not exist, and +// would create a queue of requests waiting for a human, which is the thing +// the automation is meant to remove. + +import { serviceClient } from "@/lib/supabase/service"; +import { ago, loadCrawlerStats } from "@/lib/lx/feedCrawlStats"; + +// Always fresh: a status page served from a cache reports the cache's health, +// not the crawler's, and would go on saying "healthy" through an outage. +export const dynamic = "force-dynamic"; + +function Stat({ + label, + value, + note, +}: { + label: string; + value: string; + note?: string; +}) { + return ( +
+
{value}
+
+ {label} +
+ {note ? ( +
{note}
+ ) : null} +
+ ); +} + +export default async function CrawlerStatusPage() { + const { stats, sources } = await loadCrawlerStats(serviceClient()); + const now = Date.now(); + + return ( +
+

Crawler status

+

+ Live view of the daemon that reads the RSS Amplifier topic feeds the + autoblog cites from. The feed list is derived from every active + site's master keywords — nothing here is curated by hand. The same + numbers are available as{" "} + + JSON + + . +

+ + {stats.stalled ? ( +
+ Stalled. {stats.dueNow.toLocaleString()} feed(s) are + due and nothing has been read successfully since{" "} + {ago(stats.lastSuccessAt, now)}. Either the worker is not running or + the directory is unreachable — this page cannot tell which, and the + two have different fixes. +
+ ) : null} + +
+ + + + +
+ +

+ Last successful fetch {ago(stats.lastSuccessAt, now)} ·{" "} + {stats.neverFetched.toLocaleString()} never fetched · generated{" "} + {stats.generatedAt} +

+ +

Feeds

+

+ One row per subject any active blog covers. A feed is re-read at most + every 6 hours, least-recently-fetched first, 25 per sweep — so the work + per tick is fixed however many subjects the platform grows to. +

+ + {sources.length === 0 ? ( +

+ No feeds yet. The first sweep derives them from the active sites' + master keywords; if this stays empty, no site has master keywords set. +

+ ) : ( +
+ + + + + + + + + + + + + {sources.map((source) => ( + + + + + + + + + ))} + +
TopicStatusLast readLast successItemsLast error
+ + {source.topic} + + + {source.status === "given_up" ? ( + + given up + + ) : source.consecutive_failures > 0 ? ( + failing ({source.consecutive_failures}) + ) : ( + active + )} + {ago(source.last_fetch_at, now)} + {ago(source.last_success_at, now)} + + {source.item_count.toLocaleString()} + + {source.last_error ?? ""} +
+
+ )} +
+ ); +} diff --git a/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx b/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx index e21e7329..5ab4bc9f 100644 --- a/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx +++ b/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx @@ -23,8 +23,10 @@ type Existing = { target_audiences: string[]; description: string; seed_keywords: string[]; + master_keywords: string[]; modifiers: string[]; preserve_keywords: boolean; + ads_enabled: boolean; keywords: string[]; seo_title: string | null; seo_description: string | null; @@ -109,6 +111,14 @@ export function SetupForm({ (initial?.target_audiences ?? []).join(", "), ); const [description, setDescription] = useState(initial?.description ?? ""); + const [masterKeywords, setMasterKeywords] = useState( + // Falls back to the seed list for any site the backfill has not reached, + // so the field is never blank on a project that already has subjects. + (initial?.master_keywords?.length + ? initial.master_keywords + : (initial?.seed_keywords ?? []) + ).join(", "), + ); const [seedKeywords, setSeedKeywords] = useState( (initial?.seed_keywords ?? []).join(", "), ); @@ -118,6 +128,11 @@ export function SetupForm({ const [preserveKeywords, setPreserveKeywords] = useState( initial?.preserve_keywords ?? false, ); + // Default ON, matching the column default. A new project is in the network + // unless its owner takes the tick out. + const [adsEnabled, setAdsEnabled] = useState( + initial?.ads_enabled ?? true, + ); const [keywords, setKeywords] = useState( (initial?.keywords ?? []).join("\n"), ); @@ -425,6 +440,13 @@ export function SetupForm({ // Each line is a CSV row ",". For seed-building and // dedupe we only care about the keyword half (everything before the // first comma). + function masterKeywordsAsArray(): string[] { + return masterKeywords + .split(",") + .map((s) => s.trim()) + .filter(Boolean); + } + function keywordsAsArray(): string[] { return keywords .split("\n") @@ -569,9 +591,11 @@ export function SetupForm({ niche, targetAudiences: audiences, description, + masterKeywords, seedKeywords, modifiers, preserveKeywords, + adsEnabled, keywords, seoTitle, seoDescription, @@ -843,6 +867,34 @@ export function SetupForm({ onChange={(e) => setDescription(e.target.value)} /> +
+ + setMasterKeywords(e.target.value)} + /> +

+ The subjects this blog covers, and the list it always rebuilds from. + The planner splits every research run evenly across{" "} + all of them and fills whichever is furthest behind + first, so a subject cannot run away with the schedule. Each one is + researched crossed with a modifier below — never on its own.{" "} + {masterKeywordsAsArray().length > 12 ? ( + + {masterKeywordsAsArray().length} entered — only the first 12 are + kept. + + ) : ( + <>{masterKeywordsAsArray().length} subject(s). + )} +

+
+
+
+ +
+
diff --git a/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx b/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx index fa990710..b5201701 100644 --- a/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx +++ b/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx @@ -7,7 +7,7 @@ import { SetupForm } from "./form"; export const metadata = { title: "Autoblog · Setup" }; const SITE_COLUMNS = - "id, domain, blog_root_url, sitemap_url, niche, target_audiences, description, seed_keywords, modifiers, preserve_keywords, keywords, seo_title, seo_description, tone, competitors, webhook_url, webhook_secret, daily_article_count, publish_days, publish_hour, internal_links_per_article, backlinks_enabled, external_links_per_article, banner_style, status"; + "id, domain, blog_root_url, sitemap_url, niche, target_audiences, description, seed_keywords, master_keywords, modifiers, preserve_keywords, ads_enabled, keywords, seo_title, seo_description, tone, competitors, webhook_url, webhook_secret, daily_article_count, publish_days, publish_hour, internal_links_per_article, backlinks_enabled, external_links_per_article, banner_style, status"; export default async function AutoblogSetupPage({ params, diff --git a/app/actions/linkExchange.ts b/app/actions/linkExchange.ts index 514ff120..1d109971 100644 --- a/app/actions/linkExchange.ts +++ b/app/actions/linkExchange.ts @@ -289,6 +289,9 @@ export type SiteInput = { niche: string; targetAudiences: string; description: string; + // Comma-separated 3-12 subjects this blog covers. The durable list the + // keyword planner allocates evenly across; see lib/lx/topicPlan. + masterKeywords: string; // Comma-separated 1-3 word head terms for DataForSEO expansion. seedKeywords: string; // Comma-separated tail terms ("payments", "merchant account") that the @@ -297,6 +300,9 @@ export type SiteInput = { // When true, Refetch flows skip overwriting the keywords text[]. Used // after a hand-curated build via seeds × modifiers. preserveKeywords: boolean; + // Network opt-in: ad units and partner/directory links in published + // articles. Default true — see the migration note on lx_site.ads_enabled. + adsEnabled: boolean; // Comma-separated. parseList() into a text[]. keywords: string; seoTitle: string; @@ -360,6 +366,12 @@ export async function createOrUpdateSite( const seedKeywords = parseList(input.seedKeywords, 50, 40); const modifiers = parseList(input.modifiers, 20, 40); const preserveKeywords = !!input.preserveKeywords; + // The DB constraint caps this at 12; parse to the same number so an + // oversized paste is trimmed here rather than rejected as a write error the + // form cannot explain. + const masterKeywords = parseList(input.masterKeywords, 12, MAX.keyword); + // Absent reads as opted in, matching the column default. + const adsEnabled = input.adsEnabled !== false; // Keywords use parseKeywordRows so each row keeps its `,` hint. // Allow up to ~80 chars per row to fit "keyword phrase here,12345". const keywords = parseKeywordRows(input.keywords, MAX.keywords, MAX.keyword + 20); @@ -433,6 +445,8 @@ export async function createOrUpdateSite( target_audiences: audiences, description, seed_keywords: seedKeywords, + master_keywords: masterKeywords, + ads_enabled: adsEnabled, modifiers, preserve_keywords: preserveKeywords, keywords, @@ -489,6 +503,8 @@ export async function createOrUpdateSite( target_audiences: audiences, description, seed_keywords: seedKeywords, + master_keywords: masterKeywords, + ads_enabled: adsEnabled, modifiers, preserve_keywords: preserveKeywords, keywords, @@ -534,6 +550,8 @@ export async function createOrUpdateSite( target_audiences: audiences, description, seed_keywords: seedKeywords, + master_keywords: masterKeywords, + ads_enabled: adsEnabled, modifiers, preserve_keywords: preserveKeywords, keywords, diff --git a/app/api/lx/feed-crawl/status/route.ts b/app/api/lx/feed-crawl/status/route.ts new file mode 100644 index 00000000..d53cbdba --- /dev/null +++ b/app/api/lx/feed-crawl/status/route.ts @@ -0,0 +1,20 @@ +import { NextResponse } from "next/server"; +import { serviceClient } from "@/lib/supabase/service"; +import { loadCrawlerStats } from "@/lib/lx/feedCrawlStats"; + +export const runtime = "nodejs"; +// A cached status response reports the cache's health rather than the +// crawler's, and would keep answering "fine" straight through an outage. +export const dynamic = "force-dynamic"; + +/** + * Counts only. + * + * The per-feed rows stay on the dashboard page behind a session: this is the + * shape a monitor polls, and it has no reason to carry the error strings some + * upstream server wrote. + */ +export async function GET() { + const { stats } = await loadCrawlerStats(serviceClient()); + return NextResponse.json(stats, { headers: { "cache-control": "no-store" } }); +} diff --git a/lib/lx/feedCrawl.ts b/lib/lx/feedCrawl.ts new file mode 100644 index 00000000..f20c4ae5 --- /dev/null +++ b/lib/lx/feedCrawl.ts @@ -0,0 +1,285 @@ +// The daemon side of the directory feeds. +// +// `postsFromTopicFeeds` used to fetch RSS live, inside article delivery, on a +// three-second budget. That is a third-party HTTP call on the critical path of +// the one operation a customer pays for, and it made the source invisible: a +// topic feed going missing showed up as an empty block and nothing else. +// +// This moves the fetching onto the worker's own schedule and leaves a record +// of every attempt. Delivery then reads rows. The two halves fail +// independently — a dead directory ages the cache instead of slowing down a +// publish, and a publish never waits on anybody else's server. +// +// The source list is derived on every sweep from the master_keywords of every +// active site. Nobody maintains it. A blog that adds a subject has that +// subject's feed crawled on the next tick, which is the behaviour "no human +// intervention" actually requires — a curated feed list is a queue of requests +// waiting for somebody. + +import type { SupabaseClient } from "@supabase/supabase-js"; +import { itemEntries } from "./feedTopics"; +import { resolveMasters, tokens } from "./topicPlan"; + +const RSSAMPLIFIER = "https://rssamplifier.com"; + +/** Per-feed budget. Generous next to delivery's, because nothing waits on it. */ +const FETCH_TIMEOUT_MS = 12_000; + +/** Feeds read per sweep. Bounded so one tick cannot run for an hour. */ +const PER_SWEEP = 25; + +/** Don't re-read a feed more often than this. */ +export const REFRESH_AFTER_MS = 6 * 60 * 60 * 1000; // 6h + +/** + * Consecutive failures before a source is parked. + * + * Parked, not deleted. A deleted row is re-derived on the very next sweep from + * the same master keyword and starts failing again immediately — an infinite + * retry wearing the costume of a clean table. The row stays so the status page + * can say "this topic has no feed, here is why". + */ +const GIVE_UP_AFTER = 8; + +/** Items kept per source. Older ones are pruned so the table stays bounded. */ +const KEEP_PER_SOURCE = 40; + +/** + * The directory's slug for a subject. + * + * Mirrors `topicSlug` in feedTopics but drops the stopword-free tokenisation + * through `tokens` first, so a master keyword of "merchant account payments" + * slugs to something the directory actually has a topic for rather than a long + * phrase that will always 404. + */ +export function topicFor(master: string): string | null { + const parts = tokens(master); + if (parts.length === 0) { + // A short subject ("iptv", "weed") has no tokens over the length floor but + // is still a real topic — fall back to the raw slug rather than dropping + // exactly the short, high-value subjects. + const raw = (master ?? "").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, ""); + return raw || null; + } + return parts[0]; +} + +export type FeedSweepResult = { + sources: number; + fetched: number; + succeeded: number; + newItems: number; + gaveUp: number; +}; + +/** + * Make sure every subject any active site covers has a source row. + * + * Insert-only, ignoring conflicts: a topic already present keeps its history, + * and two sites sharing a subject share one crawl. + */ +export async function syncFeedSources( + supabase: SupabaseClient, +): Promise { + const { data: sites } = await supabase + .from("lx_site") + .select("master_keywords, seed_keywords") + .eq("status", "active"); + + const topics = new Set(); + for (const site of (sites ?? []) as Array<{ + master_keywords: string[] | null; + seed_keywords: string[] | null; + }>) { + for (const master of resolveMasters(site)) { + const topic = topicFor(master); + if (topic) topics.add(topic); + } + } + if (topics.size === 0) return 0; + + const rows = Array.from(topics).map((topic) => ({ + topic, + url: `${RSSAMPLIFIER}/topics/${encodeURIComponent(topic)}.rss`, + })); + + const { error } = await supabase + .from("lx_feed_source") + .upsert(rows, { onConflict: "topic", ignoreDuplicates: true }); + if (error) console.warn("[lx] feed source sync:", error.message); + + return topics.size; +} + +/** + * Read the feeds that are due. + * + * Least-recently-fetched first, capped, so the sweep is a fixed amount of work + * regardless of how many subjects the platform covers. + */ +export async function crawlFeeds( + supabase: SupabaseClient, + fetchImpl: typeof fetch = fetch, +): Promise { + const sources = await syncFeedSources(supabase); + const due = new Date(Date.now() - REFRESH_AFTER_MS).toISOString(); + + const { data: batch } = await supabase + .from("lx_feed_source") + .select("id, topic, url, consecutive_failures") + .eq("status", "active") + .or(`last_fetch_at.is.null,last_fetch_at.lt.${due}`) + .order("last_fetch_at", { ascending: true, nullsFirst: true }) + .limit(PER_SWEEP); + + const result: FeedSweepResult = { + sources, + fetched: 0, + succeeded: 0, + newItems: 0, + gaveUp: 0, + }; + + for (const source of (batch ?? []) as Array<{ + id: string; + topic: string; + url: string; + consecutive_failures: number; + }>) { + result.fetched += 1; + const now = new Date().toISOString(); + + let status: number | null = null; + let entries: Array<{ title: string; link: string | null }> = []; + let error: string | null = null; + + try { + const res = await fetchImpl(source.url, { + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + headers: { accept: "application/rss+xml, application/xml;q=0.9" }, + }); + status = res.status; + if (res.ok) { + entries = itemEntries(await res.text()).filter((e) => e.link); + } else { + error = `HTTP ${res.status}`; + } + } catch (err) { + error = err instanceof Error ? err.message : String(err); + } + + // A 200 carrying no items is not a success. It is what the directory + // returns for a topic nobody publishes under, and counting it as one + // would let a permanently empty topic sit at the front of the + // least-recently-fetched queue for ever, crowding out real feeds. + const ok = error === null && entries.length > 0; + if (!ok && !error) error = "no usable items"; + + if (ok) { + result.succeeded += 1; + const rows = entries.slice(0, KEEP_PER_SOURCE).map((e) => ({ + source_id: source.id, + title: e.title, + link: e.link as string, + })); + const { data: inserted } = await supabase + .from("lx_feed_item") + .upsert(rows, { onConflict: "source_id,link", ignoreDuplicates: true }) + .select("id"); + result.newItems += inserted?.length ?? 0; + await pruneItems(supabase, source.id); + } + + const failures = ok ? 0 : source.consecutive_failures + 1; + const givingUp = failures >= GIVE_UP_AFTER; + if (givingUp) result.gaveUp += 1; + + await supabase + .from("lx_feed_source") + .update({ + last_fetch_at: now, + ...(ok ? { last_success_at: now, item_count: entries.length } : {}), + last_status: status, + last_error: ok ? null : error, + consecutive_failures: failures, + status: givingUp ? "given_up" : "active", + updated_at: now, + }) + .eq("id", source.id); + } + + return result; +} + +/** Keep the newest KEEP_PER_SOURCE items and drop the rest. */ +async function pruneItems(supabase: SupabaseClient, sourceId: string): Promise { + const { data: keep } = await supabase + .from("lx_feed_item") + .select("id") + .eq("source_id", sourceId) + .order("first_seen_at", { ascending: false }) + .limit(KEEP_PER_SOURCE); + const ids = (keep ?? []).map((r: { id: string }) => r.id); + if (ids.length < KEEP_PER_SOURCE) return; + await supabase + .from("lx_feed_item") + .delete() + .eq("source_id", sourceId) + .not("id", "in", `(${ids.join(",")})`); +} + +/** + * Cached posts for a set of subjects. + * + * The read side of the crawl, used by article delivery. Returns nothing rather + * than falling back to a live fetch: a cache miss here means the sweep has not + * reached that topic yet, and the correct behaviour is an article without a + * citation block, not a publish that blocks on somebody else's server. The + * next article gets the block. + */ +export async function cachedFeedPosts( + supabase: SupabaseClient, + masters: string[], + limit = 3, +): Promise> { + const topics = Array.from( + new Set(masters.map(topicFor).filter((t): t is string => !!t)), + ); + if (topics.length === 0) return []; + + const { data } = await supabase + .from("lx_feed_item") + .select("title, link, source:lx_feed_source!inner(topic)") + .in("source.topic", topics) + .order("first_seen_at", { ascending: false }) + .limit(limit * 12); + + type Row = { title: string; link: string; source: { topic: string } | { topic: string }[] }; + const byTopic = new Map>(); + for (const row of (data ?? []) as Row[]) { + const source = Array.isArray(row.source) ? row.source[0] : row.source; + const topic = source?.topic; + if (!topic) continue; + const bucket = byTopic.get(topic) ?? []; + bucket.push({ title: row.title, link: row.link, topic }); + byTopic.set(topic, bucket); + } + + // One per topic before a second from any, so three citations show three + // subjects rather than three posts from whichever feed is busiest. + const out: Array<{ title: string; link: string; topic: string }> = []; + const queues = Array.from(byTopic.values()); + let live = true; + while (live && out.length < limit) { + live = false; + for (const queue of queues) { + if (out.length >= limit) break; + const next = queue.shift(); + if (next) { + out.push(next); + live = true; + } + } + } + return out; +} diff --git a/lib/lx/feedCrawlStats.ts b/lib/lx/feedCrawlStats.ts new file mode 100644 index 00000000..745bbcad --- /dev/null +++ b/lib/lx/feedCrawlStats.ts @@ -0,0 +1,119 @@ +// The numbers behind the crawler status page. +// +// Kept out of the page so the HTML view and the JSON endpoint cannot drift — +// a status page and a machine-readable feed of the same status that disagree +// are worse than having only one of them. + +import type { SupabaseClient } from "@supabase/supabase-js"; +import { REFRESH_AFTER_MS } from "./feedCrawl"; + +export type FeedSourceRow = { + id: string; + topic: string; + url: string; + status: string; + last_fetch_at: string | null; + last_success_at: string | null; + last_status: number | null; + last_error: string | null; + item_count: number; + consecutive_failures: number; +}; + +export type CrawlerStats = { + sources: number; + active: number; + gaveUp: number; + dueNow: number; + neverFetched: number; + erroring: number; + items: number; + newItems24h: number; + lastSuccessAt: string | null; + /** + * The daemon looks stopped. + * + * Sources are due and nothing has succeeded in a while. Stated as a + * suspicion rather than a fact because this cannot distinguish "the worker + * is down" from "the directory is down" — and the operator's next step + * differs, so claiming to know which would send them the wrong way. + */ + stalled: boolean; + generatedAt: string; +}; + +/** Nothing succeeded in this long, with work waiting, reads as stalled. */ +const STALL_AFTER_MS = 2 * 60 * 60 * 1000; // 2h — four missed 30-min ticks + +export async function loadCrawlerStats( + supabase: SupabaseClient, +): Promise<{ stats: CrawlerStats; sources: FeedSourceRow[] }> { + const now = Date.now(); + const dueBefore = new Date(now - REFRESH_AFTER_MS).toISOString(); + const since24h = new Date(now - 24 * 60 * 60 * 1000).toISOString(); + + const [{ data: sources }, itemCount, new24h] = await Promise.all([ + supabase + .from("lx_feed_source") + .select( + "id, topic, url, status, last_fetch_at, last_success_at, last_status, last_error, item_count, consecutive_failures", + ) + .order("last_fetch_at", { ascending: true, nullsFirst: true }) + .limit(500), + supabase + .from("lx_feed_item") + .select("id", { count: "exact", head: true }) + .then((r) => r.count ?? 0), + supabase + .from("lx_feed_item") + .select("id", { count: "exact", head: true }) + .gte("first_seen_at", since24h) + .then((r) => r.count ?? 0), + ]); + + const rows = (sources ?? []) as FeedSourceRow[]; + const active = rows.filter((r) => r.status === "active"); + const dueNow = active.filter( + (r) => !r.last_fetch_at || r.last_fetch_at < dueBefore, + ).length; + + const lastSuccessAt = rows + .map((r) => r.last_success_at) + .filter((v): v is string => !!v) + .sort() + .at(-1) ?? null; + + const stalled = + dueNow > 0 && + (!lastSuccessAt || now - Date.parse(lastSuccessAt) > STALL_AFTER_MS); + + return { + stats: { + sources: rows.length, + active: active.length, + gaveUp: rows.filter((r) => r.status === "given_up").length, + dueNow, + neverFetched: rows.filter((r) => !r.last_fetch_at).length, + erroring: rows.filter((r) => r.consecutive_failures > 0).length, + items: itemCount, + newItems24h: new24h, + lastSuccessAt, + stalled, + generatedAt: new Date(now).toISOString(), + }, + sources: rows, + }; +} + +/** "15m ago", "3h ago", "never" — the only formatting this page needs. */ +export function ago(iso: string | null, now = Date.now()): string { + if (!iso) return "never"; + const ms = now - Date.parse(iso); + if (!Number.isFinite(ms) || ms < 0) return "just now"; + const mins = Math.floor(ms / 60000); + if (mins < 1) return "just now"; + if (mins < 60) return `${mins}m ago`; + const hours = Math.floor(mins / 60); + if (hours < 24) return `${hours}h ago`; + return `${Math.floor(hours / 24)}d ago`; +} diff --git a/lib/lx/feedTopics.ts b/lib/lx/feedTopics.ts index 80215bb9..6d67d574 100644 --- a/lib/lx/feedTopics.ts +++ b/lib/lx/feedTopics.ts @@ -62,7 +62,26 @@ export function topicSlug(keyword: string): string { * @returns post titles, channel title and sponsored items removed */ export function itemTitles(xml: string): string[] { - const out: string[] = []; + return itemEntries(xml).map((e) => e.title); +} + +/** A usable post from a topic feed. */ +export type FeedEntry = { title: string; link: string | null }; + +/** + * Posts in a topic feed, with their links. + * + * The link is what a *citation* needs; `itemTitles` only ever needed the + * subject, and is now a projection of this. Keeping one parser means the + * sponsored-item exclusion below cannot be enforced on one path and forgotten + * on the other — and the path that carries links out to a published page is + * precisely the one where forgetting it would be worst. + * + * @param xml an RSS document + * @returns entries, channel-level and sponsored items removed + */ +export function itemEntries(xml: string): FeedEntry[] { + const out: FeedEntry[] = []; // Item blocks only — this is what keeps the channel's own (the name // of the topic) out of the candidate list. @@ -84,12 +103,75 @@ export function itemTitles(xml: string): string[] { if (/\(sponsored\)\s*$/i.test(title)) continue; if (title.length < MIN_TITLE_LEN || title.length > MAX_TITLE_LEN) continue; - out.push(title); + + const href = decodeXml(item.match(/<link>([\s\S]*?)<\/link>/i)?.[1] ?? "").trim(); + // Only absolute http(s) links survive. A feed carrying a relative link, a + // javascript: URL or a bare guid must not be able to put either into an + // anchor on a customer's published page. + const link = /^https?:\/\/\S+$/i.test(href) ? href : null; + + out.push({ title, link }); } return out; } +/** + * Real posts from the directory on the subjects given, with links. + * + * Unlike `subjectFromTopicFeeds`, which wants one subject to write about, this + * wants several posts to cite — so it reads across every requested topic + * rather than stopping at the first that answers, and returns only entries + * that carry a usable link. + * + * @param keywords subject words, tried in random order + * @param limit most entries to return + * @param fetchImpl injected by the tests + */ +export async function postsFromTopicFeeds( + keywords: string[], + limit = 3, + fetchImpl: typeof fetch = fetch, +): Promise<Array<{ title: string; link: string; topic: string }>> { + const slugs = shuffle( + Array.from(new Set((keywords ?? []).map(topicSlug).filter(Boolean))), + ).slice(0, MAX_FEEDS); + + const out: Array<{ title: string; link: string; topic: string }> = []; + const seen = new Set<string>(); + + for (const slug of slugs) { + if (out.length >= limit) break; + let xml: string; + try { + const res = await fetchImpl( + `${RSSAMPLIFIER}/topics/${encodeURIComponent(slug)}.rss`, + { + signal: AbortSignal.timeout(TIMEOUT_MS), + headers: { accept: "application/rss+xml, application/xml;q=0.9" }, + }, + ); + if (!res.ok) continue; + xml = await res.text(); + } catch { + continue; + } + + // One entry per topic before taking a second from any of them, so a block + // of three citations shows three subjects rather than three posts from + // whichever feed happened to be longest. + const entries = shuffle(itemEntries(xml).filter((e) => e.link)); + for (const entry of entries) { + if (!entry.link || seen.has(entry.link)) continue; + seen.add(entry.link); + out.push({ title: entry.title, link: entry.link, topic: slug.replace(/-/g, " ") }); + break; + } + } + + return out.slice(0, limit); +} + /** * Undo the escaping a feed document applies, and nothing else. * diff --git a/lib/lx/keywordsResearch.ts b/lib/lx/keywordsResearch.ts index 0de6e48f..7f5bbc80 100644 --- a/lib/lx/keywordsResearch.ts +++ b/lib/lx/keywordsResearch.ts @@ -1,14 +1,20 @@ // Keyword research pipeline (PRD §6.1, §15). // -// Inputs: siteId — site must have niche + (optionally) target_audiences set. -// Outputs: up to 30 new rows in lx_keyword (status='queued'), scheduled -// across the next ~6 weeks honoring publish_days + daily_article_count. +// Inputs: siteId — the site's `master_keywords` (the durable subject list) and +// `modifiers` (what the site actually does) are the two lists this +// works from. `niche` backfills the modifiers when that column is empty. +// Outputs: up to 30 new rows in lx_keyword (status='queued'), allocated evenly +// across every master subject and scheduled across the next ~6 weeks +// honoring publish_days + daily_article_count. // -// Caching: lx_keyword_metrics rows live 60 days. Before paying DataForSEO -// for a seed, we check whether we already have a recent expansion cached -// (heuristic: same seed string, region='us'). For v1 we always re-fetch on -// manual trigger; cache is queried per-keyword to short-circuit the -// volume backfill step. +// **Every subject is researched, and none is researched alone.** Both halves +// matter and both were broken. See lib/lx/topicPlan.ts for the failure this +// was written against — nineteen articles about peptide vendors on a payments +// blog — and why the fix is a cross product rather than a bigger slice. +// +// Cost note: the cross queries all go into a *single* keywordIdeas call, which +// accepts 200 seeds. Covering ten subjects properly is therefore cheaper than +// the three separate calls this replaced, not more expensive. // // Spend ledger: every DataForSEO call writes a row to lx_dataforseo_usage. @@ -25,6 +31,18 @@ import { } from "./buyerJourneyKeywords"; import { DataForSeoClient, filterOutliers, type DfsKeywordRow } from "./dataforseo"; import { nextPublishAt } from "./schedule"; +import { + allocate, + anchorTokens, + crossQueries, + dropDuplicates, + isOnNiche, + resolveMasters, + resolveModifiers, + signature, + stem, + tokens, +} from "./topicPlan"; type SiteRow = { id: string; @@ -33,6 +51,8 @@ type SiteRow = { target_audiences: string[]; description: string | null; seed_keywords: string[]; + master_keywords: string[]; + modifiers: string[]; keywords: string[]; competitors: string[]; tone: string | null; @@ -45,24 +65,20 @@ const TARGET_KEYWORDS = 30; const MIN_VOLUME = 50; const MIN_BUYER_JOURNEY_VOLUME = 10; const MIN_WORDS = 2; -const PER_SEED_LIMIT = 200; +const IDEAS_LIMIT = 600; const MAX_BUYER_JOURNEY_VOLUME_LOOKUP = 160; -const SEED_TOKEN_STOPLIST = new Set([ - "the","and","for","with","you","your","that","this","from","into","over", - "but","not","are","was","were","has","had","have","its","off","out","all", - "any","new","get","how","why","what","who","best","top", -]); - -function buildSeeds(site: SiteRow): string[] { - const seeds: string[] = []; - for (const s of site.seed_keywords ?? []) seeds.push(s.trim()); - if (site.niche) seeds.push(site.niche.trim()); - for (const a of (site.target_audiences ?? []).slice(0, 3)) { - if (site.niche && a) seeds.push(`${site.niche} for ${a}`); - else if (a) seeds.push(a); - } - return Array.from(new Set(seeds.filter((s) => s.length > 0))).slice(0, 5); -} + +/** + * Cross depth per subject. + * + * Three narrowing terms per subject: enough that a subject yields more than a + * single phrasing, few enough that ten subjects still fit one API call with + * room for the seed list to grow. + */ +const CROSS_PER_MASTER = 3; + +/** A candidate with the subject it belongs to. */ +type Candidate = { row: DfsKeywordRow; master: string }; function parseStoredKeyword(row: string): DfsKeywordRow | null { const idx = row.indexOf(","); @@ -82,20 +98,43 @@ function parseStoredKeyword(row: string): DfsKeywordRow | null { }; } -function seedTokens(seed: string): string[] { - return seed - .toLowerCase() - .split(/[\s-]+/) - .map((t) => t.replace(/[^a-z0-9]/g, "")) - .filter((t) => t.length >= 4 && !SEED_TOKEN_STOPLIST.has(t)); -} - type KeywordBoost = { priority: number; intent: BuyerJourneyKeywordIntent; clusterType: BuyerJourneyClusterType; }; +/** + * Which subject a candidate belongs to. + * + * Longest match wins, so a site covering both "crypto" and "cryptocurrency" + * attributes "cryptocurrency merchant account" to the more specific of the + * two rather than to whichever happens to sort first. Returns null when no + * subject claims it — the caller drops those, which is the same verdict the + * niche gate would reach a moment later. + */ +function attribute(keyword: string, masters: string[]): string | null { + let best: string | null = null; + let bestLen = 0; + // Stemmed on both sides, so attribution and the gate agree about what + // "promo codes" and "promo code" are. They disagreed before, and a keyword + // attributed to a subject the gate then could not match was dropped for a + // reason nobody would have guessed from the strings. + const candidate = new Set(tokens(keyword).map(stem)); + for (const master of masters) { + const masterTokens = tokens(master).map(stem); + if (masterTokens.length === 0) continue; + const hit = masterTokens.filter((t) => candidate.has(t)); + if (hit.length === 0) continue; + const len = hit.join("").length; + if (len > bestLen) { + bestLen = len; + best = master; + } + } + return best; +} + function rankKeywords( rows: DfsKeywordRow[], boosts = new Map<string, KeywordBoost>(), @@ -136,26 +175,22 @@ function rankKeywords( return scored.map((s) => s.row); } -function primarySeedQuery(site: SiteRow, seeds: string[]): string { - return ( - seeds[0] ?? - site.niche ?? - site.keywords?.[0] ?? - site.description ?? - site.domain ?? - "site keyword research" - ).trim(); -} - -function buildBuyerJourneyInput(site: SiteRow, seeds: string[]): BuyerJourneyKeywordInput { +function buildBuyerJourneyInput( + site: SiteRow, + masters: string[], + modifiers: string[], +): BuyerJourneyKeywordInput { const brand = site.domain?.replace(/^www\./, "") || site.niche || "the website"; const offer = [site.niche, site.description] .filter((s): s is string => !!s && s.trim().length > 0) .join(" — ") .slice(0, 900); return { - seedQuery: primarySeedQuery(site, seeds), - additionalSeeds: seeds.slice(1, 8), + // Every subject, not just the first one. The old code passed + // `seeds[0]` here and `seeds.slice(1, 8)` from an already-truncated + // list, which is how one subject came to own the entire model run. + seedQuery: masters.join(", "), + additionalSeeds: modifiers.slice(0, 8), offer: offer || brand, audience: site.target_audiences?.join(", ") || "the site's target customers", brand, @@ -182,21 +217,6 @@ function rowFromCandidate( }; } -function dedupeBySite( - candidates: DfsKeywordRow[], - existing: Set<string>, -): DfsKeywordRow[] { - const out: DfsKeywordRow[] = []; - const seen = new Set<string>(); - for (const r of candidates) { - const k = r.keyword.toLowerCase(); - if (existing.has(k) || seen.has(k)) continue; - seen.add(k); - out.push(r); - } - return out; -} - function scheduleKeywords( count: number, publishDays: number[], @@ -218,10 +238,39 @@ function scheduleKeywords( return dates; } +/** + * Interleave per-subject picks so the schedule alternates topics. + * + * Allocation decides *how many* rows each subject gets; this decides the order + * they are scheduled in, and the two are separate concerns. A fair allocation + * emitted subject-by-subject would still publish six consecutive peptide posts + * and then six consecutive casino ones — fair over a quarter, and visibly + * spammy over a fortnight. Round-robining the emission is what makes the fix + * legible to a reader of the blog rather than only to a reader of the database. + */ +function interleave(byMaster: Map<string, Candidate[]>): Candidate[] { + const queues = Array.from(byMaster.values()).map((c) => [...c]); + const out: Candidate[] = []; + let live = true; + while (live) { + live = false; + for (const queue of queues) { + const next = queue.shift(); + if (next) { + out.push(next); + live = true; + } + } + } + return out; +} + export type KeywordResearchResult = { ok: boolean; inserted: number; apiCost: number; + /** Rows allocated per subject — surfaced so a skewed queue is visible. */ + perMaster?: Record<string, number>; error?: string; }; @@ -240,7 +289,7 @@ export async function researchKeywords( const { data: site } = await supabase .from("lx_site") .select( - "id, domain, niche, target_audiences, description, seed_keywords, keywords, competitors, tone, publish_days, publish_hour, daily_article_count", + "id, domain, niche, target_audiences, description, seed_keywords, master_keywords, modifiers, keywords, competitors, tone, publish_days, publish_hour, daily_article_count", ) .eq("id", siteId) .maybeSingle<SiteRow>(); @@ -248,153 +297,258 @@ export async function researchKeywords( return { ok: false, inserted: 0, apiCost: 0, error: "site not found" }; } - const savedKeywords = (site.keywords ?? []) - .map(parseStoredKeyword) - .filter((r): r is DfsKeywordRow => !!r); - const seeds = buildSeeds(site); - if (savedKeywords.length === 0 && seeds.length === 0) { + const masters = resolveMasters(site); + // Masters are passed through so the derived terms can subtract them: an + // anchor word that is also a subject word lets a candidate satisfy both + // halves of the gate with one token. See anchorTokens. + const modifiers = resolveModifiers(site, masters); + const anchors = anchorTokens(site, masters); + + if (masters.length === 0) { return { ok: false, inserted: 0, apiCost: 0, - error: "add saved keywords or seed keywords first", + error: "add master keywords (the subjects this blog covers) first", + }; + } + // Refusing here rather than falling back to an unanchored expansion is the + // point. An anchorless run is exactly the run that produced the vendor + // articles, so it must be an error the operator sees and fixes, not a + // degraded mode that quietly publishes. + if (anchors.size === 0) { + return { + ok: false, + inserted: 0, + apiCost: 0, + error: + "set a niche or modifiers first — keywords are only researched crossed with what this site does, never on their own", }; } - // Skip keywords this site already has in active/history states. Failed - // rows are intentionally ignored here: an upstream outage should not - // permanently poison a topic and prevent the top-up sweep from - // refilling the queue. + // Existing rows serve two purposes: the duplicate fingerprints, and the + // per-subject coverage the allocator balances against. Failed rows are + // intentionally excluded from the duplicate set — an upstream outage should + // not permanently poison a topic — but they still count toward coverage, so + // a subject that keeps failing does not monopolise every top-up. const { data: existingRows } = await supabase .from("lx_keyword") - .select("keyword, status") + .select("keyword, status, master_keyword") .eq("site_id", site.id); - const existingSet = new Set( - (existingRows ?? []) - .filter((r: { status: string }) => r.status !== "failed") - .map((r: { keyword: string }) => r.keyword.toLowerCase()), - ); - const savedChosen = dedupeBySite(savedKeywords, existingSet).slice(0, TARGET_KEYWORDS); + const publishedSignatures = new Set<string>(); + const coverage = new Map<string, number>(); + for (const row of (existingRows ?? []) as Array<{ + keyword: string; + status: string; + master_keyword: string | null; + }>) { + if (row.status !== "failed") { + const sig = signature(row.keyword); + if (sig) publishedSignatures.add(sig); + } + // Attribute legacy rows (written before provenance existed) so the + // allocator sees the real history rather than treating a blog with + // twenty-three peptide posts as having no coverage at all. Without this + // the balancing would take a full cycle to notice the existing skew. + const master = row.master_keyword ?? attribute(row.keyword, masters); + if (master) { + const key = master.toLowerCase(); + coverage.set(key, (coverage.get(key) ?? 0) + 1); + } + } - // If the saved long-tail list does not fill the target queue, top it - // up using the same DataForSEO Labs endpoint + relevance gate used by - // the settings page's "Refetch keywords" flow. - const allRows: DfsKeywordRow[] = []; + const allocation = allocate(masters, coverage, TARGET_KEYWORDS); + + // ------------------------------------------------------------------ + // Candidate sources. All three are gated identically; they differ only + // in how they are obtained and how likely they are to be unavailable. + // ------------------------------------------------------------------ + const candidates: Candidate[] = []; const buyerJourneyBoosts = new Map<string, KeywordBoost>(); let totalCost = 0; - const seedErrors: string[] = []; - if (savedChosen.length < TARGET_KEYWORDS && seeds.length > 0) { - if (deps.openai || deps.anthropic) { - try { - const buyerJourney = await generateBuyerJourneyKeywordOpportunities( - buildBuyerJourneyInput(site, seeds), - { - openai: deps.openai, - anthropic: deps.anthropic, - backendAiProvider: deps.backendAiProvider, - }, - ); - const candidates = flattenBuyerJourneyKeywords( - buyerJourney.output, - MAX_BUYER_JOURNEY_VOLUME_LOOKUP, - ); - const volumeKeywords = candidates.map((c) => c.keyword); - const volume = volumeKeywords.length > 0 - ? await dfs.searchVolume(volumeKeywords) - : { rows: [], cost: 0, taskId: null }; - totalCost += volume.cost; - if (volume.cost > 0 || volume.taskId) { - await supabase.from("lx_dataforseo_usage").insert({ - task_id: volume.taskId, - endpoint: "search_volume/live", - cost: volume.cost, - site_id: site.id, - }); - } - - const metricsByKeyword = new Map( - volume.rows.map((r) => [r.keyword.toLowerCase(), r]), - ); - for (const candidate of candidates) { - const metrics = metricsByKeyword.get(candidate.keyword.toLowerCase()); - const volumeValue = metrics?.search_volume ?? 0; - if ( - volumeValue >= MIN_BUYER_JOURNEY_VOLUME || - candidate.priority >= 4 - ) { - allRows.push(rowFromCandidate(candidate, metrics)); - buyerJourneyBoosts.set(candidate.keyword.toLowerCase(), { - priority: candidate.priority, - intent: candidate.intent, - clusterType: candidate.clusterType, - }); - } - } - } catch (err) { - seedErrors.push( - `buyer-journey model: ${err instanceof Error ? err.message : String(err)}`, - ); + const sourceErrors: string[] = []; + + // Source 0 — the floor. Subject × modifier, built locally from two columns + // the operator controls. No network, no failure mode, always on-niche by + // construction. Everything below is an improvement on this, never a + // prerequisite for it. + const crosses = crossQueries(masters, modifiers, CROSS_PER_MASTER); + for (const { master, query } of crosses) { + candidates.push({ + row: { + keyword: query, + search_volume: null, + competition: null, + competition_index: null, + cpc: null, + low_top_of_page_bid: null, + high_top_of_page_bid: null, + monthly_searches: null, + }, + master, + }); + } + + // Source 1 — hand-saved long-tail from the settings page. + for (const parsed of (site.keywords ?? []).map(parseStoredKeyword)) { + if (!parsed) continue; + const master = attribute(parsed.keyword, masters); + if (master) candidates.push({ row: parsed, master }); + } + + // Source 2 — the buyer-journey model, now seeded with every subject. + if (deps.openai || deps.anthropic) { + try { + const buyerJourney = await generateBuyerJourneyKeywordOpportunities( + buildBuyerJourneyInput(site, masters, modifiers), + { + openai: deps.openai, + anthropic: deps.anthropic, + backendAiProvider: deps.backendAiProvider, + }, + ); + const flattened = flattenBuyerJourneyKeywords( + buyerJourney.output, + MAX_BUYER_JOURNEY_VOLUME_LOOKUP, + ); + const volumeKeywords = flattened.map((c) => c.keyword); + const volume = volumeKeywords.length > 0 + ? await dfs.searchVolume(volumeKeywords) + : { rows: [], cost: 0, taskId: null }; + totalCost += volume.cost; + if (volume.cost > 0 || volume.taskId) { + await supabase.from("lx_dataforseo_usage").insert({ + task_id: volume.taskId, + endpoint: "search_volume/live", + cost: volume.cost, + site_id: site.id, + }); + } + + const metricsByKeyword = new Map( + volume.rows.map((r) => [r.keyword.toLowerCase(), r]), + ); + for (const candidate of flattened) { + const metrics = metricsByKeyword.get(candidate.keyword.toLowerCase()); + const volumeValue = metrics?.search_volume ?? 0; + if (volumeValue < MIN_BUYER_JOURNEY_VOLUME && candidate.priority < 4) continue; + const master = attribute(candidate.keyword, masters); + if (!master) continue; + candidates.push({ row: rowFromCandidate(candidate, metrics), master }); + buyerJourneyBoosts.set(candidate.keyword.toLowerCase(), { + priority: candidate.priority, + intent: candidate.intent, + clusterType: candidate.clusterType, + }); } + } catch (err) { + sourceErrors.push( + `buyer-journey model: ${err instanceof Error ? err.message : String(err)}`, + ); } + } - for (const seed of seeds.slice(0, 3)) { - try { - const result = await dfs.keywordIdeas([seed], { - limit: PER_SEED_LIMIT, + // Source 3 — DataForSEO expansion of the CROSSED phrases. + // + // One call carrying every cross, because keywordIdeas accepts 200 seeds. + // The bare subject is never sent: "peptide" on its own is what returned + // "skye peptides" and "pure peptide labs", and no downstream filter can + // reliably tell those from a keyword worth writing about. + if (crosses.length > 0) { + try { + const result = await dfs.keywordIdeas( + crosses.map((c) => c.query), + { + limit: IDEAS_LIMIT, minVolume: MIN_VOLUME, minWords: MIN_WORDS, closelyVariants: false, - }); - totalCost += result.cost; - await supabase.from("lx_dataforseo_usage").insert({ - task_id: result.taskId, - endpoint: "keyword_ideas/live", - cost: result.cost, - site_id: site.id, - }); - - const tokens = seedTokens(seed); - const relevant = tokens.length === 0 - ? result.rows - : result.rows.filter((r) => { - const kw = r.keyword.toLowerCase(); - return tokens.some((t) => kw.includes(t)); - }); - allRows.push(...relevant); - } catch (err) { - seedErrors.push( - `"${seed}": ${err instanceof Error ? err.message : String(err)}`, - ); + }, + ); + totalCost += result.cost; + await supabase.from("lx_dataforseo_usage").insert({ + task_id: result.taskId, + endpoint: "keyword_ideas/live", + cost: result.cost, + site_id: site.id, + }); + for (const row of result.rows) { + const master = attribute(row.keyword, masters); + if (master) candidates.push({ row, master }); } + } catch (err) { + sourceErrors.push( + `keyword ideas: ${err instanceof Error ? err.message : String(err)}`, + ); } } - const filtered = filterOutliers(allRows).filter( - (r) => { - const boost = buyerJourneyBoosts.get(r.keyword.toLowerCase()); - if (boost && boost.priority >= 4) return true; - return (r.search_volume ?? 0) >= MIN_VOLUME; - }, - ); - const ranked = rankKeywords(filtered, buyerJourneyBoosts); - const existingWithSaved = new Set(existingSet); - for (const r of savedChosen) existingWithSaved.add(r.keyword.toLowerCase()); - const researchedChosen = dedupeBySite(ranked, existingWithSaved).slice( - 0, - TARGET_KEYWORDS - savedChosen.length, + // ------------------------------------------------------------------ + // Gate, rank, allocate. + // ------------------------------------------------------------------ + const onNiche = candidates.filter((c) => isOnNiche(c.row.keyword, c.master, anchors)); + + // Volume filtering applies only to what came back from an API with a volume + // attached. The locally-built crosses have no volume by construction and + // must not be discarded for it — they are the floor that keeps the queue + // from emptying when everything upstream is unavailable. + const withVolume = onNiche.filter((c) => c.row.search_volume !== null); + const withoutVolume = onNiche.filter((c) => c.row.search_volume === null); + const volumeKept = filterOutliers(withVolume.map((c) => c.row)).filter((r) => { + const boost = buyerJourneyBoosts.get(r.keyword.toLowerCase()); + if (boost && boost.priority >= 4) return true; + return (r.search_volume ?? 0) >= MIN_VOLUME; + }); + const volumeKeptKeys = new Set(volumeKept.map((r) => r.keyword.toLowerCase())); + + const ranked = rankKeywords(volumeKept, buyerJourneyBoosts); + const rankIndex = new Map(ranked.map((r, i) => [r.keyword.toLowerCase(), i])); + + const survivors = [ + ...withVolume + .filter((c) => volumeKeptKeys.has(c.row.keyword.toLowerCase())) + .sort( + (a, b) => + (rankIndex.get(a.row.keyword.toLowerCase()) ?? Infinity) - + (rankIndex.get(b.row.keyword.toLowerCase()) ?? Infinity), + ), + // Crosses last within each subject: a real long-tail phrase with measured + // demand is a better article than a two-word construction, but the + // construction is a better article than nothing. + ...withoutVolume, + ]; + + const deduped = dropDuplicates( + survivors.map((c) => ({ ...c, keyword: c.row.keyword })), + publishedSignatures, ); - const chosen = [...savedChosen, ...researchedChosen].slice(0, TARGET_KEYWORDS); + + // Take each subject's allocated share, then interleave so the published + // sequence alternates subjects rather than running one to exhaustion. + const byMaster = new Map<string, Candidate[]>(); + for (const master of masters) byMaster.set(master, []); + for (const candidate of deduped) { + const bucket = byMaster.get(candidate.master); + if (!bucket) continue; + if (bucket.length >= (allocation.get(candidate.master) ?? 0)) continue; + bucket.push({ row: candidate.row, master: candidate.master }); + } + + // Subjects that could not fill their share hand it back, so a subject with + // no available candidates costs the run coverage rather than volume. + const chosen = interleave(byMaster).slice(0, TARGET_KEYWORDS); + if (chosen.length === 0) { - const details = seedErrors.length > 0 - ? ` Seed errors: ${seedErrors.join("; ")}` + const details = sourceErrors.length > 0 + ? ` Source errors: ${sourceErrors.join("; ")}` : ""; return { ok: false, inserted: 0, apiCost: totalCost, error: - "No new keyword candidates found. Saved keywords may already be published or queued; add new seed keywords/settings and try again." + + "No new keyword candidates found. Every candidate was already published or off-niche; add master keywords or modifiers and try again." + details, }; } @@ -405,7 +559,6 @@ export async function researchKeywords( // without first reshaping it. We re-introduce it when keyword overlap // across customers becomes measurable. - // Schedule them across publish_days. const slots = scheduleKeywords( chosen.length, site.publish_days, @@ -413,15 +566,16 @@ export async function researchKeywords( site.daily_article_count, ); - const insertRows = chosen.map((r, i) => ({ + const insertRows = chosen.map((c, i) => ({ site_id: site.id, - keyword: r.keyword, + keyword: c.row.keyword, + master_keyword: c.master, scheduled_for: slots[i]?.toISOString().slice(0, 10) ?? new Date(Date.now() + (i + 1) * 86400000).toISOString().slice(0, 10), status: "queued", source: "auto", - search_volume: r.search_volume, - cpc_usd: r.cpc, + search_volume: c.row.search_volume, + cpc_usd: c.row.cpc, })); const { error: insErr } = await supabase.from("lx_keyword").insert(insertRows); @@ -434,5 +588,10 @@ export async function researchKeywords( }; } - return { ok: true, inserted: insertRows.length, apiCost: totalCost }; + const perMaster: Record<string, number> = {}; + for (const row of insertRows) { + perMaster[row.master_keyword] = (perMaster[row.master_keyword] ?? 0) + 1; + } + + return { ok: true, inserted: insertRows.length, apiCost: totalCost, perMaster }; } diff --git a/lib/lx/networkBlock.ts b/lib/lx/networkBlock.ts new file mode 100644 index 00000000..13928d72 --- /dev/null +++ b/lib/lx/networkBlock.ts @@ -0,0 +1,232 @@ +// What a published article carries besides the article: an ad unit, and links +// out to the rest of the network. +// +// Both are governed by one column — `lx_site.ads_enabled`, default true — and +// that is deliberate. Splitting "show ads" from "join the link network" into +// two switches produces four states, two of which are incoherent (take +// backlinks from partners, refuse to give them) and all four of which need a +// human to reason about. One switch, on by default, is the whole opt-in. +// +// Three things this is careful about. +// +// **The slot is provisioned, not requested.** A blog with no ad slot used to +// mean a support conversation. Slots are created here, active, on first +// delivery — `ad_slots.status` defaults to 'inactive' and `serveAd()` returns +// null before the house-ad fallback for a non-active slot, so a slot created +// at the default would render an empty div on every article forever and look +// exactly like a broken embed. Provisioning it inactive would be worse than +// not provisioning it at all. +// +// **Everything interpolated is escaped.** The titles and links in the +// insertion block come from other people's RSS feeds. They are written into +// HTML that lands on a customer's domain, which makes this an injection sink +// with a hostile upstream, and the fact that the immediate source is our own +// directory changes nothing about that — the directory is a crawl of the open +// web. +// +// **A failure here costs the block, never the article.** Every lookup is +// wrapped and degrades to "no block". A post that publishes without an ad unit +// has cost us an impression; a post that fails to publish because the ad +// lookup threw has cost the customer the thing they are paying for. + +import type { SupabaseClient } from "@supabase/supabase-js"; +import { cachedFeedPosts } from "./feedCrawl"; + +/** Format asked of the slot. 728x90 is the in-article leaderboard. */ +const AD_FORMAT = "banner_728x90"; + +/** Most partner links in one insertion. */ +const MAX_PARTNER_LINKS = 3; + +/** Most directory posts in one insertion. */ +const MAX_FEED_LINKS = 3; + +/** + * Escape text for HTML interpolation. + * + * Ampersand first, or the escapes introduced by the later replacements get + * double-escaped. Quotes are included because these values are also written + * into attributes. + */ +export function escapeHtml(value: string): string { + return String(value ?? "") + .replace(/&/g, "&") + .replace(/</g, "<") + .replace(/>/g, ">") + .replace(/"/g, """) + .replace(/'/g, "'"); +} + +/** + * Is this a link we are willing to put on a customer's page? + * + * An allowlist of two schemes rather than a denylist of the dangerous ones: + * `javascript:`, `data:` and `vbscript:` are the ones anybody thinks to block, + * and the list of what else a browser will execute is not one to maintain by + * hand. + */ +export function isSafeHref(href: string): boolean { + try { + const url = new URL(href); + return url.protocol === "https:" || url.protocol === "http:"; + } catch { + return false; + } +} + +export type NetworkLink = { title: string; url: string; source: "partner" | "directory" }; + +/** + * The slot this project's articles should fill. + * + * Reuses an existing active slot before creating one, so re-delivering an + * article — or publishing the second post on a blog — does not mint a second + * slot and split the site's reporting across two rows. + * + * @returns a slot id, or null when one could not be resolved or created + */ +export async function resolveAdSlot( + supabase: SupabaseClient<any>, + projectId: string, + ownerId: string, + niche: string | null, +): Promise<string | null> { + try { + const { data: existing } = await supabase + .from("ad_slots") + .select("id") + .eq("project_id", projectId) + .eq("status", "active") + .limit(1) + .maybeSingle<{ id: string }>(); + if (existing?.id) return existing.id; + + const { data: created } = await supabase + .from("ad_slots") + .insert({ + project_id: projectId, + owner_id: ownerId, + placement: "inline", + formats: [AD_FORMAT, "banner_300x250", "text_link"], + niche, + // Explicit, against the column default. See the note at the top of + // this file: an inactive slot is indistinguishable from a broken one. + status: "active", + }) + .select("id") + .maybeSingle<{ id: string }>(); + return created?.id ?? null; + } catch { + return null; + } +} + +/** + * The ad unit markup. + * + * `data-cp-ad` carries no `data-slot` in the server-rendered case elsewhere in + * the codebase to avoid a fill race; here there is no client to race with — + * this HTML is delivered to a third-party blog and rendered as-is — so the + * slot travels on the element and ad.js's own DOMContentLoaded pass fills it. + */ +export function adUnitHtml(slotId: string, origin: string): string { + const slot = escapeHtml(slotId); + const src = `${origin.replace(/\/$/, "")}/ad.js`; + return [ + `<div data-cp-ad data-slot="${slot}" data-format="${AD_FORMAT}"></div>`, + `<script async src="${escapeHtml(src)}"></script>`, + ].join("\n"); +} + +/** + * The "elsewhere in the network" block. + * + * Partner articles first, directory posts after: a partner opted into the + * exchange and gets the more valuable position, while the directory posts are + * what keep the block from being visibly the same three domains on every + * article — which is the shape that gets a link network discounted. + * + * Directory links are `rel="nofollow ugc"`. They are not exchange partners and + * have not agreed to anything; passing them ranking signal would be us + * spending someone else's reputation. Partner links are followed, because that + * reciprocity is the entire point of the exchange and is recorded on both + * sides in lx_backlink. + */ +export function networkLinksHtml(links: NetworkLink[]): string { + const safe = links.filter((l) => isSafeHref(l.url)); + if (safe.length === 0) return ""; + + const items = safe + .map((link) => { + const rel = link.source === "directory" + ? ' rel="nofollow ugc noopener"' + : ' rel="noopener"'; + return ` <li><a href="${escapeHtml(link.url)}"${rel}>${escapeHtml(link.title)}</a></li>`; + }) + .join("\n"); + + return [ + `<aside class="cp-network-links" data-cp-network>`, + ` <h2>Elsewhere on this topic</h2>`, + ` <ul>`, + items, + ` </ul>`, + `</aside>`, + ].join("\n"); +} + +/** + * Everything appended to an article, for a site that is in the network. + * + * @param supabase service client — this runs from the delivery path, no session + * @param site the blog that will HOST the post (the guest-post target, when + * the article is one), because it is that site's readers who see the + * block and that site's owner who opted in + * @param topics subjects to pull directory posts for + * @param partnerLinks already-ranked exchange candidates from the caller + * @returns HTML to append, or "" when the site is opted out or nothing resolved + */ +export async function buildNetworkBlock( + supabase: SupabaseClient<any>, + site: { + id: string; + project_id: string; + user_id: string; + niche: string | null; + ads_enabled?: boolean | null; + }, + topics: string[], + partnerLinks: NetworkLink[], + origin: string, +): Promise<string> { + // Absent column reads as opted in, matching the migration default, so this + // behaves the same before and after the schema lands. + if (site.ads_enabled === false) return ""; + + const parts: string[] = []; + + // Read from the crawl cache, never live. A publish must not wait on the + // directory's server; a topic the sweep has not reached yet costs this + // article its citation block and the next one gets it. + const feedLinks: NetworkLink[] = await cachedFeedPosts(supabase, topics, MAX_FEED_LINKS) + .then((posts) => + posts.map((p) => ({ title: p.title, url: p.link, source: "directory" as const })), + ) + .catch(() => []); + + const block = networkLinksHtml([ + ...partnerLinks.slice(0, MAX_PARTNER_LINKS), + ...feedLinks, + ]); + if (block) parts.push(block); + + const slotId = await resolveAdSlot( + supabase, + site.project_id, + site.user_id, + site.niche, + ); + if (slotId) parts.push(adUnitHtml(slotId, origin)); + + return parts.join("\n"); +} diff --git a/lib/lx/topicPlan.ts b/lib/lx/topicPlan.ts new file mode 100644 index 00000000..7514129c --- /dev/null +++ b/lib/lx/topicPlan.ts @@ -0,0 +1,436 @@ +// Deciding which subjects an autoblog writes about, and in what proportion. +// +// This module exists because of a specific failure, and the shape of it is a +// direct response to that failure. coinpayportal.com — a crypto payment +// processor — published nineteen consecutive articles about peptide vendors: +// "skye peptides", "pure peptide labs", "wolverine stack peptides". Those are +// competitor storefronts in an industry the site *serves*, not one it is in. +// +// Two independent defects produced it, and fixing either alone would have left +// the other running. +// +// 1. **Truncation.** The site had ten subjects. The pipeline sliced them to +// five, then to three, and handed subject #1 to the buyer-journey model as +// its entire query. Five subjects had never produced a keyword in the +// site's lifetime. The top-up sweep re-ran the same truncated set every +// time the queue drained, so the concentration compounded rather than +// averaged out. +// +// 2. **An unanchored relevance gate.** A candidate was kept if it contained +// the seed token. Expanding the bare word "peptide" against a keyword +// tool returns the peptide industry's own vocabulary, and every one of +// those passed a test that only ever asked "is this about peptides?" — +// never "is this about what we sell to them?". +// +// The answer to both is the same: never expand a subject on its own. A subject +// is only ever researched *crossed with a modifier* — the tail terms that +// describe what this site actually does ("merchant account", "payment +// gateway"). "peptide" is not a topic. "peptide merchant account" is. +// +// That cross is also the floor. Every other source here can return nothing — +// the keyword API can be down, the model can refuse, the gate can reject +// everything — and the cross product still yields on-niche subjects, because +// it is built from two lists the operator controls rather than fetched. A +// blog that cannot reach any upstream still publishes, and still publishes +// about itself. Going dark and going off-topic are both failures; this +// prefers a smaller, correct queue to a full, spammy one. + +/** Subjects past this point can't be given a meaningful share of a 30-row target. */ +export const MAX_MASTERS = 12; + +/** + * Words that carry no topic and would pass any gate built on them. + * + * Deliberately not a general English stoplist: this is scoped to the words + * that appear in a *niche description* and a keyword phrase without narrowing + * either. "best" and "top" are here because they are the two most common + * prefixes in keyword-tool output and matching on them would re-admit the + * whole vendor-listicle class this module exists to reject. + */ +const STOPWORDS = new Set([ + "the", "and", "for", "with", "you", "your", "that", "this", "from", "into", + "over", "but", "not", "are", "was", "were", "has", "had", "have", "its", + "off", "out", "all", "any", "new", "get", "how", "why", "what", "who", + "best", "top", "guide", "list", "using", "about", "when", "where", "which", +]); + +/** + * A token worth matching on. + * + * Four characters because three-letter fragments ("pay", "ads", "seo") match + * inside unrelated words often enough to be worse than useless in a gate whose + * entire job is rejecting near-misses. + */ +const MIN_TOKEN_LEN = 4; + +/** + * Split a phrase into matchable tokens. + * + * Punctuation is dropped rather than split on, so "high-risk" yields "high" + * and "risk" — both of which are real narrowing terms for the site that wrote + * that niche, and neither of which survives a naive whitespace split. + */ +export function tokens(phrase: string): string[] { + return (phrase ?? "") + .toLowerCase() + .split(/[^a-z0-9]+/i) + .map((t) => t.trim()) + .filter((t) => t.length >= MIN_TOKEN_LEN && !STOPWORDS.has(t)); +} + +/** + * Reduce a token to a form that survives pluralisation. + * + * A crude suffix strip rather than a real stemmer, because the only job is + * collapsing "payment"/"payments" and "transaction"/"transactions" so the + * duplicate check can see that "peptide payments" and "peptide payment" are + * the same article. A real stemmer would be a dependency and a behaviour + * change in the gate, for a class of match this never needs to make. + */ +export function stem(token: string): string { + if (token.length > 4 && token.endsWith("ies")) return `${token.slice(0, -3)}y`; + + if (token.length > 3 && token.endsWith("es")) { + // "-es" is two different plurals and stripping a fixed number of + // characters gets one of them wrong. Taking two always ("codes" → "cod") + // does not collide with the singular ("code" → "code"), so a subject of + // "promo codes" stopped matching the keyword "promo code" — the exact + // shape this function exists to collapse. Strip one by default and two + // only after a sibilant, which is where the extra "e" is really epenthetic: + // "boxes" → "box", "matches" → "match", but "codes" → "code". + const short = token.slice(0, -2); + return /(?:s|x|z|ch|sh)$/.test(short) ? short : token.slice(0, -1); + } + + if (token.length > 3 && token.endsWith("s")) return token.slice(0, -1); + return token; +} + +/** + * An order-independent fingerprint of what a keyword is about. + * + * Sorted, so "merchant account peptide" and "peptide merchant account" collide + * — they would produce the same article, and a blog publishing both is the + * duplicate-content problem this is here to prevent. + */ +export function signature(keyword: string): string { + return Array.from(new Set(tokens(keyword).map(stem))).sort().join(" "); +} + +type SiteTopicFields = { + master_keywords?: string[] | null; + seed_keywords?: string[] | null; + modifiers?: string[] | null; + niche?: string | null; +}; + +/** + * The durable subject list for a site. + * + * Falls back to `seed_keywords` for any site the backfill has not reached, so + * this is safe to deploy ahead of the migration rather than after it. Capped + * on read as well as by the database constraint: a list that arrives oversized + * from anywhere must still be allocated over, not rejected at publish time. + */ +export function resolveMasters(site: SiteTopicFields): string[] { + const raw = (site.master_keywords?.length ? site.master_keywords : site.seed_keywords) ?? []; + const seen = new Set<string>(); + const out: string[] = []; + for (const entry of raw) { + const trimmed = (entry ?? "").trim(); + if (!trimmed) continue; + const key = trimmed.toLowerCase(); + if (seen.has(key)) continue; + seen.add(key); + out.push(trimmed); + } + return out.slice(0, MAX_MASTERS); +} + +/** + * The tail terms that turn any subject into an article this blog would write. + * + * A last-resort vocabulary, used when a site has no modifiers column and its + * niche says nothing the subjects do not already say. That case is common and + * it is where the worst output came from: vu1nz.com covers "ci/cd security" + * and "supply chain security" under the niche "CI/CD and supply chain + * security", so mining the niche yields only words the subjects already + * contain — and a gate built from those admits "adt home security" and + * "brinks home security" on a supply-chain blog. + * + * These are commercial and comparative rather than topical on purpose. They + * are what distinguishes an article a B2B blog publishes ("screen sharing + * software for teams") from a search result about a physical object + * ("garage door opener remote"), and they generalise across every niche, + * which a topical list could not. + */ +// Every entry has to be a word that a *commercial software* search uses and an +// ordinary one does not. "teams" and "business" were in an earlier version of +// this list and both had to come out: "community emergency response team" is a +// real queued keyword on a SOC blog, and it passed the gate on the token +// "team". A generic English noun cannot carry the second half of a two-part +// test, however natural it reads in a keyword phrase. +export const DEFAULT_MODIFIERS = [ + "software", + "tools", + "platform", + "alternatives", + "comparison", + "pricing", + "integration", + "automation", + "checklist", + "best practices", +]; + +/** + * The tail terms that anchor a subject to this site's own business. + * + * Three sources, in descending order of how much the operator meant them. + * The middle one subtracts the subjects: a niche word that is also a subject + * word cannot narrow anything, and keeping it is what lets a candidate satisfy + * both halves of the gate with a single token. + * + * @param site the row + * @param masters from `resolveMasters` — subtracted from the derived terms + */ +export function resolveModifiers( + site: SiteTopicFields, + masters: string[] = [], +): string[] { + const explicit = (site.modifiers ?? []) + .map((m) => (m ?? "").trim()) + .filter((m) => m.length > 0); + if (explicit.length > 0) return explicit.slice(0, 20); + + const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem))); + const fromNiche = Array.from(new Set(tokens(site.niche ?? ""))).filter( + (t) => !masterTokens.has(stem(t)), + ); + if (fromNiche.length > 0) return fromNiche.slice(0, 8); + + return DEFAULT_MODIFIERS; +} + +/** + * Every token that means "this keyword is about our business, not just our + * subject". + * + * Master tokens are excluded, and that exclusion is the fix for the sharpest + * version of the original bug. On vu1nz.com the subject "devops security" and + * the niche both contain "security"; without the subtraction, "adt home + * security" matches the subject on `security` and then matches the anchor on + * the very same word, satisfying a two-part test with one token. The anchor + * has to be evidence the subject match did not already provide. + */ +export function anchorTokens( + site: SiteTopicFields, + masters: string[] = [], +): Set<string> { + const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem))); + const out = new Set<string>(); + + const add = (phrase: string) => { + for (const token of tokens(phrase)) { + const stemmed = stem(token); + if (!masterTokens.has(stemmed)) out.add(stemmed); + } + }; + + for (const modifier of resolveModifiers(site, masters)) add(modifier); + add(site.niche ?? ""); + + // The commercial vocabulary is always unioned in, not just used as a + // fallback. bl0ggers.com's niche ("human-in-the-loop AI publishing") yields + // exactly {human, loop} once its own subjects are removed — non-empty, so + // the fallback never fired, and a thin anchor set rejected "ai writing + // tools", which is precisely what that blog should write. + // + // Unioning is safe because these words are commercial-software words: they + // rescue the good keywords without admitting any of the junk. None of "adt + // home security", "samsung tv remote", "palantir technologies" or "bayesian + // optimization" contains one — and where a site's own subject already + // claims one ("developer tools" on logicsrc), the master-token subtraction + // above removes it, so "mac tools" stays rejected. + for (const modifier of DEFAULT_MODIFIERS) add(modifier); + + return out; +} + +/** + * Is this candidate about one of our subjects *and* about what we do? + * + * Both halves are required, on different words, and that conjunction is the + * whole fix. The old gate asked only the first question, which is why + * nineteen articles about other people's peptide shops passed it — and why + * a supply-chain security blog was queued to write about home alarm + * installers. + * + * @param keyword the candidate + * @param master the subject it was researched for + * @param anchors from `anchorTokens`, which has already removed subject words + */ +export function isOnNiche( + keyword: string, + master: string, + anchors: Set<string>, +): boolean { + const candidate = new Set(tokens(keyword).map(stem)); + if (candidate.size === 0) return false; + + const masterTokens = tokens(master).map(stem); + if (masterTokens.length === 0) return false; + + const hits = masterTokens.filter((t) => candidate.has(t)).length; + if (hits === 0) return false; + + // A COMPLETE match on a multi-word subject is its own evidence, and needs no + // anchor. + // + // This is what separates the two failure shapes. Every bad keyword found on + // live sites matched exactly one generic word out of a multi-word subject — + // "security" from "devops security", "remote" from "remote control", + // "response" from "incident response", "tools" from "developer tools". None + // matched a subject in full. Meanwhile "abercrombie promo code" matches + // "promo codes" completely and is precisely what a coupon blog should write, + // yet an anchor rule alone rejects it, because a coupon site's subject IS + // its topic and it has no narrowing term to offer. + // + // Single-word subjects are excluded from this, and that exclusion is the + // original bug: "peptide" is one token, so "skye peptides" would match it + // "completely". A one-word subject can only ever be a vertical the site + // serves or a word too broad to stand alone, so it always needs the anchor. + if (masterTokens.length > 1 && hits === masterTokens.length) return true; + + // Otherwise: a partial or single-word subject match, which has to be backed + // by a word the subject did not supply. + // + // An anchorless site cannot answer that, and answering "yes" by default + // would restore the old behaviour exactly. `resolveModifiers` always yields + // something, so an empty set here means a site with no subjects configured. + if (anchors.size === 0) return false; + + for (const token of candidate) { + if (anchors.has(token)) return true; + } + return false; +} + +/** + * The subjects to research, each already crossed with a narrowing term. + * + * Ordered subject-major — every subject's first cross comes before any + * subject's second — so that a caller which runs out of API budget partway + * through has still touched every subject rather than exhausting the first + * one. That ordering is the difference between a budget cut costing depth and + * costing coverage, and coverage is what was broken. + * + * A cross whose modifier tokens are already in the subject is skipped: + * "crypto" × "crypto payments" would otherwise research "crypto crypto + * payments". + */ +export function crossQueries( + masters: string[], + modifiers: string[], + perMaster = 3, +): Array<{ master: string; query: string }> { + const out: Array<{ master: string; query: string }> = []; + if (masters.length === 0 || modifiers.length === 0) return out; + + const seen = new Set<string>(); + for (let depth = 0; depth < perMaster; depth += 1) { + for (const master of masters) { + const masterTokens = new Set(tokens(master).map(stem)); + // Each subject walks the modifier list from a different offset, so the + // narrowing terms are spread across subjects instead of every subject + // getting the same first modifier and the tail never being used. + const offset = masters.indexOf(master); + let taken = 0; + for (let i = 0; i < modifiers.length && taken <= depth; i += 1) { + const modifier = modifiers[(i + offset) % modifiers.length]; + if (tokens(modifier).every((t) => masterTokens.has(stem(t)))) continue; + if (taken < depth) { + taken += 1; + continue; + } + const query = `${master} ${modifier}`.replace(/\s+/g, " ").trim(); + const key = signature(query); + if (key && !seen.has(key)) { + seen.add(key); + out.push({ master, query }); + } + taken += 1; + } + } + } + return out; +} + +/** + * How many new keywords each subject should get. + * + * Fair share of the remaining target, but weighted toward whatever is *behind* + * — a subject with no coverage is filled before one that already has twenty + * articles. Without that, a queue topped up repeatedly stays in whatever + * proportion it started in, and the site that published nineteen peptide posts + * would go on publishing them at exactly the rate it always had. + * + * @param masters the subject list + * @param coverage existing keyword count per subject (any status) + * @param target how many rows to allocate in total + */ +export function allocate( + masters: string[], + coverage: Map<string, number>, + target: number, +): Map<string, number> { + const out = new Map<string, number>(); + if (masters.length === 0 || target <= 0) return out; + + for (const m of masters) out.set(m, 0); + + // Hand out one row at a time to whichever subject is furthest behind, + // counting what has already been handed out in this pass. An O(target × + // masters) loop over at most 30 × 12, which is not worth a heap to avoid and + // is far easier to prove correct than an apportionment formula. + for (let i = 0; i < target; i += 1) { + let pick = masters[0]; + let lowest = Infinity; + for (const master of masters) { + const total = (coverage.get(master.toLowerCase()) ?? 0) + (out.get(master) ?? 0); + if (total < lowest) { + lowest = total; + pick = master; + } + } + out.set(pick, (out.get(pick) ?? 0) + 1); + } + return out; +} + +/** + * Drop candidates that would produce an article the blog already has. + * + * Compares fingerprints rather than strings, so the pluralised and reordered + * restatements of an existing post are caught. This is the direct answer to + * "spamming blogs with same content": the old dedupe compared lowercased + * keywords exactly, which let "peptide payments" and "peptide payment" + * both through, and both were published — nine days apart, in May. + * + * @param candidates in preference order + * @param published fingerprints already on the blog, from `signature` + */ +export function dropDuplicates<T extends { keyword: string }>( + candidates: T[], + published: Set<string>, +): T[] { + const out: T[] = []; + const seen = new Set(published); + for (const candidate of candidates) { + const key = signature(candidate.keyword); + if (!key || seen.has(key)) continue; + seen.add(key); + out.push(candidate); + } + return out; +} diff --git a/lib/lx/webhookDeliver.ts b/lib/lx/webhookDeliver.ts index 49bf3cf8..fb876632 100644 --- a/lib/lx/webhookDeliver.ts +++ b/lib/lx/webhookDeliver.ts @@ -14,6 +14,8 @@ import type { SupabaseClient } from "@supabase/supabase-js"; import { buildEvent, sendWebhook, type Post } from "@profullstack/autoblog"; import { env } from "../env"; +import { findExchangeCandidates } from "./exchangeMatcher"; +import { buildNetworkBlock } from "./networkBlock"; type ArticleRow = { id: string; @@ -44,6 +46,13 @@ type SiteRow = { webhook_secret: string | null; author_name: string | null; author_url: string | null; + // Network opt-in, resolved for the site that will HOST the post. For a guest + // post that is the partner, not the author: it is the host's readers who see + // the ad unit and the host's owner who agreed to carry it. + project_id: string; + user_id: string; + niche: string | null; + ads_enabled: boolean | null; }; /** @@ -109,7 +118,7 @@ export type DeliveryResult = { error?: string; }; -function articleToPost(article: ArticleRow, site: SiteRow): Post { +function articleToPost(article: ArticleRow, site: SiteRow, networkHtml = ""): Post { const blogRoot = site.blog_root_url.replace(/\/$/, ""); const url = `${blogRoot}/${article.slug}`; @@ -140,7 +149,13 @@ function articleToPost(article: ArticleRow, site: SiteRow): Post { title: article.title, slug: article.slug, excerpt: article.excerpt || article.meta_description || null, - html: `${jsonLd}\n${article.content_html}`, + // The network block goes after the article and outside the JSON-LD, so an + // ad unit and a list of other people's links are never described to a + // crawler as part of this article's body. + html: [jsonLd, article.content_html, networkHtml].filter(Boolean).join("\n"), + // Markdown deliberately does NOT carry the block. A receiver rendering the + // markdown path would have to trust our HTML through its own sanitiser, + // and the ad unit needs a script tag that no markdown renderer will emit. markdown: article.content_markdown, status: "published", published_at: publishedAt, @@ -187,7 +202,7 @@ export async function deliverArticle( const { data: site } = await supabase .from("lx_site") .select( - "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url", + "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url, project_id, user_id, niche, ads_enabled", ) .eq("id", deliveryTargetId) .maybeSingle<SiteRow>(); @@ -208,7 +223,42 @@ export async function deliverArticle( }; } - const post = articleToPost(claimed, site); + // Ads and partner links for the hosting site, if it is in the network. + // + // Built here rather than at generation time because the host is only known + // once the guest-post target has been resolved, and because a redelivery + // should carry a current block rather than one frozen weeks ago. The whole + // thing is best-effort: `buildNetworkBlock` swallows its own failures, and + // this catch covers the rest, because an article that publishes without an + // ad has cost an impression while an article that fails to publish has cost + // the customer the thing they pay for. + const partnerLinks = await findExchangeCandidates(supabase, { + selfSiteId: site.id, + selfNiche: site.niche, + keyword: (claimed.tags ?? []).join(" ") || claimed.title, + slots: 3, + }) + .then((r) => + r.candidates.map((c) => ({ + title: c.title, + url: c.url, + source: "partner" as const, + })), + ) + .catch(() => []); + + const networkHtml = await buildNetworkBlock( + supabase, + site, + claimed.tags ?? [], + partnerLinks, + env.siteUrl, + ).catch((err: unknown) => { + console.warn("[lx] network block failed:", err); + return ""; + }); + + const post = articleToPost(claimed, site, networkHtml); // Reuse the saved delivery id on retries so receivers idempotently // dedupe. SDK uses event.id as the webhook-id header. const event = buildEvent(post, { diff --git a/scripts/purge-offniche-keywords.ts b/scripts/purge-offniche-keywords.ts new file mode 100644 index 00000000..3fc4a32b --- /dev/null +++ b/scripts/purge-offniche-keywords.ts @@ -0,0 +1,128 @@ +// Decide which QUEUED keywords the new gate would never have created. +// +// Reads a JSON dump of queued rows and prints the ids to delete, plus a +// per-site summary. Prints SQL rather than executing it: this deletes +// scheduled work on live blogs, and the diff between "what the gate rejects" +// and "what I am about to remove" should be readable by a person before it +// runs, not inferred from an exit code. +// +// Only `queued` rows are ever considered. A published article is a URL that +// exists on somebody's blog and possibly in an index; removing its keyword row +// would not unpublish it, it would only lose the record that we wrote it. +// +// Usage: npx tsx scripts/purge-offniche-keywords.ts <dump.json> + +import { readFileSync } from "node:fs"; +import { anchorTokens, isOnNiche, resolveMasters, stem, tokens } from "../lib/lx/topicPlan"; + +type Row = { + id: string; + keyword: string; + domain: string; + niche: string | null; + master_keywords: string[] | null; + modifiers: string[] | null; +}; + +/** Same longest-match attribution the research pipeline uses. */ +function attribute(keyword: string, masters: string[]): string | null { + let best: string | null = null; + let bestLen = 0; + // Stemmed on both sides, so attribution and the gate agree about what + // "promo codes" and "promo code" are. They disagreed before, and a keyword + // attributed to a subject the gate then could not match was dropped for a + // reason nobody would have guessed from the strings. + const candidate = new Set(tokens(keyword).map(stem)); + for (const master of masters) { + const masterTokens = tokens(master).map(stem); + if (masterTokens.length === 0) continue; + const hit = masterTokens.filter((t) => candidate.has(t)); + if (hit.length === 0) continue; + const len = hit.join("").length; + if (len > bestLen) { + bestLen = len; + best = master; + } + } + return best; +} + +/** + * Pull the rows out of whatever wrapper the dump arrived in. + * + * A plain array, or the MCP tool envelope — which is the JSON *escaped* inside + * a JSON string, so the quotes need unescaping before it will parse, and the + * rows then sit under a `payload` key. + */ +function extractRows(raw: string): Row[] { + const slice = raw.slice(raw.indexOf("[{"), raw.lastIndexOf("}]") + 2); + const text = slice.includes('\\"') ? slice.replace(/\\"/g, '"') : slice; + const parsed = JSON.parse(text); + const first = parsed[0]; + return Array.isArray(first?.payload) ? first.payload : parsed; +} + +const rows: Row[] = extractRows(readFileSync(process.argv[2], "utf8")); + +const bySite = new Map<string, { kept: string[]; dropped: string[]; ids: string[] }>(); + +const unjudgeable = new Set<string>(); + +for (const row of rows) { + const masters = resolveMasters(row); + const anchors = anchorTokens(row, masters); + const bucket = bySite.get(row.domain) ?? { kept: [], dropped: [], ids: [] }; + + // A site with no subjects configured cannot be judged, and "reject + // everything" is the wrong reading of "I have no basis for an opinion" — + // it would empty a queue whose keywords may be perfectly good, as they are + // on khipu-agency. Left alone, and named in the summary so the real fix + // (set master keywords) is visible. + if (masters.length === 0) { + unjudgeable.add(row.domain); + bucket.kept.push(row.keyword); + bySite.set(row.domain, bucket); + continue; + } + + const master = attribute(row.keyword, masters); + const ok = master !== null && anchors.size > 0 && isOnNiche(row.keyword, master, anchors); + + if (ok) bucket.kept.push(row.keyword); + else { + bucket.dropped.push(row.keyword); + bucket.ids.push(row.id); + } + bySite.set(row.domain, bucket); +} + +const allIds: string[] = []; +let totalKept = 0; +let totalDropped = 0; + +for (const [domain, b] of Array.from(bySite).sort()) { + totalKept += b.kept.length; + totalDropped += b.dropped.length; + allIds.push(...b.ids); + console.log( + `\n=== ${domain}: keep ${b.kept.length}, drop ${b.dropped.length}`, + ); + console.log(` dropping : ${b.dropped.slice(0, 12).join(" | ")}${b.dropped.length > 12 ? " | …" : ""}`); + console.log(` keeping : ${b.kept.slice(0, 8).join(" | ")}${b.kept.length > 8 ? " | …" : ""}`); +} + +console.log(`\n--- total: keep ${totalKept}, drop ${totalDropped} of ${rows.length}`); +if (unjudgeable.size > 0) { + console.log( + `--- left alone (no master keywords set): ${Array.from(unjudgeable).join(", ")}`, + ); +} +console.log("\n-- SQL:"); +for (let i = 0; i < allIds.length; i += 200) { + const chunk = allIds.slice(i, i + 200); + console.log( + `delete from lx_keyword where status='queued' and id in (${chunk + .map((id) => `'${id}'`) + .join(",")});`, + ); +} diff --git a/scripts/verify-topic-plan.ts b/scripts/verify-topic-plan.ts new file mode 100644 index 00000000..4c567971 --- /dev/null +++ b/scripts/verify-topic-plan.ts @@ -0,0 +1,96 @@ +// Dry-run the topic planner against live rows and print what it WOULD do. +// +// No writes, no API calls. Reads each active site's real subjects, modifiers +// and existing keyword history, and reports the allocation plus the cross +// queries. Exists so the fix can be checked against production shapes rather +// than only against the fixtures in tests/lx/topic-plan.test.ts. +// +// Usage: npx tsx scripts/verify-topic-plan.ts (needs .env with the service key) + +import { createClient } from "@supabase/supabase-js"; +import { + allocate, + anchorTokens, + crossQueries, + isOnNiche, + resolveMasters, + resolveModifiers, +} from "../lib/lx/topicPlan"; + +const supabase = createClient( + process.env.NEXT_PUBLIC_SUPABASE_URL!, + process.env.SUPABASE_SERVICE_ROLE_KEY!, + { auth: { persistSession: false } }, +); + +async function main() { + const { data: sites } = await supabase + .from("lx_site") + .select("id, domain, niche, master_keywords, seed_keywords, modifiers") + .eq("status", "active") + .order("domain"); + + for (const site of sites ?? []) { + const masters = resolveMasters(site as never); + const modifiers = resolveModifiers(site as never); + const anchors = anchorTokens(site as never); + + const { data: existing } = await supabase + .from("lx_keyword") + .select("keyword, master_keyword") + .eq("site_id", site.id); + + const coverage = new Map<string, number>(); + for (const row of existing ?? []) { + const master = + row.master_keyword ?? + masters.find((m) => + row.keyword.toLowerCase().includes(m.toLowerCase()), + ); + if (master) { + const k = master.toLowerCase(); + coverage.set(k, (coverage.get(k) ?? 0) + 1); + } + } + + const plan = allocate(masters, coverage, 30); + const crosses = crossQueries(masters, modifiers, 3); + const coveredByCross = new Set(crosses.map((c) => c.master)); + + console.log(`\n=== ${site.domain}`); + console.log(` niche : ${site.niche ?? "(none)"}`); + console.log(` masters : ${masters.length ? masters.join(", ") : "(NONE)"}`); + console.log(` modifiers : ${modifiers.length ? modifiers.join(", ") : "(NONE)"}`); + console.log(` anchors : ${anchors.size}`); + if (anchors.size === 0) { + console.log(" !! would ERROR: no niche and no modifiers"); + continue; + } + const starved = masters.filter((m) => !coveredByCross.has(m)); + if (starved.length) { + console.log(` !! no cross floor for: ${starved.join(", ")}`); + } + console.log( + ` allocation : ${masters + .map((m) => `${m}=${plan.get(m) ?? 0}(have ${coverage.get(m.toLowerCase()) ?? 0})`) + .join(" ")}`, + ); + console.log(` sample : ${crosses.slice(0, 6).map((c) => c.query).join(" | ")}`); + + // How the existing published keywords score under the new gate. + const kept = (existing ?? []).filter((r) => { + const master = masters.find((m) => + r.keyword.toLowerCase().includes(m.toLowerCase()), + ); + return master ? isOnNiche(r.keyword, master, anchors) : false; + }); + console.log( + ` gate : ${kept.length}/${(existing ?? []).length} existing keywords would pass`, + ); + } +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql b/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql new file mode 100644 index 00000000..7f764ed4 --- /dev/null +++ b/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql @@ -0,0 +1,82 @@ +-- Master keywords, topic provenance, and the network opt-in. +-- +-- Three columns, one root cause. +-- +-- `seed_keywords` had become two lists wearing one name: the durable set of +-- subjects a blog is *about*, and the working set the research pipeline was +-- allowed to chew on. Nothing enforced a size on it, so it grew — ten entries +-- on coinpayportal — and the pipeline quietly truncated it twice on the way +-- through (`buildSeeds` to five, the DataForSEO loop to three). The seeds past +-- the cut had never produced a single keyword: thirty-two of them, across nine +-- sites. Splitting the durable list out under its own name and capping it is +-- what makes that class of bug impossible to reintroduce, because a list a +-- human can hold in their head is one the planner can afford to cover *all* of. +-- +-- `master_keyword` on lx_keyword is the provenance the planner needs to do +-- that. Fair allocation across topics is not expressible without knowing which +-- topic each queued row came from, and inferring it after the fact by +-- substring-matching the keyword back to a seed is exactly the loose matching +-- that let "skye peptides" through the relevance gate in the first place. +-- +-- `ads_enabled` is the network opt-in: house ads and partner guest posts in +-- published articles. Default true, because a network everybody has to be +-- asked to join is one that stays empty. + +alter table public.lx_site + add column if not exists master_keywords text[] not null default '{}', + add column if not exists ads_enabled boolean not null default true; + +comment on column public.lx_site.master_keywords is + 'The durable 3-12 subjects this blog covers. The keyword planner allocates its target evenly across ALL of these; nothing truncates it. Crossed with `modifiers` to stay anchored to the site''s own niche.'; + +comment on column public.lx_site.ads_enabled is + 'Opted into the CrawlProof network: house/partner ad units and guest posts are placed in published articles. Default true.'; + +-- A cap the planner can honour rather than a limit it has to work around. +-- Twelve is the point past which an even allocation of a 30-keyword target +-- stops giving each topic enough rows to be worth researching separately. +do $$ +begin + if not exists ( + select 1 from pg_constraint where conname = 'lx_site_master_keywords_len' + ) then + alter table public.lx_site + add constraint lx_site_master_keywords_len + check (coalesce(array_length(master_keywords, 1), 0) <= 12); + end if; +end $$; + +-- Which master subject a queued keyword was researched for. +-- +-- Nullable, and stays nullable: every row written before this migration has no +-- honest answer, and guessing one would corrupt the very coverage figures the +-- planner reads. Those rows are simply invisible to the fair-share maths, which +-- is the correct behaviour — they are already published or queued, so the +-- topics they belong to need no further filling on their account. +alter table public.lx_keyword + add column if not exists master_keyword text; + +comment on column public.lx_keyword.master_keyword is + 'The lx_site.master_keywords entry this keyword was researched for. Null on rows predating topic provenance; those are excluded from coverage rather than guessed at.'; + +create index if not exists lx_keyword_site_master_idx + on public.lx_keyword (site_id, master_keyword); + +-- Seed the durable list from what each site already had. +-- +-- First ten only, mirroring the cap, and only where the column is still empty +-- so a re-run cannot stomp a hand-curated list. This is the one place the old +-- truncation is preserved on purpose: a site with more than ten seeds had no +-- coverage of the tail anyway, and the operator should choose which ten matter +-- rather than have position in an unordered array decide it. +update public.lx_site +set master_keywords = ( + select coalesce(array_agg(s order by ord), '{}') + from ( + select s, ord from unnest(seed_keywords) with ordinality as t(s, ord) + where length(trim(s)) > 0 + limit 10 + ) picked +) +where coalesce(array_length(master_keywords, 1), 0) = 0 + and coalesce(array_length(seed_keywords, 1), 0) > 0; diff --git a/supabase/migrations/20260828140000_lx_feed_sources.sql b/supabase/migrations/20260828140000_lx_feed_sources.sql new file mode 100644 index 00000000..99211548 --- /dev/null +++ b/supabase/migrations/20260828140000_lx_feed_sources.sql @@ -0,0 +1,66 @@ +-- The directory feeds the autoblog reads, and what happened last time. +-- +-- Until now these were fetched live, inside article delivery, on a 3-second +-- timeout. That put a third-party HTTP call on the critical path of the one +-- operation a customer is paying for, and it made the whole source invisible: +-- when a topic feed went missing the block simply came back empty, and there +-- was nowhere to look. Both problems have the same fix — crawl on a schedule, +-- cache what came back, and keep the record of the attempt. +-- +-- The source list is derived, never curated. Topics are the slugged +-- master_keywords of every active site, so a blog that adds a subject gets its +-- feed crawled on the next sweep with nobody filing a request. That is the +-- point: the operator sets subjects, and the feed list follows. + +create table if not exists public.lx_feed_source ( + id uuid primary key default gen_random_uuid(), + -- The directory's own topic slug. Unique because two sites covering the + -- same subject must share one crawl rather than each causing their own. + topic text not null unique, + url text not null, + -- active: crawl it. given_up: too many consecutive failures, left in place + -- so the status page can still explain the absence rather than the row + -- silently vanishing and the topic looking like it was never wanted. + status text not null default 'active' + check (status in ('active', 'given_up')), + last_fetch_at timestamptz, + last_success_at timestamptz, + last_status integer, + last_error text, + item_count integer not null default 0, + consecutive_failures integer not null default 0, + created_at timestamptz not null default now(), + updated_at timestamptz not null default now() +); + +comment on table public.lx_feed_source is + 'RSS Amplifier topic feeds the autoblog cites from. Derived from active sites'' master_keywords by the worker''s feed sweep — not hand-curated.'; + +-- The sweep picks the least-recently-fetched active source, so this is the +-- index that decides throughput. +create index if not exists lx_feed_source_due_idx + on public.lx_feed_source (status, last_fetch_at nulls first); + +create table if not exists public.lx_feed_item ( + id uuid primary key default gen_random_uuid(), + source_id uuid not null references public.lx_feed_source(id) on delete cascade, + title text not null, + link text not null, + first_seen_at timestamptz not null default now(), + -- Link rather than guid: the link is what gets rendered, and it is the only + -- field whose uniqueness actually prevents the same anchor appearing twice. + unique (source_id, link) +); + +comment on table public.lx_feed_item is + 'Cached posts from a topic feed. Read by the article delivery path so a third-party fetch is never on the critical path of publishing.'; + +create index if not exists lx_feed_item_source_seen_idx + on public.lx_feed_item (source_id, first_seen_at desc); + +-- Service-role only. Nothing here is user data and no browser reads it +-- directly; the status page goes through a server component holding the +-- service client. Enabling RLS with no policy is what makes that explicit +-- rather than incidental. +alter table public.lx_feed_source enable row level security; +alter table public.lx_feed_item enable row level security; diff --git a/tests/lx/feed-crawl.test.ts b/tests/lx/feed-crawl.test.ts new file mode 100644 index 00000000..ac5d29a7 --- /dev/null +++ b/tests/lx/feed-crawl.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, it } from "vitest"; +import { topicFor } from "@/lib/lx/feedCrawl"; +import { ago } from "@/lib/lx/feedCrawlStats"; + +describe("topicFor", () => { + it("uses the first meaningful token of a multi-word subject", () => { + // "merchant account payments" is not a directory topic; "merchant" is. + expect(topicFor("merchant account payments")).toBe("merchant"); + }); + + it("keeps a short subject that the token floor would otherwise drop", () => { + // These are exactly the high-value subjects — four of the five that had + // never produced a keyword were short words. Losing them here would + // reintroduce the same blind spot one layer down. + expect(topicFor("iptv")).toBe("iptv"); + expect(topicFor("weed")).toBe("weed"); + }); + + it("slugs punctuation rather than emitting an unfetchable path", () => { + expect(topicFor("high-risk")).toBe("high"); + }); + + it("returns null for a subject with nothing in it", () => { + expect(topicFor(" ")).toBeNull(); + expect(topicFor("!!!")).toBeNull(); + }); +}); + +describe("ago", () => { + const now = Date.parse("2026-08-28T12:00:00.000Z"); + + it("reports never for a source that has not been read", () => { + expect(ago(null, now)).toBe("never"); + }); + + it.each([ + ["2026-08-28T11:45:00.000Z", "15m ago"], + ["2026-08-28T09:00:00.000Z", "3h ago"], + ["2026-08-26T12:00:00.000Z", "2d ago"], + ["2026-08-28T11:59:45.000Z", "just now"], + ])("formats %j as %j", (iso, expected) => { + expect(ago(iso, now)).toBe(expected); + }); + + it("does not render a negative age when a clock is ahead", () => { + // Worker and web can be on different hosts; "-3m ago" reads as a bug in + // the page rather than a skewed clock, so it is clamped. + expect(ago("2026-08-28T12:03:00.000Z", now)).toBe("just now"); + }); +}); diff --git a/tests/lx/network-block.test.ts b/tests/lx/network-block.test.ts new file mode 100644 index 00000000..fad6b1f4 --- /dev/null +++ b/tests/lx/network-block.test.ts @@ -0,0 +1,133 @@ +// The insertion block writes other people's text into a customer's page. +// +// Most of what follows is injection and link-safety, because that is what this +// module actually risks. The titles come from RSS feeds crawled off the open +// web; the fact that they reach us via our own directory does not make them +// ours, and the page they land on belongs to somebody paying us. + +import { describe, expect, it } from "vitest"; +import { + adUnitHtml, + escapeHtml, + isSafeHref, + networkLinksHtml, + type NetworkLink, +} from "@/lib/lx/networkBlock"; +import { itemEntries } from "@/lib/lx/feedTopics"; + +describe("escapeHtml", () => { + it("neutralises a script tag in a feed title", () => { + expect(escapeHtml('<script>alert(1)</script>')).toBe( + "<script>alert(1)</script>", + ); + }); + + it("escapes the ampersand first so nothing is double-escaped", () => { + // "<" must not become "&lt;" — which is what a naive ordering does. + expect(escapeHtml("<")).toBe("<"); + expect(escapeHtml("&")).toBe("&"); + expect(escapeHtml("&<")).toBe("&<"); + }); + + it("escapes both quote characters, since these land in attributes", () => { + expect(escapeHtml(`" onload="x`)).toBe("" onload="x"); + expect(escapeHtml("' onload='x")).toBe("' onload='x"); + }); +}); + +describe("isSafeHref", () => { + it.each(["javascript:alert(1)", "data:text/html,<script>", "vbscript:msgbox", "file:///etc/passwd"])( + "rejects %j", + (href) => expect(isSafeHref(href)).toBe(false), + ); + + it.each(["https://example.com/post", "http://example.com/post"])( + "allows %j", + (href) => expect(isSafeHref(href)).toBe(true), + ); + + it("rejects a relative link rather than emitting a broken anchor", () => { + expect(isSafeHref("/relative/path")).toBe(false); + }); +}); + +describe("networkLinksHtml", () => { + const links: NetworkLink[] = [ + { title: "Partner post", url: "https://partner.example/a", source: "partner" }, + { title: "Directory post", url: "https://stranger.example/b", source: "directory" }, + ]; + + it("follows partner links and nofollows directory ones", () => { + const html = networkLinksHtml(links); + // The partner opted into an exchange; the stranger did not, and passing + // them ranking signal would be spending someone else's reputation. + expect(html).toMatch(/href="https:\/\/partner\.example\/a" rel="noopener"/); + expect(html).toMatch(/href="https:\/\/stranger\.example\/b" rel="nofollow ugc noopener"/); + }); + + it("drops an unsafe link instead of rendering the whole block", () => { + const html = networkLinksHtml([ + ...links, + { title: "Bad", url: "javascript:alert(1)", source: "directory" }, + ]); + expect(html).not.toContain("javascript:"); + expect(html).toContain("partner.example"); + }); + + it("escapes a hostile title", () => { + const html = networkLinksHtml([ + { + title: '</a><script>alert(document.cookie)</script>', + url: "https://example.com/x", + source: "directory", + }, + ]); + expect(html).not.toContain("<script>"); + expect(html).toContain("<script>"); + }); + + it("returns nothing at all when every link was unsafe", () => { + // An empty <ul> under a heading reads as a broken feature. + expect(networkLinksHtml([{ title: "x", url: "javascript:1", source: "directory" }])).toBe(""); + }); + + it("returns nothing for an empty list", () => { + expect(networkLinksHtml([])).toBe(""); + }); +}); + +describe("adUnitHtml", () => { + it("carries the slot on the element so ad.js can fill it", () => { + const html = adUnitHtml("slot-123", "https://crawlproof.com"); + expect(html).toContain('data-cp-ad data-slot="slot-123"'); + expect(html).toContain('src="https://crawlproof.com/ad.js"'); + }); + + it("does not double the slash when the origin has a trailing one", () => { + expect(adUnitHtml("s", "https://crawlproof.com/")).toContain("https://crawlproof.com/ad.js"); + }); +}); + +describe("itemEntries", () => { + const feed = `<rss><channel><title>topic — RSS Amplifier + A perfectly ordinary post about build systemshttps://good.example/1 + SponsoredOur own advertisement, laundered as editorialhttps://ad.example/2 + Another post that is long enough to qualify (sponsored)https://ad.example/3 + Relative links must not become anchors here/relative + `; + + it("excludes sponsored items by category and by title suffix", () => { + const titles = itemEntries(feed).map((e) => e.title); + expect(titles).toEqual(["A perfectly ordinary post about build systems", "Relative links must not become anchors here"]); + }); + + it("keeps an absolute link and nulls a relative one", () => { + const entries = itemEntries(feed); + expect(entries[0].link).toBe("https://good.example/1"); + expect(entries[1].link).toBeNull(); + }); + + it("ignores the channel title", () => { + expect(itemEntries(feed).map((e) => e.title)).not.toContain("topic — RSS Amplifier"); + }); +}); diff --git a/tests/lx/topic-plan-platform.test.ts b/tests/lx/topic-plan-platform.test.ts new file mode 100644 index 00000000..aae2ea35 --- /dev/null +++ b/tests/lx/topic-plan-platform.test.ts @@ -0,0 +1,239 @@ +// The same bug on every other blog, pinned. +// +// The peptide articles were what got noticed, but pulling the real +// lx_keyword rows for all seventeen active sites showed the unanchored gate +// producing wholesale garbage everywhere. Every string below was actually in +// the queue or already published on the site named. +// +// Read the "rejects" lists as the specification: these are the searches a +// keyword tool returns when you hand it a bare subject and ask for related +// terms, and no amount of ranking distinguishes them from a real topic. +// Only the second half of the gate does. + +import { describe, expect, it } from "vitest"; +import { + anchorTokens, + isOnNiche, + resolveMasters, + resolveModifiers, +} from "@/lib/lx/topicPlan"; + +type Case = { + domain: string; + site: { + master_keywords: string[]; + modifiers: string[]; + niche: string; + }; + /** [keyword, the subject it was researched for] */ + rejects: Array<[string, string]>; + keeps: Array<[string, string]>; +}; + +const CASES: Case[] = [ + { + // Home-alarm installers on a CI/CD supply-chain security blog. This is the + // case that forced anchors to exclude subject words: "security" is both a + // subject word and a niche word, so before that fix "adt home security" + // satisfied both halves of the gate using the single token "security". + domain: "vu1nz.com", + site: { + master_keywords: [ + "ci/cd security", "supply chain security", "github actions security", + "devops security", "npm security", + ], + modifiers: [], + niche: "CI/CD and supply chain security", + }, + rejects: [ + ["adt home security", "devops security"], + ["brinks home security", "devops security"], + ["vivint security", "devops security"], + ["security public storage", "devops security"], + ["security dodge", "devops security"], + ["safe haven security", "devops security"], + ["security clearance", "devops security"], + ["weiser security", "devops security"], + ], + keeps: [ + ["npm security best practices", "npm security"], + ["github actions security checklist", "github actions security"], + ["supply chain security tools", "supply chain security"], + ], + }, + { + // Pregnancy-test evaporation lines and a Roblox game, on a SOC blog. + domain: "threatcrush.com", + site: { + master_keywords: [ + "security operations", "threat detection", "incident response", + "soc tools", "detection engineering", "network security monitoring", + ], + modifiers: [], + niche: "security operations and threat detection", + }, + rejects: [ + ["evap line first response", "incident response"], + ["first response evap line", "incident response"], + ["emergency response liberty county", "incident response"], + ["911 live incident", "incident response"], + ["community emergency response team", "incident response"], + ["personal emergency response system", "incident response"], + ], + keeps: [ + ["incident response platform", "incident response"], + ["soc tools comparison", "soc tools"], + ["threat detection software", "threat detection"], + ], + }, + { + // TV remotes and a lawn mower, on screen-sharing software. + domain: "pairux.com", + site: { + master_keywords: [ + "screen sharing", "remote desktop", "remote collaboration", + "remote control", "video conferencing", "pair programming", + ], + modifiers: [], + niche: "collaborative screen sharing software", + }, + rejects: [ + ["samsung tv remote", "remote control"], + ["vizio tv remote", "remote control"], + ["roku voice remote", "remote control"], + ["garage door opener remote", "remote control"], + ["firestick remote", "remote control"], + ["universal remote", "remote control"], + ], + keeps: [ + ["screen sharing software for remote teams", "screen sharing"], + ["remote desktop software", "remote desktop"], + ], + }, + { + // Publicly traded companies, on a torrent-streaming blog. + domain: "bittorrented.com", + site: { + master_keywords: [ + "streaming", "torrents", "iptv", "media tech", "vpn", "cord cutting", + ], + modifiers: [], + niche: "streaming torrent media tech", + }, + rejects: [ + ["tyler technologies", "media tech"], + ["lumen technologies", "media tech"], + ["micron technology", "media tech"], + ["palantir technologies", "media tech"], + ["trane technologies", "media tech"], + ["applied industrial technologies", "media tech"], + ["wake tech", "media tech"], + ["lanier tech", "media tech"], + ], + keeps: [["streaming media software", "streaming"]], + }, + { + // Numerical optimization and a dog breed, on an AEO blog. "xoloitzcuintli + // price" was genuinely queued. + domain: "crawlproof.com", + site: { + master_keywords: [ + "answer engine optimization", "AEO", "LLM crawlers", + "AI answer engines", "schema markup", + ], + modifiers: [], + niche: "answer engine optimization for websites", + }, + rejects: [ + ["bayesian optimization", "answer engine optimization"], + ["convex optimization", "answer engine optimization"], + ["scipy optimization minimize", "answer engine optimization"], + ["ant colony optimization algorithms", "answer engine optimization"], + ["topology optimization", "answer engine optimization"], + ["charles babbage analytical engine", "answer engine optimization"], + ["matlab optimization toolbox", "answer engine optimization"], + ], + keeps: [["answer engine optimization for websites", "answer engine optimization"]], + }, + { + // Hardware tool brands and biochemistry, on an AI-agent-standards blog. + domain: "logicsrc.com", + site: { + master_keywords: [ + "ai agents", "agent orchestration", "developer tools", + "api integration", "open standards", "mcp", + ], + modifiers: [], + niche: "open AI agent standards", + }, + rejects: [ + ["mac tools", "developer tools"], + ["metabo tools", "developer tools"], + ["jb tools", "developer tools"], + ["osint tools", "developer tools"], + ["intercalating agent", "ai agents"], + ["causative agent", "ai agents"], + ["principal agent problem", "ai agents"], + ], + keeps: [["agent orchestration platform", "agent orchestration"]], + }, +]; + +// A limitation worth stating rather than engineering around. +// +// "remote control lawn mower" contains pairux's subject "remote control" in +// full, so the complete-match rule admits it. No lexical test separates that +// from "abercrombie promo code", which contains "promo codes" in full and is +// exactly what a coupon blog should write. +// +// The real defect is upstream: "remote control" is two common English words +// and is too broad to be a subject at all. The fix is editorial — drop it from +// pairux's master keywords, where "screen sharing" and "pair programming" +// already cover the ground — not another clause here. Pinned so the tradeoff +// is visible if someone later wonders why this one gets through. +describe("known limitation: an over-broad subject", () => { + const site = { + master_keywords: ["screen sharing", "remote control"], + modifiers: [], + niche: "collaborative screen sharing software", + }; + const anchors = anchorTokens(site, resolveMasters(site)); + + it("admits a keyword that contains the whole over-broad subject", () => { + expect(isOnNiche("remote control lawn mower", "remote control", anchors)).toBe(true); + }); + + it("still rejects the partial matches, which is most of the damage", () => { + for (const junk of ["samsung tv remote", "universal remote", "firestick remote"]) { + expect(isOnNiche(junk, "remote control", anchors)).toBe(false); + } + }); +}); + +describe.each(CASES)("$domain", ({ site, rejects, keeps }) => { + const masters = resolveMasters(site); + const anchors = anchorTokens(site, masters); + + it("resolves an anchor set, so the site is never researched unanchored", () => { + expect(resolveModifiers(site, masters).length).toBeGreaterThan(0); + expect(anchors.size).toBeGreaterThan(0); + }); + + it("shares no token between the anchors and the subjects", () => { + // The invariant that makes the two halves of the gate independent. + const masterTokens = new Set( + masters.flatMap((m) => m.toLowerCase().split(/[^a-z0-9]+/i)), + ); + for (const anchor of anchors) { + expect(masterTokens.has(anchor)).toBe(false); + } + }); + + it.each(rejects)("rejects %j", (keyword, master) => { + expect(isOnNiche(keyword, master, anchors)).toBe(false); + }); + + it.each(keeps)("keeps %j", (keyword, master) => { + expect(isOnNiche(keyword, master, anchors)).toBe(true); + }); +}); diff --git a/tests/lx/topic-plan.test.ts b/tests/lx/topic-plan.test.ts new file mode 100644 index 00000000..acbc5ebb --- /dev/null +++ b/tests/lx/topic-plan.test.ts @@ -0,0 +1,289 @@ +// The peptide regression, pinned. +// +// Every fixture here is real: the master list, the modifiers and the niche are +// coinpayportal.com's live values, and the rejected keywords are titles the +// site actually published in June and July 2026. If these pass, the specific +// nineteen articles that prompted this work cannot be generated again. + +import { describe, expect, it } from "vitest"; +import { + allocate, + DEFAULT_MODIFIERS, + anchorTokens, + crossQueries, + dropDuplicates, + isOnNiche, + MAX_MASTERS, + resolveMasters, + resolveModifiers, + signature, + stem, +} from "@/lib/lx/topicPlan"; + +// coinpayportal.com, exactly as stored. +const SITE = { + master_keywords: [ + "peptide", "crypto", "blockchain", "cryptocurrency", "casino", + "marijuana", "dispensary", "weed", "iptv", "torrents", + ], + seed_keywords: [ + "peptide", "crypto", "blockchain", "cryptocurrency", "casino", + "marijuana", "dispensary", "weed", "iptv", "torrents", + ], + modifiers: [ + "payments", "transactions", "merchant account", + "payment gateway", "payment processing", + ], + niche: "crypto payments for high-risk merchants", +}; + +describe("resolveMasters", () => { + it("keeps every subject — the truncation that caused the outage is gone", () => { + // The old buildSeeds() ended in .slice(0, 5) and the expansion loop then + // took .slice(0, 3). marijuana, dispensary, weed, iptv and torrents had + // never produced a single keyword in the site's lifetime. + const masters = resolveMasters(SITE); + expect(masters).toHaveLength(10); + for (const forgotten of ["marijuana", "dispensary", "weed", "iptv", "torrents"]) { + expect(masters).toContain(forgotten); + } + }); + + it("falls back to seed_keywords so it is safe to deploy before the backfill", () => { + expect(resolveMasters({ seed_keywords: ["alpha", "beta"] })).toEqual(["alpha", "beta"]); + }); + + it("caps an oversized list rather than rejecting it", () => { + const many = Array.from({ length: 40 }, (_, i) => `subject${i}`); + expect(resolveMasters({ master_keywords: many })).toHaveLength(MAX_MASTERS); + }); + + it("drops duplicates and blanks without shifting the rest", () => { + expect(resolveMasters({ master_keywords: ["crypto", " ", "Crypto", "casino"] })) + .toEqual(["crypto", "casino"]); + }); +}); + +describe("resolveModifiers", () => { + it("prefers the explicit column", () => { + expect(resolveModifiers(SITE, resolveMasters(SITE))).toContain("merchant account"); + }); + + it("mines the niche when the column is empty — 13 of 17 live sites", () => { + const derived = resolveModifiers( + { modifiers: [], niche: "security operations and threat detection" }, + ["siem", "edr"], + ); + expect(derived).toEqual( + expect.arrayContaining(["security", "operations", "threat", "detection"]), + ); + // "and" carries no narrowing and would weaken the gate. + expect(derived).not.toContain("and"); + }); + + it("subtracts the subjects from the niche-derived terms", () => { + // vu1nz.com: niche "CI/CD and supply chain security" says nothing its + // own subjects do not already say. A term that is also a subject cannot + // narrow anything, and keeping it is what admits "adt home security". + const derived = resolveModifiers( + { modifiers: [], niche: "CI/CD and supply chain security" }, + ["ci/cd security", "supply chain security", "devops security"], + ); + expect(derived).not.toContain("security"); + expect(derived).not.toContain("supply"); + }); + + it("falls back to the commercial vocabulary when the niche adds nothing", () => { + const derived = resolveModifiers( + { modifiers: [], niche: "CI/CD and supply chain security" }, + ["ci/cd security", "supply chain security", "devops security"], + ); + expect(derived).toBe(DEFAULT_MODIFIERS); + }); +}); + +describe("isOnNiche — the gate that was missing", () => { + const anchors = anchorTokens(SITE, resolveMasters(SITE)); + + // These are real published titles. Each is a peptide *vendor* — a + // competitor storefront in an industry coinpayportal sells payment + // processing TO, not one it writes about. + it.each([ + "skye peptides", + "pure peptide labs", + "wolverine stack peptides", + "apex peptides", + "biotech peptides", + "melanotan 2 peptides", + "ghk peptide", + "aod 9604 peptide", + "peptide crafters", + "lab 34 peptides and proteins", + ])("rejects the vendor term %j", (keyword) => { + expect(isOnNiche(keyword, "peptide", anchors)).toBe(false); + }); + + // These are the May 2026 posts — the period before the regression, when + // the pipeline still crossed seeds with modifiers. + it.each([ + "peptide merchant account", + "peptide payment processing", + "peptide payment gateway", + "peptide payments", + "peptide transactions", + ])("keeps the on-niche keyword %j", (keyword) => { + expect(isOnNiche(keyword, "peptide", anchors)).toBe(true); + }); + + it("requires the subject as well as the anchor", () => { + // Anchored, but about a subject this blog does not cover. + expect(isOnNiche("plumbing merchant account", "peptide", anchors)).toBe(false); + }); + + it("refuses everything when a site has no anchor at all", () => { + // Defaulting to "allow" here would restore the exact behaviour that + // published the vendor articles, so an anchorless site must fail closed + // and let the caller raise it. + expect(isOnNiche("peptide merchant account", "peptide", new Set())).toBe(false); + }); +}); + +describe("crossQueries", () => { + const crosses = crossQueries( + resolveMasters(SITE), + resolveModifiers(SITE, resolveMasters(SITE)), + 3, + ); + + it("gives every subject coverage, including the five that never had any", () => { + const covered = new Set(crosses.map((c) => c.master)); + expect(covered.size).toBe(10); + for (const forgotten of ["marijuana", "dispensary", "weed", "iptv", "torrents"]) { + expect(covered.has(forgotten)).toBe(true); + } + }); + + it("never emits a bare subject — that is what returned the vendor terms", () => { + for (const { master, query } of crosses) { + expect(query).not.toBe(master); + expect(query.length).toBeGreaterThan(master.length); + } + }); + + it("orders subject-major so a truncated run loses depth, not coverage", () => { + // The first ten entries must be ten *different* subjects: a budget cut + // partway through should still have touched everything. + const firstPass = crosses.slice(0, 10).map((c) => c.master); + expect(new Set(firstPass).size).toBe(10); + }); + + it("skips a cross whose modifier is already in the subject", () => { + const out = crossQueries(["crypto"], ["crypto", "payments"], 2); + expect(out.map((c) => c.query)).not.toContain("crypto crypto"); + expect(out.map((c) => c.query)).toContain("crypto payments"); + }); + + it("produces nothing without modifiers, rather than expanding bare", () => { + expect(crossQueries(["peptide"], [], 3)).toEqual([]); + }); +}); + +describe("allocate", () => { + it("fills the subjects that are behind first", () => { + // The live skew: 23 peptide articles, 11 crypto, nothing else. + const coverage = new Map([["peptide", 23], ["crypto", 11]]); + const out = allocate(resolveMasters(SITE), coverage, 30); + + expect(out.get("peptide") ?? 0).toBe(0); + expect(out.get("crypto") ?? 0).toBe(0); + // Everything with no history gets a share. + for (const starved of ["marijuana", "dispensary", "weed", "iptv", "torrents"]) { + expect(out.get(starved) ?? 0).toBeGreaterThan(0); + } + }); + + it("splits evenly when nothing has history", () => { + const out = allocate(["a", "b", "c"], new Map(), 30); + expect([out.get("a"), out.get("b"), out.get("c")]).toEqual([10, 10, 10]); + }); + + it("allocates exactly the target", () => { + const masters = resolveMasters(SITE); + const total = Array.from(allocate(masters, new Map(), 30).values()) + .reduce((a, b) => a + b, 0); + expect(total).toBe(30); + }); + + it("is a no-op for an empty subject list", () => { + expect(allocate([], new Map(), 30).size).toBe(0); + }); +}); + +describe("stem — plural collapsing", () => { + // The subtle one. An earlier version stripped two characters from every + // "-es", so "codes" became "cod" while "code" stayed "code" and the two + // never matched. That silently broke plural matching everywhere, and made + // several of the peptide rejections above pass for the wrong reason: + // "skye peptides" was rejected because "peptides" stemmed to "peptid" and + // failed to match the subject at all, not because it lacked an anchor. + it.each([ + ["codes", "code"], + ["peptides", "peptide"], + ["payments", "payment"], + ["alternatives", "alternative"], + ["practices", "practice"], + ["transactions", "transaction"], + ])("collapses %j onto %j", (plural, singular) => { + expect(stem(plural)).toBe(stem(singular)); + }); + + it.each([ + ["boxes", "box"], + ["matches", "match"], + ["searches", "search"], + ])("still strips the epenthetic e in %j", (plural, singular) => { + expect(stem(plural)).toBe(stem(singular)); + }); + + it("handles -ies", () => { + expect(stem("companies")).toBe(stem("company")); + }); + + it("leaves a short word alone rather than mangling it", () => { + // "aeo" and "soc" must not lose characters they cannot spare. + expect(stem("aeo")).toBe("aeo"); + expect(stem("gas")).toBe("gas"); + }); +}); + +describe("signature + dropDuplicates", () => { + it("collapses the plural restatement that shipped twice", () => { + // Both published, nine days apart, in May 2026. + expect(signature("peptide payments")).toBe(signature("peptide payment")); + }); + + it("collapses a reordering", () => { + expect(signature("merchant account peptide")).toBe(signature("peptide merchant account")); + }); + + it("keeps genuinely different subjects apart", () => { + expect(signature("peptide payments")).not.toBe(signature("casino payments")); + }); + + it("drops candidates already on the blog", () => { + const published = new Set([signature("peptide payments")]); + const out = dropDuplicates( + [{ keyword: "peptide payment" }, { keyword: "casino payments" }], + published, + ); + expect(out.map((c) => c.keyword)).toEqual(["casino payments"]); + }); + + it("drops duplicates within the candidate list itself", () => { + const out = dropDuplicates( + [{ keyword: "iptv payments" }, { keyword: "iptv payment" }], + new Set(), + ); + expect(out).toHaveLength(1); + }); +}); diff --git a/worker/index.ts b/worker/index.ts index f695caa1..b1d5f992 100644 --- a/worker/index.ts +++ b/worker/index.ts @@ -42,6 +42,7 @@ import { processUserAlerts } from "../lib/alerts/worker"; import { processDuePortScans } from "../lib/prober-queue"; import { processDueMonitors } from "../lib/uptime"; import { processDuePromoteLists } from "../lib/promote/sweep"; +import { crawlFeeds } from "../lib/lx/feedCrawl"; import { reapStalePublishingJobs } from "../lib/promote/jobs"; import { ingestDueFeeds } from "../lib/promote/ingest"; import { refreshCookieSessions } from "../lib/sp/sessionRefresh"; @@ -1475,6 +1476,39 @@ setInterval( SESSION_REFRESH_TICK_MS, ); +// Directory feed crawl. +// +// The autoblog cites real posts from RSS Amplifier topic feeds. Those used to +// be fetched live inside article delivery on a 3s budget — a third-party HTTP +// call on the critical path of the thing customers pay for, and a source with +// no visibility when it broke. This reads them here instead, on the worker's +// own clock, and delivery reads rows. +// +// The source list is derived from every active site's master_keywords on each +// sweep, so adding a subject to a blog gets its feed crawled without anybody +// filing a request. Tick is 30 minutes; individual sources refresh at most +// every 6h (REFRESH_AFTER_MS), so most ticks find only a few due and are cheap. +const FEED_CRAWL_TICK_MS = 30 * 60 * 1000; // 30 min +let feedCrawlRunning = false; +async function feedCrawlSweep() { + if (feedCrawlRunning) return; // a slow directory must not stack up sweeps + feedCrawlRunning = true; + try { + const r = await crawlFeeds(supabase); + if (r.fetched > 0) { + console.log( + `[worker] feed crawl sources=${r.sources} fetched=${r.fetched} ok=${r.succeeded} new=${r.newItems} gave_up=${r.gaveUp}`, + ); + } + } finally { + feedCrawlRunning = false; + } +} +setInterval( + () => feedCrawlSweep().catch((e) => console.error("[worker] feed crawl sweep", e)), + FEED_CRAWL_TICK_MS, +); + // Bind to loopback by default so the worker isn't reachable from the public // internet when colocated with the app. Override with WORKER_BIND=0.0.0.0 to // run as a separate Railway service. @@ -1487,4 +1521,5 @@ server.listen(port, bindHost, () => { promoteSweep().catch((e) => console.error("[worker] promote sweep", e)); promoteIngestSweep().catch((e) => console.error("[worker] promote ingest", e)); sessionRefreshSweep().catch((e) => console.error("[worker] session refresh sweep", e)); + feedCrawlSweep().catch((e) => console.error("[worker] feed crawl sweep", e)); });