From 3f772681a2b68038afc82d7f325efb60d473bd26 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 28 Aug 2026 15:00:49 +0000 Subject: [PATCH 1/5] Write about every subject, and never one on its own MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit coinpayportal.com published nineteen consecutive articles about peptide vendors — "skye peptides", "pure peptide labs", "wolverine stack peptides". Those are competitor storefronts in an industry the site sells payment processing to. Two independent defects produced them. Truncation. The site listed ten subjects. buildSeeds() sliced them to five, the DataForSEO loop took three of those, and subject #1 was handed to the buyer-journey model as its entire query. marijuana, dispensary, weed, iptv and torrents had never produced a single keyword. Platform-wide that is thirty-two subjects across nine sites that have never once been written about. The top-up sweep re-ran the same truncated set every time the queue drained, so the skew compounded instead of averaging out. An unanchored gate. A candidate survived if it contained the seed token. Expanding the bare word "peptide" returns the peptide industry's own vocabulary, and all of it passed a test that only ever asked "is this about peptides?" — never "is this about what we sell to them?". Both answers are the same: a subject is only ever researched crossed with a modifier. "peptide" is not a topic; "peptide merchant account" is. The cross is also the floor — it is built from two operator-controlled columns, so a blog that cannot reach any upstream still publishes, and still publishes about itself. An anchorless site now errors rather than falling back to the expansion that caused this. Allocation is fair-share weighted toward whatever is behind, and emission is interleaved: fairness measured over a quarter still reads as spam if it arrives as six peptide posts in a row. Duplicate detection moved from exact string match to a stemmed, order-independent fingerprint, which is what lets it see that "peptide payments" and "peptide payment" are the same article. Both shipped, nine days apart, in May. Covering ten subjects costs less than the three it replaced: the crosses go into one keywordIdeas call, which takes 200 seeds. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_013cbcXdpLFUgJZuVfCB4Gzv --- lib/lx/keywordsResearch.ts | 519 +++++++++++------- lib/lx/topicPlan.ts | 327 +++++++++++ ...20000_autoblog_master_keywords_and_ads.sql | 82 +++ tests/lx/topic-plan.test.ts | 226 ++++++++ 4 files changed, 970 insertions(+), 184 deletions(-) create mode 100644 lib/lx/topicPlan.ts create mode 100644 supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql create mode 100644 tests/lx/topic-plan.test.ts diff --git a/lib/lx/keywordsResearch.ts b/lib/lx/keywordsResearch.ts index 0de6e48f..65f8c3b9 100644 --- a/lib/lx/keywordsResearch.ts +++ b/lib/lx/keywordsResearch.ts @@ -1,14 +1,20 @@ // Keyword research pipeline (PRD §6.1, §15). // -// Inputs: siteId — site must have niche + (optionally) target_audiences set. -// Outputs: up to 30 new rows in lx_keyword (status='queued'), scheduled -// across the next ~6 weeks honoring publish_days + daily_article_count. +// Inputs: siteId — the site's `master_keywords` (the durable subject list) and +// `modifiers` (what the site actually does) are the two lists this +// works from. `niche` backfills the modifiers when that column is empty. +// Outputs: up to 30 new rows in lx_keyword (status='queued'), allocated evenly +// across every master subject and scheduled across the next ~6 weeks +// honoring publish_days + daily_article_count. // -// Caching: lx_keyword_metrics rows live 60 days. Before paying DataForSEO -// for a seed, we check whether we already have a recent expansion cached -// (heuristic: same seed string, region='us'). For v1 we always re-fetch on -// manual trigger; cache is queried per-keyword to short-circuit the -// volume backfill step. +// **Every subject is researched, and none is researched alone.** Both halves +// matter and both were broken. See lib/lx/topicPlan.ts for the failure this +// was written against — nineteen articles about peptide vendors on a payments +// blog — and why the fix is a cross product rather than a bigger slice. +// +// Cost note: the cross queries all go into a *single* keywordIdeas call, which +// accepts 200 seeds. Covering ten subjects properly is therefore cheaper than +// the three separate calls this replaced, not more expensive. // // Spend ledger: every DataForSEO call writes a row to lx_dataforseo_usage. @@ -25,6 +31,17 @@ import { } from "./buyerJourneyKeywords"; import { DataForSeoClient, filterOutliers, type DfsKeywordRow } from "./dataforseo"; import { nextPublishAt } from "./schedule"; +import { + allocate, + anchorTokens, + crossQueries, + dropDuplicates, + isOnNiche, + resolveMasters, + resolveModifiers, + signature, + tokens, +} from "./topicPlan"; type SiteRow = { id: string; @@ -33,6 +50,8 @@ type SiteRow = { target_audiences: string[]; description: string | null; seed_keywords: string[]; + master_keywords: string[]; + modifiers: string[]; keywords: string[]; competitors: string[]; tone: string | null; @@ -45,24 +64,20 @@ const TARGET_KEYWORDS = 30; const MIN_VOLUME = 50; const MIN_BUYER_JOURNEY_VOLUME = 10; const MIN_WORDS = 2; -const PER_SEED_LIMIT = 200; +const IDEAS_LIMIT = 600; const MAX_BUYER_JOURNEY_VOLUME_LOOKUP = 160; -const SEED_TOKEN_STOPLIST = new Set([ - "the","and","for","with","you","your","that","this","from","into","over", - "but","not","are","was","were","has","had","have","its","off","out","all", - "any","new","get","how","why","what","who","best","top", -]); - -function buildSeeds(site: SiteRow): string[] { - const seeds: string[] = []; - for (const s of site.seed_keywords ?? []) seeds.push(s.trim()); - if (site.niche) seeds.push(site.niche.trim()); - for (const a of (site.target_audiences ?? []).slice(0, 3)) { - if (site.niche && a) seeds.push(`${site.niche} for ${a}`); - else if (a) seeds.push(a); - } - return Array.from(new Set(seeds.filter((s) => s.length > 0))).slice(0, 5); -} + +/** + * Cross depth per subject. + * + * Three narrowing terms per subject: enough that a subject yields more than a + * single phrasing, few enough that ten subjects still fit one API call with + * room for the seed list to grow. + */ +const CROSS_PER_MASTER = 3; + +/** A candidate with the subject it belongs to. */ +type Candidate = { row: DfsKeywordRow; master: string }; function parseStoredKeyword(row: string): DfsKeywordRow | null { const idx = row.indexOf(","); @@ -82,20 +97,39 @@ function parseStoredKeyword(row: string): DfsKeywordRow | null { }; } -function seedTokens(seed: string): string[] { - return seed - .toLowerCase() - .split(/[\s-]+/) - .map((t) => t.replace(/[^a-z0-9]/g, "")) - .filter((t) => t.length >= 4 && !SEED_TOKEN_STOPLIST.has(t)); -} - type KeywordBoost = { priority: number; intent: BuyerJourneyKeywordIntent; clusterType: BuyerJourneyClusterType; }; +/** + * Which subject a candidate belongs to. + * + * Longest match wins, so a site covering both "crypto" and "cryptocurrency" + * attributes "cryptocurrency merchant account" to the more specific of the + * two rather than to whichever happens to sort first. Returns null when no + * subject claims it — the caller drops those, which is the same verdict the + * niche gate would reach a moment later. + */ +function attribute(keyword: string, masters: string[]): string | null { + let best: string | null = null; + let bestLen = 0; + const candidate = new Set(tokens(keyword)); + for (const master of masters) { + const masterTokens = tokens(master); + if (masterTokens.length === 0) continue; + const hit = masterTokens.filter((t) => candidate.has(t)); + if (hit.length === 0) continue; + const len = hit.join("").length; + if (len > bestLen) { + bestLen = len; + best = master; + } + } + return best; +} + function rankKeywords( rows: DfsKeywordRow[], boosts = new Map(), @@ -136,26 +170,22 @@ function rankKeywords( return scored.map((s) => s.row); } -function primarySeedQuery(site: SiteRow, seeds: string[]): string { - return ( - seeds[0] ?? - site.niche ?? - site.keywords?.[0] ?? - site.description ?? - site.domain ?? - "site keyword research" - ).trim(); -} - -function buildBuyerJourneyInput(site: SiteRow, seeds: string[]): BuyerJourneyKeywordInput { +function buildBuyerJourneyInput( + site: SiteRow, + masters: string[], + modifiers: string[], +): BuyerJourneyKeywordInput { const brand = site.domain?.replace(/^www\./, "") || site.niche || "the website"; const offer = [site.niche, site.description] .filter((s): s is string => !!s && s.trim().length > 0) .join(" — ") .slice(0, 900); return { - seedQuery: primarySeedQuery(site, seeds), - additionalSeeds: seeds.slice(1, 8), + // Every subject, not just the first one. The old code passed + // `seeds[0]` here and `seeds.slice(1, 8)` from an already-truncated + // list, which is how one subject came to own the entire model run. + seedQuery: masters.join(", "), + additionalSeeds: modifiers.slice(0, 8), offer: offer || brand, audience: site.target_audiences?.join(", ") || "the site's target customers", brand, @@ -182,21 +212,6 @@ function rowFromCandidate( }; } -function dedupeBySite( - candidates: DfsKeywordRow[], - existing: Set, -): DfsKeywordRow[] { - const out: DfsKeywordRow[] = []; - const seen = new Set(); - for (const r of candidates) { - const k = r.keyword.toLowerCase(); - if (existing.has(k) || seen.has(k)) continue; - seen.add(k); - out.push(r); - } - return out; -} - function scheduleKeywords( count: number, publishDays: number[], @@ -218,10 +233,39 @@ function scheduleKeywords( return dates; } +/** + * Interleave per-subject picks so the schedule alternates topics. + * + * Allocation decides *how many* rows each subject gets; this decides the order + * they are scheduled in, and the two are separate concerns. A fair allocation + * emitted subject-by-subject would still publish six consecutive peptide posts + * and then six consecutive casino ones — fair over a quarter, and visibly + * spammy over a fortnight. Round-robining the emission is what makes the fix + * legible to a reader of the blog rather than only to a reader of the database. + */ +function interleave(byMaster: Map): Candidate[] { + const queues = Array.from(byMaster.values()).map((c) => [...c]); + const out: Candidate[] = []; + let live = true; + while (live) { + live = false; + for (const queue of queues) { + const next = queue.shift(); + if (next) { + out.push(next); + live = true; + } + } + } + return out; +} + export type KeywordResearchResult = { ok: boolean; inserted: number; apiCost: number; + /** Rows allocated per subject — surfaced so a skewed queue is visible. */ + perMaster?: Record; error?: string; }; @@ -240,7 +284,7 @@ export async function researchKeywords( const { data: site } = await supabase .from("lx_site") .select( - "id, domain, niche, target_audiences, description, seed_keywords, keywords, competitors, tone, publish_days, publish_hour, daily_article_count", + "id, domain, niche, target_audiences, description, seed_keywords, master_keywords, modifiers, keywords, competitors, tone, publish_days, publish_hour, daily_article_count", ) .eq("id", siteId) .maybeSingle(); @@ -248,153 +292,255 @@ export async function researchKeywords( return { ok: false, inserted: 0, apiCost: 0, error: "site not found" }; } - const savedKeywords = (site.keywords ?? []) - .map(parseStoredKeyword) - .filter((r): r is DfsKeywordRow => !!r); - const seeds = buildSeeds(site); - if (savedKeywords.length === 0 && seeds.length === 0) { + const masters = resolveMasters(site); + const modifiers = resolveModifiers(site); + const anchors = anchorTokens(site); + + if (masters.length === 0) { return { ok: false, inserted: 0, apiCost: 0, - error: "add saved keywords or seed keywords first", + error: "add master keywords (the subjects this blog covers) first", + }; + } + // Refusing here rather than falling back to an unanchored expansion is the + // point. An anchorless run is exactly the run that produced the vendor + // articles, so it must be an error the operator sees and fixes, not a + // degraded mode that quietly publishes. + if (anchors.size === 0) { + return { + ok: false, + inserted: 0, + apiCost: 0, + error: + "set a niche or modifiers first — keywords are only researched crossed with what this site does, never on their own", }; } - // Skip keywords this site already has in active/history states. Failed - // rows are intentionally ignored here: an upstream outage should not - // permanently poison a topic and prevent the top-up sweep from - // refilling the queue. + // Existing rows serve two purposes: the duplicate fingerprints, and the + // per-subject coverage the allocator balances against. Failed rows are + // intentionally excluded from the duplicate set — an upstream outage should + // not permanently poison a topic — but they still count toward coverage, so + // a subject that keeps failing does not monopolise every top-up. const { data: existingRows } = await supabase .from("lx_keyword") - .select("keyword, status") + .select("keyword, status, master_keyword") .eq("site_id", site.id); - const existingSet = new Set( - (existingRows ?? []) - .filter((r: { status: string }) => r.status !== "failed") - .map((r: { keyword: string }) => r.keyword.toLowerCase()), - ); - const savedChosen = dedupeBySite(savedKeywords, existingSet).slice(0, TARGET_KEYWORDS); + const publishedSignatures = new Set(); + const coverage = new Map(); + for (const row of (existingRows ?? []) as Array<{ + keyword: string; + status: string; + master_keyword: string | null; + }>) { + if (row.status !== "failed") { + const sig = signature(row.keyword); + if (sig) publishedSignatures.add(sig); + } + // Attribute legacy rows (written before provenance existed) so the + // allocator sees the real history rather than treating a blog with + // twenty-three peptide posts as having no coverage at all. Without this + // the balancing would take a full cycle to notice the existing skew. + const master = row.master_keyword ?? attribute(row.keyword, masters); + if (master) { + const key = master.toLowerCase(); + coverage.set(key, (coverage.get(key) ?? 0) + 1); + } + } - // If the saved long-tail list does not fill the target queue, top it - // up using the same DataForSEO Labs endpoint + relevance gate used by - // the settings page's "Refetch keywords" flow. - const allRows: DfsKeywordRow[] = []; + const allocation = allocate(masters, coverage, TARGET_KEYWORDS); + + // ------------------------------------------------------------------ + // Candidate sources. All three are gated identically; they differ only + // in how they are obtained and how likely they are to be unavailable. + // ------------------------------------------------------------------ + const candidates: Candidate[] = []; const buyerJourneyBoosts = new Map(); let totalCost = 0; - const seedErrors: string[] = []; - if (savedChosen.length < TARGET_KEYWORDS && seeds.length > 0) { - if (deps.openai || deps.anthropic) { - try { - const buyerJourney = await generateBuyerJourneyKeywordOpportunities( - buildBuyerJourneyInput(site, seeds), - { - openai: deps.openai, - anthropic: deps.anthropic, - backendAiProvider: deps.backendAiProvider, - }, - ); - const candidates = flattenBuyerJourneyKeywords( - buyerJourney.output, - MAX_BUYER_JOURNEY_VOLUME_LOOKUP, - ); - const volumeKeywords = candidates.map((c) => c.keyword); - const volume = volumeKeywords.length > 0 - ? await dfs.searchVolume(volumeKeywords) - : { rows: [], cost: 0, taskId: null }; - totalCost += volume.cost; - if (volume.cost > 0 || volume.taskId) { - await supabase.from("lx_dataforseo_usage").insert({ - task_id: volume.taskId, - endpoint: "search_volume/live", - cost: volume.cost, - site_id: site.id, - }); - } - - const metricsByKeyword = new Map( - volume.rows.map((r) => [r.keyword.toLowerCase(), r]), - ); - for (const candidate of candidates) { - const metrics = metricsByKeyword.get(candidate.keyword.toLowerCase()); - const volumeValue = metrics?.search_volume ?? 0; - if ( - volumeValue >= MIN_BUYER_JOURNEY_VOLUME || - candidate.priority >= 4 - ) { - allRows.push(rowFromCandidate(candidate, metrics)); - buyerJourneyBoosts.set(candidate.keyword.toLowerCase(), { - priority: candidate.priority, - intent: candidate.intent, - clusterType: candidate.clusterType, - }); - } - } - } catch (err) { - seedErrors.push( - `buyer-journey model: ${err instanceof Error ? err.message : String(err)}`, - ); + const sourceErrors: string[] = []; + + // Source 0 — the floor. Subject × modifier, built locally from two columns + // the operator controls. No network, no failure mode, always on-niche by + // construction. Everything below is an improvement on this, never a + // prerequisite for it. + const crosses = crossQueries(masters, modifiers, CROSS_PER_MASTER); + for (const { master, query } of crosses) { + candidates.push({ + row: { + keyword: query, + search_volume: null, + competition: null, + competition_index: null, + cpc: null, + low_top_of_page_bid: null, + high_top_of_page_bid: null, + monthly_searches: null, + }, + master, + }); + } + + // Source 1 — hand-saved long-tail from the settings page. + for (const parsed of (site.keywords ?? []).map(parseStoredKeyword)) { + if (!parsed) continue; + const master = attribute(parsed.keyword, masters); + if (master) candidates.push({ row: parsed, master }); + } + + // Source 2 — the buyer-journey model, now seeded with every subject. + if (deps.openai || deps.anthropic) { + try { + const buyerJourney = await generateBuyerJourneyKeywordOpportunities( + buildBuyerJourneyInput(site, masters, modifiers), + { + openai: deps.openai, + anthropic: deps.anthropic, + backendAiProvider: deps.backendAiProvider, + }, + ); + const flattened = flattenBuyerJourneyKeywords( + buyerJourney.output, + MAX_BUYER_JOURNEY_VOLUME_LOOKUP, + ); + const volumeKeywords = flattened.map((c) => c.keyword); + const volume = volumeKeywords.length > 0 + ? await dfs.searchVolume(volumeKeywords) + : { rows: [], cost: 0, taskId: null }; + totalCost += volume.cost; + if (volume.cost > 0 || volume.taskId) { + await supabase.from("lx_dataforseo_usage").insert({ + task_id: volume.taskId, + endpoint: "search_volume/live", + cost: volume.cost, + site_id: site.id, + }); + } + + const metricsByKeyword = new Map( + volume.rows.map((r) => [r.keyword.toLowerCase(), r]), + ); + for (const candidate of flattened) { + const metrics = metricsByKeyword.get(candidate.keyword.toLowerCase()); + const volumeValue = metrics?.search_volume ?? 0; + if (volumeValue < MIN_BUYER_JOURNEY_VOLUME && candidate.priority < 4) continue; + const master = attribute(candidate.keyword, masters); + if (!master) continue; + candidates.push({ row: rowFromCandidate(candidate, metrics), master }); + buyerJourneyBoosts.set(candidate.keyword.toLowerCase(), { + priority: candidate.priority, + intent: candidate.intent, + clusterType: candidate.clusterType, + }); } + } catch (err) { + sourceErrors.push( + `buyer-journey model: ${err instanceof Error ? err.message : String(err)}`, + ); } + } - for (const seed of seeds.slice(0, 3)) { - try { - const result = await dfs.keywordIdeas([seed], { - limit: PER_SEED_LIMIT, + // Source 3 — DataForSEO expansion of the CROSSED phrases. + // + // One call carrying every cross, because keywordIdeas accepts 200 seeds. + // The bare subject is never sent: "peptide" on its own is what returned + // "skye peptides" and "pure peptide labs", and no downstream filter can + // reliably tell those from a keyword worth writing about. + if (crosses.length > 0) { + try { + const result = await dfs.keywordIdeas( + crosses.map((c) => c.query), + { + limit: IDEAS_LIMIT, minVolume: MIN_VOLUME, minWords: MIN_WORDS, closelyVariants: false, - }); - totalCost += result.cost; - await supabase.from("lx_dataforseo_usage").insert({ - task_id: result.taskId, - endpoint: "keyword_ideas/live", - cost: result.cost, - site_id: site.id, - }); - - const tokens = seedTokens(seed); - const relevant = tokens.length === 0 - ? result.rows - : result.rows.filter((r) => { - const kw = r.keyword.toLowerCase(); - return tokens.some((t) => kw.includes(t)); - }); - allRows.push(...relevant); - } catch (err) { - seedErrors.push( - `"${seed}": ${err instanceof Error ? err.message : String(err)}`, - ); + }, + ); + totalCost += result.cost; + await supabase.from("lx_dataforseo_usage").insert({ + task_id: result.taskId, + endpoint: "keyword_ideas/live", + cost: result.cost, + site_id: site.id, + }); + for (const row of result.rows) { + const master = attribute(row.keyword, masters); + if (master) candidates.push({ row, master }); } + } catch (err) { + sourceErrors.push( + `keyword ideas: ${err instanceof Error ? err.message : String(err)}`, + ); } } - const filtered = filterOutliers(allRows).filter( - (r) => { - const boost = buyerJourneyBoosts.get(r.keyword.toLowerCase()); - if (boost && boost.priority >= 4) return true; - return (r.search_volume ?? 0) >= MIN_VOLUME; - }, - ); - const ranked = rankKeywords(filtered, buyerJourneyBoosts); - const existingWithSaved = new Set(existingSet); - for (const r of savedChosen) existingWithSaved.add(r.keyword.toLowerCase()); - const researchedChosen = dedupeBySite(ranked, existingWithSaved).slice( - 0, - TARGET_KEYWORDS - savedChosen.length, + // ------------------------------------------------------------------ + // Gate, rank, allocate. + // ------------------------------------------------------------------ + const onNiche = candidates.filter((c) => isOnNiche(c.row.keyword, c.master, anchors)); + + // Volume filtering applies only to what came back from an API with a volume + // attached. The locally-built crosses have no volume by construction and + // must not be discarded for it — they are the floor that keeps the queue + // from emptying when everything upstream is unavailable. + const withVolume = onNiche.filter((c) => c.row.search_volume !== null); + const withoutVolume = onNiche.filter((c) => c.row.search_volume === null); + const volumeKept = filterOutliers(withVolume.map((c) => c.row)).filter((r) => { + const boost = buyerJourneyBoosts.get(r.keyword.toLowerCase()); + if (boost && boost.priority >= 4) return true; + return (r.search_volume ?? 0) >= MIN_VOLUME; + }); + const volumeKeptKeys = new Set(volumeKept.map((r) => r.keyword.toLowerCase())); + + const ranked = rankKeywords(volumeKept, buyerJourneyBoosts); + const rankIndex = new Map(ranked.map((r, i) => [r.keyword.toLowerCase(), i])); + + const survivors = [ + ...withVolume + .filter((c) => volumeKeptKeys.has(c.row.keyword.toLowerCase())) + .sort( + (a, b) => + (rankIndex.get(a.row.keyword.toLowerCase()) ?? Infinity) - + (rankIndex.get(b.row.keyword.toLowerCase()) ?? Infinity), + ), + // Crosses last within each subject: a real long-tail phrase with measured + // demand is a better article than a two-word construction, but the + // construction is a better article than nothing. + ...withoutVolume, + ]; + + const deduped = dropDuplicates( + survivors.map((c) => ({ ...c, keyword: c.row.keyword })), + publishedSignatures, ); - const chosen = [...savedChosen, ...researchedChosen].slice(0, TARGET_KEYWORDS); + + // Take each subject's allocated share, then interleave so the published + // sequence alternates subjects rather than running one to exhaustion. + const byMaster = new Map(); + for (const master of masters) byMaster.set(master, []); + for (const candidate of deduped) { + const bucket = byMaster.get(candidate.master); + if (!bucket) continue; + if (bucket.length >= (allocation.get(candidate.master) ?? 0)) continue; + bucket.push({ row: candidate.row, master: candidate.master }); + } + + // Subjects that could not fill their share hand it back, so a subject with + // no available candidates costs the run coverage rather than volume. + const chosen = interleave(byMaster).slice(0, TARGET_KEYWORDS); + if (chosen.length === 0) { - const details = seedErrors.length > 0 - ? ` Seed errors: ${seedErrors.join("; ")}` + const details = sourceErrors.length > 0 + ? ` Source errors: ${sourceErrors.join("; ")}` : ""; return { ok: false, inserted: 0, apiCost: totalCost, error: - "No new keyword candidates found. Saved keywords may already be published or queued; add new seed keywords/settings and try again." + + "No new keyword candidates found. Every candidate was already published or off-niche; add master keywords or modifiers and try again." + details, }; } @@ -405,7 +551,6 @@ export async function researchKeywords( // without first reshaping it. We re-introduce it when keyword overlap // across customers becomes measurable. - // Schedule them across publish_days. const slots = scheduleKeywords( chosen.length, site.publish_days, @@ -413,15 +558,16 @@ export async function researchKeywords( site.daily_article_count, ); - const insertRows = chosen.map((r, i) => ({ + const insertRows = chosen.map((c, i) => ({ site_id: site.id, - keyword: r.keyword, + keyword: c.row.keyword, + master_keyword: c.master, scheduled_for: slots[i]?.toISOString().slice(0, 10) ?? new Date(Date.now() + (i + 1) * 86400000).toISOString().slice(0, 10), status: "queued", source: "auto", - search_volume: r.search_volume, - cpc_usd: r.cpc, + search_volume: c.row.search_volume, + cpc_usd: c.row.cpc, })); const { error: insErr } = await supabase.from("lx_keyword").insert(insertRows); @@ -434,5 +580,10 @@ export async function researchKeywords( }; } - return { ok: true, inserted: insertRows.length, apiCost: totalCost }; + const perMaster: Record = {}; + for (const row of insertRows) { + perMaster[row.master_keyword] = (perMaster[row.master_keyword] ?? 0) + 1; + } + + return { ok: true, inserted: insertRows.length, apiCost: totalCost, perMaster }; } diff --git a/lib/lx/topicPlan.ts b/lib/lx/topicPlan.ts new file mode 100644 index 00000000..f836a154 --- /dev/null +++ b/lib/lx/topicPlan.ts @@ -0,0 +1,327 @@ +// Deciding which subjects an autoblog writes about, and in what proportion. +// +// This module exists because of a specific failure, and the shape of it is a +// direct response to that failure. coinpayportal.com — a crypto payment +// processor — published nineteen consecutive articles about peptide vendors: +// "skye peptides", "pure peptide labs", "wolverine stack peptides". Those are +// competitor storefronts in an industry the site *serves*, not one it is in. +// +// Two independent defects produced it, and fixing either alone would have left +// the other running. +// +// 1. **Truncation.** The site had ten subjects. The pipeline sliced them to +// five, then to three, and handed subject #1 to the buyer-journey model as +// its entire query. Five subjects had never produced a keyword in the +// site's lifetime. The top-up sweep re-ran the same truncated set every +// time the queue drained, so the concentration compounded rather than +// averaged out. +// +// 2. **An unanchored relevance gate.** A candidate was kept if it contained +// the seed token. Expanding the bare word "peptide" against a keyword +// tool returns the peptide industry's own vocabulary, and every one of +// those passed a test that only ever asked "is this about peptides?" — +// never "is this about what we sell to them?". +// +// The answer to both is the same: never expand a subject on its own. A subject +// is only ever researched *crossed with a modifier* — the tail terms that +// describe what this site actually does ("merchant account", "payment +// gateway"). "peptide" is not a topic. "peptide merchant account" is. +// +// That cross is also the floor. Every other source here can return nothing — +// the keyword API can be down, the model can refuse, the gate can reject +// everything — and the cross product still yields on-niche subjects, because +// it is built from two lists the operator controls rather than fetched. A +// blog that cannot reach any upstream still publishes, and still publishes +// about itself. Going dark and going off-topic are both failures; this +// prefers a smaller, correct queue to a full, spammy one. + +/** Subjects past this point can't be given a meaningful share of a 30-row target. */ +export const MAX_MASTERS = 12; + +/** + * Words that carry no topic and would pass any gate built on them. + * + * Deliberately not a general English stoplist: this is scoped to the words + * that appear in a *niche description* and a keyword phrase without narrowing + * either. "best" and "top" are here because they are the two most common + * prefixes in keyword-tool output and matching on them would re-admit the + * whole vendor-listicle class this module exists to reject. + */ +const STOPWORDS = new Set([ + "the", "and", "for", "with", "you", "your", "that", "this", "from", "into", + "over", "but", "not", "are", "was", "were", "has", "had", "have", "its", + "off", "out", "all", "any", "new", "get", "how", "why", "what", "who", + "best", "top", "guide", "list", "using", "about", "when", "where", "which", +]); + +/** + * A token worth matching on. + * + * Four characters because three-letter fragments ("pay", "ads", "seo") match + * inside unrelated words often enough to be worse than useless in a gate whose + * entire job is rejecting near-misses. + */ +const MIN_TOKEN_LEN = 4; + +/** + * Split a phrase into matchable tokens. + * + * Punctuation is dropped rather than split on, so "high-risk" yields "high" + * and "risk" — both of which are real narrowing terms for the site that wrote + * that niche, and neither of which survives a naive whitespace split. + */ +export function tokens(phrase: string): string[] { + return (phrase ?? "") + .toLowerCase() + .split(/[^a-z0-9]+/i) + .map((t) => t.trim()) + .filter((t) => t.length >= MIN_TOKEN_LEN && !STOPWORDS.has(t)); +} + +/** + * Reduce a token to a form that survives pluralisation. + * + * A crude suffix strip rather than a real stemmer, because the only job is + * collapsing "payment"/"payments" and "transaction"/"transactions" so the + * duplicate check can see that "peptide payments" and "peptide payment" are + * the same article. A real stemmer would be a dependency and a behaviour + * change in the gate, for a class of match this never needs to make. + */ +function stem(token: string): string { + if (token.length > 4 && token.endsWith("ies")) return `${token.slice(0, -3)}y`; + if (token.length > 4 && token.endsWith("es")) return token.slice(0, -2); + if (token.length > 3 && token.endsWith("s")) return token.slice(0, -1); + return token; +} + +/** + * An order-independent fingerprint of what a keyword is about. + * + * Sorted, so "merchant account peptide" and "peptide merchant account" collide + * — they would produce the same article, and a blog publishing both is the + * duplicate-content problem this is here to prevent. + */ +export function signature(keyword: string): string { + return Array.from(new Set(tokens(keyword).map(stem))).sort().join(" "); +} + +type SiteTopicFields = { + master_keywords?: string[] | null; + seed_keywords?: string[] | null; + modifiers?: string[] | null; + niche?: string | null; +}; + +/** + * The durable subject list for a site. + * + * Falls back to `seed_keywords` for any site the backfill has not reached, so + * this is safe to deploy ahead of the migration rather than after it. Capped + * on read as well as by the database constraint: a list that arrives oversized + * from anywhere must still be allocated over, not rejected at publish time. + */ +export function resolveMasters(site: SiteTopicFields): string[] { + const raw = (site.master_keywords?.length ? site.master_keywords : site.seed_keywords) ?? []; + const seen = new Set(); + const out: string[] = []; + for (const entry of raw) { + const trimmed = (entry ?? "").trim(); + if (!trimmed) continue; + const key = trimmed.toLowerCase(); + if (seen.has(key)) continue; + seen.add(key); + out.push(trimmed); + } + return out.slice(0, MAX_MASTERS); +} + +/** + * The tail terms that anchor a subject to this site's own business. + * + * Thirteen of seventeen live sites have an empty `modifiers` column, so the + * niche is mined for them rather than treating the empty case as + * "no anchoring required" — that reading is precisely how an unanchored + * expansion gets to run, and it is the thing being fixed. A site with neither + * modifiers nor a niche gets an empty list, and `crossQueries` then declines + * to produce anything, which surfaces as an actionable "set a niche" error + * instead of silently publishing vendor listicles. + */ +export function resolveModifiers(site: SiteTopicFields): string[] { + const explicit = (site.modifiers ?? []) + .map((m) => (m ?? "").trim()) + .filter((m) => m.length > 0); + if (explicit.length > 0) return explicit.slice(0, 20); + return Array.from(new Set(tokens(site.niche ?? ""))).slice(0, 8); +} + +/** + * Every token that means "this keyword is about our business". + * + * The union of the modifiers and the niche, because the two are populated + * inconsistently across live sites and a candidate matching either is anchored + * in the sense the gate cares about. + */ +export function anchorTokens(site: SiteTopicFields): Set { + const out = new Set(); + for (const modifier of resolveModifiers(site)) { + for (const token of tokens(modifier)) out.add(stem(token)); + } + for (const token of tokens(site.niche ?? "")) out.add(stem(token)); + return out; +} + +/** + * Is this candidate about one of our subjects *and* about what we do? + * + * Both halves are required, and that conjunction is the whole fix. The old + * gate asked only the first question, which is why nineteen articles about + * other people's peptide shops passed it. + * + * @param keyword the candidate + * @param master the subject it was researched for + * @param anchors from `anchorTokens` + */ +export function isOnNiche( + keyword: string, + master: string, + anchors: Set, +): boolean { + const candidate = new Set(tokens(keyword).map(stem)); + if (candidate.size === 0) return false; + + const masterTokens = tokens(master).map(stem); + const onSubject = masterTokens.length === 0 + ? true + : masterTokens.some((t) => candidate.has(t)); + if (!onSubject) return false; + + // An anchorless site cannot answer the second question, and answering it + // "yes" by default would restore the old behaviour exactly. Answering "no" + // would empty every queue on the platform. Neither is acceptable, so the + // caller is required to supply anchors and this asserts rather than guesses. + if (anchors.size === 0) return false; + + for (const token of candidate) { + if (anchors.has(token)) return true; + } + return false; +} + +/** + * The subjects to research, each already crossed with a narrowing term. + * + * Ordered subject-major — every subject's first cross comes before any + * subject's second — so that a caller which runs out of API budget partway + * through has still touched every subject rather than exhausting the first + * one. That ordering is the difference between a budget cut costing depth and + * costing coverage, and coverage is what was broken. + * + * A cross whose modifier tokens are already in the subject is skipped: + * "crypto" × "crypto payments" would otherwise research "crypto crypto + * payments". + */ +export function crossQueries( + masters: string[], + modifiers: string[], + perMaster = 3, +): Array<{ master: string; query: string }> { + const out: Array<{ master: string; query: string }> = []; + if (masters.length === 0 || modifiers.length === 0) return out; + + const seen = new Set(); + for (let depth = 0; depth < perMaster; depth += 1) { + for (const master of masters) { + const masterTokens = new Set(tokens(master).map(stem)); + // Each subject walks the modifier list from a different offset, so the + // narrowing terms are spread across subjects instead of every subject + // getting the same first modifier and the tail never being used. + const offset = masters.indexOf(master); + let taken = 0; + for (let i = 0; i < modifiers.length && taken <= depth; i += 1) { + const modifier = modifiers[(i + offset) % modifiers.length]; + if (tokens(modifier).every((t) => masterTokens.has(stem(t)))) continue; + if (taken < depth) { + taken += 1; + continue; + } + const query = `${master} ${modifier}`.replace(/\s+/g, " ").trim(); + const key = signature(query); + if (key && !seen.has(key)) { + seen.add(key); + out.push({ master, query }); + } + taken += 1; + } + } + } + return out; +} + +/** + * How many new keywords each subject should get. + * + * Fair share of the remaining target, but weighted toward whatever is *behind* + * — a subject with no coverage is filled before one that already has twenty + * articles. Without that, a queue topped up repeatedly stays in whatever + * proportion it started in, and the site that published nineteen peptide posts + * would go on publishing them at exactly the rate it always had. + * + * @param masters the subject list + * @param coverage existing keyword count per subject (any status) + * @param target how many rows to allocate in total + */ +export function allocate( + masters: string[], + coverage: Map, + target: number, +): Map { + const out = new Map(); + if (masters.length === 0 || target <= 0) return out; + + for (const m of masters) out.set(m, 0); + + // Hand out one row at a time to whichever subject is furthest behind, + // counting what has already been handed out in this pass. An O(target × + // masters) loop over at most 30 × 12, which is not worth a heap to avoid and + // is far easier to prove correct than an apportionment formula. + for (let i = 0; i < target; i += 1) { + let pick = masters[0]; + let lowest = Infinity; + for (const master of masters) { + const total = (coverage.get(master.toLowerCase()) ?? 0) + (out.get(master) ?? 0); + if (total < lowest) { + lowest = total; + pick = master; + } + } + out.set(pick, (out.get(pick) ?? 0) + 1); + } + return out; +} + +/** + * Drop candidates that would produce an article the blog already has. + * + * Compares fingerprints rather than strings, so the pluralised and reordered + * restatements of an existing post are caught. This is the direct answer to + * "spamming blogs with same content": the old dedupe compared lowercased + * keywords exactly, which let "peptide payments" and "peptide payment" + * both through, and both were published — nine days apart, in May. + * + * @param candidates in preference order + * @param published fingerprints already on the blog, from `signature` + */ +export function dropDuplicates( + candidates: T[], + published: Set, +): T[] { + const out: T[] = []; + const seen = new Set(published); + for (const candidate of candidates) { + const key = signature(candidate.keyword); + if (!key || seen.has(key)) continue; + seen.add(key); + out.push(candidate); + } + return out; +} diff --git a/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql b/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql new file mode 100644 index 00000000..7f764ed4 --- /dev/null +++ b/supabase/migrations/20260828120000_autoblog_master_keywords_and_ads.sql @@ -0,0 +1,82 @@ +-- Master keywords, topic provenance, and the network opt-in. +-- +-- Three columns, one root cause. +-- +-- `seed_keywords` had become two lists wearing one name: the durable set of +-- subjects a blog is *about*, and the working set the research pipeline was +-- allowed to chew on. Nothing enforced a size on it, so it grew — ten entries +-- on coinpayportal — and the pipeline quietly truncated it twice on the way +-- through (`buildSeeds` to five, the DataForSEO loop to three). The seeds past +-- the cut had never produced a single keyword: thirty-two of them, across nine +-- sites. Splitting the durable list out under its own name and capping it is +-- what makes that class of bug impossible to reintroduce, because a list a +-- human can hold in their head is one the planner can afford to cover *all* of. +-- +-- `master_keyword` on lx_keyword is the provenance the planner needs to do +-- that. Fair allocation across topics is not expressible without knowing which +-- topic each queued row came from, and inferring it after the fact by +-- substring-matching the keyword back to a seed is exactly the loose matching +-- that let "skye peptides" through the relevance gate in the first place. +-- +-- `ads_enabled` is the network opt-in: house ads and partner guest posts in +-- published articles. Default true, because a network everybody has to be +-- asked to join is one that stays empty. + +alter table public.lx_site + add column if not exists master_keywords text[] not null default '{}', + add column if not exists ads_enabled boolean not null default true; + +comment on column public.lx_site.master_keywords is + 'The durable 3-12 subjects this blog covers. The keyword planner allocates its target evenly across ALL of these; nothing truncates it. Crossed with `modifiers` to stay anchored to the site''s own niche.'; + +comment on column public.lx_site.ads_enabled is + 'Opted into the CrawlProof network: house/partner ad units and guest posts are placed in published articles. Default true.'; + +-- A cap the planner can honour rather than a limit it has to work around. +-- Twelve is the point past which an even allocation of a 30-keyword target +-- stops giving each topic enough rows to be worth researching separately. +do $$ +begin + if not exists ( + select 1 from pg_constraint where conname = 'lx_site_master_keywords_len' + ) then + alter table public.lx_site + add constraint lx_site_master_keywords_len + check (coalesce(array_length(master_keywords, 1), 0) <= 12); + end if; +end $$; + +-- Which master subject a queued keyword was researched for. +-- +-- Nullable, and stays nullable: every row written before this migration has no +-- honest answer, and guessing one would corrupt the very coverage figures the +-- planner reads. Those rows are simply invisible to the fair-share maths, which +-- is the correct behaviour — they are already published or queued, so the +-- topics they belong to need no further filling on their account. +alter table public.lx_keyword + add column if not exists master_keyword text; + +comment on column public.lx_keyword.master_keyword is + 'The lx_site.master_keywords entry this keyword was researched for. Null on rows predating topic provenance; those are excluded from coverage rather than guessed at.'; + +create index if not exists lx_keyword_site_master_idx + on public.lx_keyword (site_id, master_keyword); + +-- Seed the durable list from what each site already had. +-- +-- First ten only, mirroring the cap, and only where the column is still empty +-- so a re-run cannot stomp a hand-curated list. This is the one place the old +-- truncation is preserved on purpose: a site with more than ten seeds had no +-- coverage of the tail anyway, and the operator should choose which ten matter +-- rather than have position in an unordered array decide it. +update public.lx_site +set master_keywords = ( + select coalesce(array_agg(s order by ord), '{}') + from ( + select s, ord from unnest(seed_keywords) with ordinality as t(s, ord) + where length(trim(s)) > 0 + limit 10 + ) picked +) +where coalesce(array_length(master_keywords, 1), 0) = 0 + and coalesce(array_length(seed_keywords, 1), 0) > 0; diff --git a/tests/lx/topic-plan.test.ts b/tests/lx/topic-plan.test.ts new file mode 100644 index 00000000..48a86f1e --- /dev/null +++ b/tests/lx/topic-plan.test.ts @@ -0,0 +1,226 @@ +// The peptide regression, pinned. +// +// Every fixture here is real: the master list, the modifiers and the niche are +// coinpayportal.com's live values, and the rejected keywords are titles the +// site actually published in June and July 2026. If these pass, the specific +// nineteen articles that prompted this work cannot be generated again. + +import { describe, expect, it } from "vitest"; +import { + allocate, + anchorTokens, + crossQueries, + dropDuplicates, + isOnNiche, + MAX_MASTERS, + resolveMasters, + resolveModifiers, + signature, +} from "@/lib/lx/topicPlan"; + +// coinpayportal.com, exactly as stored. +const SITE = { + master_keywords: [ + "peptide", "crypto", "blockchain", "cryptocurrency", "casino", + "marijuana", "dispensary", "weed", "iptv", "torrents", + ], + seed_keywords: [ + "peptide", "crypto", "blockchain", "cryptocurrency", "casino", + "marijuana", "dispensary", "weed", "iptv", "torrents", + ], + modifiers: [ + "payments", "transactions", "merchant account", + "payment gateway", "payment processing", + ], + niche: "crypto payments for high-risk merchants", +}; + +describe("resolveMasters", () => { + it("keeps every subject — the truncation that caused the outage is gone", () => { + // The old buildSeeds() ended in .slice(0, 5) and the expansion loop then + // took .slice(0, 3). marijuana, dispensary, weed, iptv and torrents had + // never produced a single keyword in the site's lifetime. + const masters = resolveMasters(SITE); + expect(masters).toHaveLength(10); + for (const forgotten of ["marijuana", "dispensary", "weed", "iptv", "torrents"]) { + expect(masters).toContain(forgotten); + } + }); + + it("falls back to seed_keywords so it is safe to deploy before the backfill", () => { + expect(resolveMasters({ seed_keywords: ["alpha", "beta"] })).toEqual(["alpha", "beta"]); + }); + + it("caps an oversized list rather than rejecting it", () => { + const many = Array.from({ length: 40 }, (_, i) => `subject${i}`); + expect(resolveMasters({ master_keywords: many })).toHaveLength(MAX_MASTERS); + }); + + it("drops duplicates and blanks without shifting the rest", () => { + expect(resolveMasters({ master_keywords: ["crypto", " ", "Crypto", "casino"] })) + .toEqual(["crypto", "casino"]); + }); +}); + +describe("resolveModifiers", () => { + it("prefers the explicit column", () => { + expect(resolveModifiers(SITE)).toContain("merchant account"); + }); + + it("mines the niche when the column is empty — 13 of 17 live sites", () => { + const derived = resolveModifiers({ + modifiers: [], + niche: "security operations and threat detection", + }); + expect(derived).toEqual( + expect.arrayContaining(["security", "operations", "threat", "detection"]), + ); + // "and" carries no narrowing and would weaken the gate. + expect(derived).not.toContain("and"); + }); +}); + +describe("isOnNiche — the gate that was missing", () => { + const anchors = anchorTokens(SITE); + + // These are real published titles. Each is a peptide *vendor* — a + // competitor storefront in an industry coinpayportal sells payment + // processing TO, not one it writes about. + it.each([ + "skye peptides", + "pure peptide labs", + "wolverine stack peptides", + "apex peptides", + "biotech peptides", + "melanotan 2 peptides", + "ghk peptide", + "aod 9604 peptide", + "peptide crafters", + "lab 34 peptides and proteins", + ])("rejects the vendor term %j", (keyword) => { + expect(isOnNiche(keyword, "peptide", anchors)).toBe(false); + }); + + // These are the May 2026 posts — the period before the regression, when + // the pipeline still crossed seeds with modifiers. + it.each([ + "peptide merchant account", + "peptide payment processing", + "peptide payment gateway", + "peptide payments", + "peptide transactions", + ])("keeps the on-niche keyword %j", (keyword) => { + expect(isOnNiche(keyword, "peptide", anchors)).toBe(true); + }); + + it("requires the subject as well as the anchor", () => { + // Anchored, but about a subject this blog does not cover. + expect(isOnNiche("plumbing merchant account", "peptide", anchors)).toBe(false); + }); + + it("refuses everything when a site has no anchor at all", () => { + // Defaulting to "allow" here would restore the exact behaviour that + // published the vendor articles, so an anchorless site must fail closed + // and let the caller raise it. + expect(isOnNiche("peptide merchant account", "peptide", new Set())).toBe(false); + }); +}); + +describe("crossQueries", () => { + const crosses = crossQueries(resolveMasters(SITE), resolveModifiers(SITE), 3); + + it("gives every subject coverage, including the five that never had any", () => { + const covered = new Set(crosses.map((c) => c.master)); + expect(covered.size).toBe(10); + for (const forgotten of ["marijuana", "dispensary", "weed", "iptv", "torrents"]) { + expect(covered.has(forgotten)).toBe(true); + } + }); + + it("never emits a bare subject — that is what returned the vendor terms", () => { + for (const { master, query } of crosses) { + expect(query).not.toBe(master); + expect(query.length).toBeGreaterThan(master.length); + } + }); + + it("orders subject-major so a truncated run loses depth, not coverage", () => { + // The first ten entries must be ten *different* subjects: a budget cut + // partway through should still have touched everything. + const firstPass = crosses.slice(0, 10).map((c) => c.master); + expect(new Set(firstPass).size).toBe(10); + }); + + it("skips a cross whose modifier is already in the subject", () => { + const out = crossQueries(["crypto"], ["crypto", "payments"], 2); + expect(out.map((c) => c.query)).not.toContain("crypto crypto"); + expect(out.map((c) => c.query)).toContain("crypto payments"); + }); + + it("produces nothing without modifiers, rather than expanding bare", () => { + expect(crossQueries(["peptide"], [], 3)).toEqual([]); + }); +}); + +describe("allocate", () => { + it("fills the subjects that are behind first", () => { + // The live skew: 23 peptide articles, 11 crypto, nothing else. + const coverage = new Map([["peptide", 23], ["crypto", 11]]); + const out = allocate(resolveMasters(SITE), coverage, 30); + + expect(out.get("peptide") ?? 0).toBe(0); + expect(out.get("crypto") ?? 0).toBe(0); + // Everything with no history gets a share. + for (const starved of ["marijuana", "dispensary", "weed", "iptv", "torrents"]) { + expect(out.get(starved) ?? 0).toBeGreaterThan(0); + } + }); + + it("splits evenly when nothing has history", () => { + const out = allocate(["a", "b", "c"], new Map(), 30); + expect([out.get("a"), out.get("b"), out.get("c")]).toEqual([10, 10, 10]); + }); + + it("allocates exactly the target", () => { + const masters = resolveMasters(SITE); + const total = Array.from(allocate(masters, new Map(), 30).values()) + .reduce((a, b) => a + b, 0); + expect(total).toBe(30); + }); + + it("is a no-op for an empty subject list", () => { + expect(allocate([], new Map(), 30).size).toBe(0); + }); +}); + +describe("signature + dropDuplicates", () => { + it("collapses the plural restatement that shipped twice", () => { + // Both published, nine days apart, in May 2026. + expect(signature("peptide payments")).toBe(signature("peptide payment")); + }); + + it("collapses a reordering", () => { + expect(signature("merchant account peptide")).toBe(signature("peptide merchant account")); + }); + + it("keeps genuinely different subjects apart", () => { + expect(signature("peptide payments")).not.toBe(signature("casino payments")); + }); + + it("drops candidates already on the blog", () => { + const published = new Set([signature("peptide payments")]); + const out = dropDuplicates( + [{ keyword: "peptide payment" }, { keyword: "casino payments" }], + published, + ); + expect(out.map((c) => c.keyword)).toEqual(["casino payments"]); + }); + + it("drops duplicates within the candidate list itself", () => { + const out = dropDuplicates( + [{ keyword: "iptv payments" }, { keyword: "iptv payment" }], + new Set(), + ); + expect(out).toHaveLength(1); + }); +}); From d2ab769262933ab1c877496a171ad9e3170139e8 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 28 Aug 2026 15:06:19 +0000 Subject: [PATCH 2/5] One tick joins the network, and it starts ticked MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Published articles now carry an ad unit and a short list of related posts — partner blogs from the exchange, plus real posts from the RSS Amplifier directory so the block is not visibly the same three domains every time. Splitting "show ads" from "join the link network" would produce four states, two of them incoherent (take partner backlinks, refuse to give them) and all four needing a human to reason about. So it is one column, ads_enabled, defaulting true. Nobody is asked. A network everybody has to opt into is one that stays empty. The slot is provisioned rather than requested. ad_slots.status defaults to 'inactive' and serveAd() returns null for a non-active slot before the house-ad fallback, so a slot created at the default renders an empty div for ever and looks exactly like a broken embed. These are created active, once per project, reused on redelivery. The block is built at delivery, not generation, because the hosting site is only known once a guest post's target is resolved — and it is the host's readers who see the ad and the host's owner who opted in, not the author's. Titles and links come from RSS feeds crawled off the open web and land in HTML on a customer's domain. Everything interpolated is escaped, hrefs are allowlisted to http(s), and directory links are nofollow ugc: those publishers agreed to nothing, and passing them ranking signal would be spending someone else's reputation. Partner links stay followed — that reciprocity is the point of the exchange and is recorded on both sides. A failure anywhere in here costs the block, never the article. Publishing without an ad costs an impression; failing to publish costs the customer the thing they pay for. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_013cbcXdpLFUgJZuVfCB4Gzv --- .../projects/[id]/autoblog/setup/form.tsx | 72 +++++- .../projects/[id]/autoblog/setup/page.tsx | 2 +- app/actions/linkExchange.ts | 18 ++ lib/lx/feedTopics.ts | 86 ++++++- lib/lx/networkBlock.ts | 229 ++++++++++++++++++ lib/lx/webhookDeliver.ts | 58 ++++- tests/lx/network-block.test.ts | 133 ++++++++++ 7 files changed, 590 insertions(+), 8 deletions(-) create mode 100644 lib/lx/networkBlock.ts create mode 100644 tests/lx/network-block.test.ts diff --git a/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx b/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx index e21e7329..5ab4bc9f 100644 --- a/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx +++ b/app/(app)/dashboard/projects/[id]/autoblog/setup/form.tsx @@ -23,8 +23,10 @@ type Existing = { target_audiences: string[]; description: string; seed_keywords: string[]; + master_keywords: string[]; modifiers: string[]; preserve_keywords: boolean; + ads_enabled: boolean; keywords: string[]; seo_title: string | null; seo_description: string | null; @@ -109,6 +111,14 @@ export function SetupForm({ (initial?.target_audiences ?? []).join(", "), ); const [description, setDescription] = useState(initial?.description ?? ""); + const [masterKeywords, setMasterKeywords] = useState( + // Falls back to the seed list for any site the backfill has not reached, + // so the field is never blank on a project that already has subjects. + (initial?.master_keywords?.length + ? initial.master_keywords + : (initial?.seed_keywords ?? []) + ).join(", "), + ); const [seedKeywords, setSeedKeywords] = useState( (initial?.seed_keywords ?? []).join(", "), ); @@ -118,6 +128,11 @@ export function SetupForm({ const [preserveKeywords, setPreserveKeywords] = useState( initial?.preserve_keywords ?? false, ); + // Default ON, matching the column default. A new project is in the network + // unless its owner takes the tick out. + const [adsEnabled, setAdsEnabled] = useState( + initial?.ads_enabled ?? true, + ); const [keywords, setKeywords] = useState( (initial?.keywords ?? []).join("\n"), ); @@ -425,6 +440,13 @@ export function SetupForm({ // Each line is a CSV row ",". For seed-building and // dedupe we only care about the keyword half (everything before the // first comma). + function masterKeywordsAsArray(): string[] { + return masterKeywords + .split(",") + .map((s) => s.trim()) + .filter(Boolean); + } + function keywordsAsArray(): string[] { return keywords .split("\n") @@ -569,9 +591,11 @@ export function SetupForm({ niche, targetAudiences: audiences, description, + masterKeywords, seedKeywords, modifiers, preserveKeywords, + adsEnabled, keywords, seoTitle, seoDescription, @@ -843,6 +867,34 @@ export function SetupForm({ onChange={(e) => setDescription(e.target.value)} /> +
+ + setMasterKeywords(e.target.value)} + /> +

+ The subjects this blog covers, and the list it always rebuilds from. + The planner splits every research run evenly across{" "} + all of them and fills whichever is furthest behind + first, so a subject cannot run away with the schedule. Each one is + researched crossed with a modifier below — never on its own.{" "} + {masterKeywordsAsArray().length > 12 ? ( + + {masterKeywordsAsArray().length} entered — only the first 12 are + kept. + + ) : ( + <>{masterKeywordsAsArray().length} subject(s). + )} +

+
+
+
+ +
+
diff --git a/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx b/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx index fa990710..b5201701 100644 --- a/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx +++ b/app/(app)/dashboard/projects/[id]/autoblog/setup/page.tsx @@ -7,7 +7,7 @@ import { SetupForm } from "./form"; export const metadata = { title: "Autoblog · Setup" }; const SITE_COLUMNS = - "id, domain, blog_root_url, sitemap_url, niche, target_audiences, description, seed_keywords, modifiers, preserve_keywords, keywords, seo_title, seo_description, tone, competitors, webhook_url, webhook_secret, daily_article_count, publish_days, publish_hour, internal_links_per_article, backlinks_enabled, external_links_per_article, banner_style, status"; + "id, domain, blog_root_url, sitemap_url, niche, target_audiences, description, seed_keywords, master_keywords, modifiers, preserve_keywords, ads_enabled, keywords, seo_title, seo_description, tone, competitors, webhook_url, webhook_secret, daily_article_count, publish_days, publish_hour, internal_links_per_article, backlinks_enabled, external_links_per_article, banner_style, status"; export default async function AutoblogSetupPage({ params, diff --git a/app/actions/linkExchange.ts b/app/actions/linkExchange.ts index 514ff120..1d109971 100644 --- a/app/actions/linkExchange.ts +++ b/app/actions/linkExchange.ts @@ -289,6 +289,9 @@ export type SiteInput = { niche: string; targetAudiences: string; description: string; + // Comma-separated 3-12 subjects this blog covers. The durable list the + // keyword planner allocates evenly across; see lib/lx/topicPlan. + masterKeywords: string; // Comma-separated 1-3 word head terms for DataForSEO expansion. seedKeywords: string; // Comma-separated tail terms ("payments", "merchant account") that the @@ -297,6 +300,9 @@ export type SiteInput = { // When true, Refetch flows skip overwriting the keywords text[]. Used // after a hand-curated build via seeds × modifiers. preserveKeywords: boolean; + // Network opt-in: ad units and partner/directory links in published + // articles. Default true — see the migration note on lx_site.ads_enabled. + adsEnabled: boolean; // Comma-separated. parseList() into a text[]. keywords: string; seoTitle: string; @@ -360,6 +366,12 @@ export async function createOrUpdateSite( const seedKeywords = parseList(input.seedKeywords, 50, 40); const modifiers = parseList(input.modifiers, 20, 40); const preserveKeywords = !!input.preserveKeywords; + // The DB constraint caps this at 12; parse to the same number so an + // oversized paste is trimmed here rather than rejected as a write error the + // form cannot explain. + const masterKeywords = parseList(input.masterKeywords, 12, MAX.keyword); + // Absent reads as opted in, matching the column default. + const adsEnabled = input.adsEnabled !== false; // Keywords use parseKeywordRows so each row keeps its `,` hint. // Allow up to ~80 chars per row to fit "keyword phrase here,12345". const keywords = parseKeywordRows(input.keywords, MAX.keywords, MAX.keyword + 20); @@ -433,6 +445,8 @@ export async function createOrUpdateSite( target_audiences: audiences, description, seed_keywords: seedKeywords, + master_keywords: masterKeywords, + ads_enabled: adsEnabled, modifiers, preserve_keywords: preserveKeywords, keywords, @@ -489,6 +503,8 @@ export async function createOrUpdateSite( target_audiences: audiences, description, seed_keywords: seedKeywords, + master_keywords: masterKeywords, + ads_enabled: adsEnabled, modifiers, preserve_keywords: preserveKeywords, keywords, @@ -534,6 +550,8 @@ export async function createOrUpdateSite( target_audiences: audiences, description, seed_keywords: seedKeywords, + master_keywords: masterKeywords, + ads_enabled: adsEnabled, modifiers, preserve_keywords: preserveKeywords, keywords, diff --git a/lib/lx/feedTopics.ts b/lib/lx/feedTopics.ts index 80215bb9..6d67d574 100644 --- a/lib/lx/feedTopics.ts +++ b/lib/lx/feedTopics.ts @@ -62,7 +62,26 @@ export function topicSlug(keyword: string): string { * @returns post titles, channel title and sponsored items removed */ export function itemTitles(xml: string): string[] { - const out: string[] = []; + return itemEntries(xml).map((e) => e.title); +} + +/** A usable post from a topic feed. */ +export type FeedEntry = { title: string; link: string | null }; + +/** + * Posts in a topic feed, with their links. + * + * The link is what a *citation* needs; `itemTitles` only ever needed the + * subject, and is now a projection of this. Keeping one parser means the + * sponsored-item exclusion below cannot be enforced on one path and forgotten + * on the other — and the path that carries links out to a published page is + * precisely the one where forgetting it would be worst. + * + * @param xml an RSS document + * @returns entries, channel-level and sponsored items removed + */ +export function itemEntries(xml: string): FeedEntry[] { + const out: FeedEntry[] = []; // Item blocks only — this is what keeps the channel's own (the name // of the topic) out of the candidate list. @@ -84,12 +103,75 @@ export function itemTitles(xml: string): string[] { if (/\(sponsored\)\s*$/i.test(title)) continue; if (title.length < MIN_TITLE_LEN || title.length > MAX_TITLE_LEN) continue; - out.push(title); + + const href = decodeXml(item.match(/<link>([\s\S]*?)<\/link>/i)?.[1] ?? "").trim(); + // Only absolute http(s) links survive. A feed carrying a relative link, a + // javascript: URL or a bare guid must not be able to put either into an + // anchor on a customer's published page. + const link = /^https?:\/\/\S+$/i.test(href) ? href : null; + + out.push({ title, link }); } return out; } +/** + * Real posts from the directory on the subjects given, with links. + * + * Unlike `subjectFromTopicFeeds`, which wants one subject to write about, this + * wants several posts to cite — so it reads across every requested topic + * rather than stopping at the first that answers, and returns only entries + * that carry a usable link. + * + * @param keywords subject words, tried in random order + * @param limit most entries to return + * @param fetchImpl injected by the tests + */ +export async function postsFromTopicFeeds( + keywords: string[], + limit = 3, + fetchImpl: typeof fetch = fetch, +): Promise<Array<{ title: string; link: string; topic: string }>> { + const slugs = shuffle( + Array.from(new Set((keywords ?? []).map(topicSlug).filter(Boolean))), + ).slice(0, MAX_FEEDS); + + const out: Array<{ title: string; link: string; topic: string }> = []; + const seen = new Set<string>(); + + for (const slug of slugs) { + if (out.length >= limit) break; + let xml: string; + try { + const res = await fetchImpl( + `${RSSAMPLIFIER}/topics/${encodeURIComponent(slug)}.rss`, + { + signal: AbortSignal.timeout(TIMEOUT_MS), + headers: { accept: "application/rss+xml, application/xml;q=0.9" }, + }, + ); + if (!res.ok) continue; + xml = await res.text(); + } catch { + continue; + } + + // One entry per topic before taking a second from any of them, so a block + // of three citations shows three subjects rather than three posts from + // whichever feed happened to be longest. + const entries = shuffle(itemEntries(xml).filter((e) => e.link)); + for (const entry of entries) { + if (!entry.link || seen.has(entry.link)) continue; + seen.add(entry.link); + out.push({ title: entry.title, link: entry.link, topic: slug.replace(/-/g, " ") }); + break; + } + } + + return out.slice(0, limit); +} + /** * Undo the escaping a feed document applies, and nothing else. * diff --git a/lib/lx/networkBlock.ts b/lib/lx/networkBlock.ts new file mode 100644 index 00000000..ba231be4 --- /dev/null +++ b/lib/lx/networkBlock.ts @@ -0,0 +1,229 @@ +// What a published article carries besides the article: an ad unit, and links +// out to the rest of the network. +// +// Both are governed by one column — `lx_site.ads_enabled`, default true — and +// that is deliberate. Splitting "show ads" from "join the link network" into +// two switches produces four states, two of which are incoherent (take +// backlinks from partners, refuse to give them) and all four of which need a +// human to reason about. One switch, on by default, is the whole opt-in. +// +// Three things this is careful about. +// +// **The slot is provisioned, not requested.** A blog with no ad slot used to +// mean a support conversation. Slots are created here, active, on first +// delivery — `ad_slots.status` defaults to 'inactive' and `serveAd()` returns +// null before the house-ad fallback for a non-active slot, so a slot created +// at the default would render an empty div on every article forever and look +// exactly like a broken embed. Provisioning it inactive would be worse than +// not provisioning it at all. +// +// **Everything interpolated is escaped.** The titles and links in the +// insertion block come from other people's RSS feeds. They are written into +// HTML that lands on a customer's domain, which makes this an injection sink +// with a hostile upstream, and the fact that the immediate source is our own +// directory changes nothing about that — the directory is a crawl of the open +// web. +// +// **A failure here costs the block, never the article.** Every lookup is +// wrapped and degrades to "no block". A post that publishes without an ad unit +// has cost us an impression; a post that fails to publish because the ad +// lookup threw has cost the customer the thing they are paying for. + +import type { SupabaseClient } from "@supabase/supabase-js"; +import { postsFromTopicFeeds } from "./feedTopics"; + +/** Format asked of the slot. 728x90 is the in-article leaderboard. */ +const AD_FORMAT = "banner_728x90"; + +/** Most partner links in one insertion. */ +const MAX_PARTNER_LINKS = 3; + +/** Most directory posts in one insertion. */ +const MAX_FEED_LINKS = 3; + +/** + * Escape text for HTML interpolation. + * + * Ampersand first, or the escapes introduced by the later replacements get + * double-escaped. Quotes are included because these values are also written + * into attributes. + */ +export function escapeHtml(value: string): string { + return String(value ?? "") + .replace(/&/g, "&") + .replace(/</g, "<") + .replace(/>/g, ">") + .replace(/"/g, """) + .replace(/'/g, "'"); +} + +/** + * Is this a link we are willing to put on a customer's page? + * + * An allowlist of two schemes rather than a denylist of the dangerous ones: + * `javascript:`, `data:` and `vbscript:` are the ones anybody thinks to block, + * and the list of what else a browser will execute is not one to maintain by + * hand. + */ +export function isSafeHref(href: string): boolean { + try { + const url = new URL(href); + return url.protocol === "https:" || url.protocol === "http:"; + } catch { + return false; + } +} + +export type NetworkLink = { title: string; url: string; source: "partner" | "directory" }; + +/** + * The slot this project's articles should fill. + * + * Reuses an existing active slot before creating one, so re-delivering an + * article — or publishing the second post on a blog — does not mint a second + * slot and split the site's reporting across two rows. + * + * @returns a slot id, or null when one could not be resolved or created + */ +export async function resolveAdSlot( + supabase: SupabaseClient<any>, + projectId: string, + ownerId: string, + niche: string | null, +): Promise<string | null> { + try { + const { data: existing } = await supabase + .from("ad_slots") + .select("id") + .eq("project_id", projectId) + .eq("status", "active") + .limit(1) + .maybeSingle<{ id: string }>(); + if (existing?.id) return existing.id; + + const { data: created } = await supabase + .from("ad_slots") + .insert({ + project_id: projectId, + owner_id: ownerId, + placement: "inline", + formats: [AD_FORMAT, "banner_300x250", "text_link"], + niche, + // Explicit, against the column default. See the note at the top of + // this file: an inactive slot is indistinguishable from a broken one. + status: "active", + }) + .select("id") + .maybeSingle<{ id: string }>(); + return created?.id ?? null; + } catch { + return null; + } +} + +/** + * The ad unit markup. + * + * `data-cp-ad` carries no `data-slot` in the server-rendered case elsewhere in + * the codebase to avoid a fill race; here there is no client to race with — + * this HTML is delivered to a third-party blog and rendered as-is — so the + * slot travels on the element and ad.js's own DOMContentLoaded pass fills it. + */ +export function adUnitHtml(slotId: string, origin: string): string { + const slot = escapeHtml(slotId); + const src = `${origin.replace(/\/$/, "")}/ad.js`; + return [ + `<div data-cp-ad data-slot="${slot}" data-format="${AD_FORMAT}"></div>`, + `<script async src="${escapeHtml(src)}"></script>`, + ].join("\n"); +} + +/** + * The "elsewhere in the network" block. + * + * Partner articles first, directory posts after: a partner opted into the + * exchange and gets the more valuable position, while the directory posts are + * what keep the block from being visibly the same three domains on every + * article — which is the shape that gets a link network discounted. + * + * Directory links are `rel="nofollow ugc"`. They are not exchange partners and + * have not agreed to anything; passing them ranking signal would be us + * spending someone else's reputation. Partner links are followed, because that + * reciprocity is the entire point of the exchange and is recorded on both + * sides in lx_backlink. + */ +export function networkLinksHtml(links: NetworkLink[]): string { + const safe = links.filter((l) => isSafeHref(l.url)); + if (safe.length === 0) return ""; + + const items = safe + .map((link) => { + const rel = link.source === "directory" + ? ' rel="nofollow ugc noopener"' + : ' rel="noopener"'; + return ` <li><a href="${escapeHtml(link.url)}"${rel}>${escapeHtml(link.title)}</a></li>`; + }) + .join("\n"); + + return [ + `<aside class="cp-network-links" data-cp-network>`, + ` <h2>Elsewhere on this topic</h2>`, + ` <ul>`, + items, + ` </ul>`, + `</aside>`, + ].join("\n"); +} + +/** + * Everything appended to an article, for a site that is in the network. + * + * @param supabase service client — this runs from the delivery path, no session + * @param site the blog that will HOST the post (the guest-post target, when + * the article is one), because it is that site's readers who see the + * block and that site's owner who opted in + * @param topics subjects to pull directory posts for + * @param partnerLinks already-ranked exchange candidates from the caller + * @returns HTML to append, or "" when the site is opted out or nothing resolved + */ +export async function buildNetworkBlock( + supabase: SupabaseClient<any>, + site: { + id: string; + project_id: string; + user_id: string; + niche: string | null; + ads_enabled?: boolean | null; + }, + topics: string[], + partnerLinks: NetworkLink[], + origin: string, +): Promise<string> { + // Absent column reads as opted in, matching the migration default, so this + // behaves the same before and after the schema lands. + if (site.ads_enabled === false) return ""; + + const parts: string[] = []; + + const feedLinks: NetworkLink[] = await postsFromTopicFeeds(topics, MAX_FEED_LINKS) + .then((posts) => + posts.map((p) => ({ title: p.title, url: p.link, source: "directory" as const })), + ) + .catch(() => []); + + const block = networkLinksHtml([ + ...partnerLinks.slice(0, MAX_PARTNER_LINKS), + ...feedLinks, + ]); + if (block) parts.push(block); + + const slotId = await resolveAdSlot( + supabase, + site.project_id, + site.user_id, + site.niche, + ); + if (slotId) parts.push(adUnitHtml(slotId, origin)); + + return parts.join("\n"); +} diff --git a/lib/lx/webhookDeliver.ts b/lib/lx/webhookDeliver.ts index 49bf3cf8..fb876632 100644 --- a/lib/lx/webhookDeliver.ts +++ b/lib/lx/webhookDeliver.ts @@ -14,6 +14,8 @@ import type { SupabaseClient } from "@supabase/supabase-js"; import { buildEvent, sendWebhook, type Post } from "@profullstack/autoblog"; import { env } from "../env"; +import { findExchangeCandidates } from "./exchangeMatcher"; +import { buildNetworkBlock } from "./networkBlock"; type ArticleRow = { id: string; @@ -44,6 +46,13 @@ type SiteRow = { webhook_secret: string | null; author_name: string | null; author_url: string | null; + // Network opt-in, resolved for the site that will HOST the post. For a guest + // post that is the partner, not the author: it is the host's readers who see + // the ad unit and the host's owner who agreed to carry it. + project_id: string; + user_id: string; + niche: string | null; + ads_enabled: boolean | null; }; /** @@ -109,7 +118,7 @@ export type DeliveryResult = { error?: string; }; -function articleToPost(article: ArticleRow, site: SiteRow): Post { +function articleToPost(article: ArticleRow, site: SiteRow, networkHtml = ""): Post { const blogRoot = site.blog_root_url.replace(/\/$/, ""); const url = `${blogRoot}/${article.slug}`; @@ -140,7 +149,13 @@ function articleToPost(article: ArticleRow, site: SiteRow): Post { title: article.title, slug: article.slug, excerpt: article.excerpt || article.meta_description || null, - html: `${jsonLd}\n${article.content_html}`, + // The network block goes after the article and outside the JSON-LD, so an + // ad unit and a list of other people's links are never described to a + // crawler as part of this article's body. + html: [jsonLd, article.content_html, networkHtml].filter(Boolean).join("\n"), + // Markdown deliberately does NOT carry the block. A receiver rendering the + // markdown path would have to trust our HTML through its own sanitiser, + // and the ad unit needs a script tag that no markdown renderer will emit. markdown: article.content_markdown, status: "published", published_at: publishedAt, @@ -187,7 +202,7 @@ export async function deliverArticle( const { data: site } = await supabase .from("lx_site") .select( - "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url", + "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url, project_id, user_id, niche, ads_enabled", ) .eq("id", deliveryTargetId) .maybeSingle<SiteRow>(); @@ -208,7 +223,42 @@ export async function deliverArticle( }; } - const post = articleToPost(claimed, site); + // Ads and partner links for the hosting site, if it is in the network. + // + // Built here rather than at generation time because the host is only known + // once the guest-post target has been resolved, and because a redelivery + // should carry a current block rather than one frozen weeks ago. The whole + // thing is best-effort: `buildNetworkBlock` swallows its own failures, and + // this catch covers the rest, because an article that publishes without an + // ad has cost an impression while an article that fails to publish has cost + // the customer the thing they pay for. + const partnerLinks = await findExchangeCandidates(supabase, { + selfSiteId: site.id, + selfNiche: site.niche, + keyword: (claimed.tags ?? []).join(" ") || claimed.title, + slots: 3, + }) + .then((r) => + r.candidates.map((c) => ({ + title: c.title, + url: c.url, + source: "partner" as const, + })), + ) + .catch(() => []); + + const networkHtml = await buildNetworkBlock( + supabase, + site, + claimed.tags ?? [], + partnerLinks, + env.siteUrl, + ).catch((err: unknown) => { + console.warn("[lx] network block failed:", err); + return ""; + }); + + const post = articleToPost(claimed, site, networkHtml); // Reuse the saved delivery id on retries so receivers idempotently // dedupe. SDK uses event.id as the webhook-id header. const event = buildEvent(post, { diff --git a/tests/lx/network-block.test.ts b/tests/lx/network-block.test.ts new file mode 100644 index 00000000..fad6b1f4 --- /dev/null +++ b/tests/lx/network-block.test.ts @@ -0,0 +1,133 @@ +// The insertion block writes other people's text into a customer's page. +// +// Most of what follows is injection and link-safety, because that is what this +// module actually risks. The titles come from RSS feeds crawled off the open +// web; the fact that they reach us via our own directory does not make them +// ours, and the page they land on belongs to somebody paying us. + +import { describe, expect, it } from "vitest"; +import { + adUnitHtml, + escapeHtml, + isSafeHref, + networkLinksHtml, + type NetworkLink, +} from "@/lib/lx/networkBlock"; +import { itemEntries } from "@/lib/lx/feedTopics"; + +describe("escapeHtml", () => { + it("neutralises a script tag in a feed title", () => { + expect(escapeHtml('<script>alert(1)</script>')).toBe( + "<script>alert(1)</script>", + ); + }); + + it("escapes the ampersand first so nothing is double-escaped", () => { + // "<" must not become "&lt;" — which is what a naive ordering does. + expect(escapeHtml("<")).toBe("<"); + expect(escapeHtml("&")).toBe("&"); + expect(escapeHtml("&<")).toBe("&<"); + }); + + it("escapes both quote characters, since these land in attributes", () => { + expect(escapeHtml(`" onload="x`)).toBe("" onload="x"); + expect(escapeHtml("' onload='x")).toBe("' onload='x"); + }); +}); + +describe("isSafeHref", () => { + it.each(["javascript:alert(1)", "data:text/html,<script>", "vbscript:msgbox", "file:///etc/passwd"])( + "rejects %j", + (href) => expect(isSafeHref(href)).toBe(false), + ); + + it.each(["https://example.com/post", "http://example.com/post"])( + "allows %j", + (href) => expect(isSafeHref(href)).toBe(true), + ); + + it("rejects a relative link rather than emitting a broken anchor", () => { + expect(isSafeHref("/relative/path")).toBe(false); + }); +}); + +describe("networkLinksHtml", () => { + const links: NetworkLink[] = [ + { title: "Partner post", url: "https://partner.example/a", source: "partner" }, + { title: "Directory post", url: "https://stranger.example/b", source: "directory" }, + ]; + + it("follows partner links and nofollows directory ones", () => { + const html = networkLinksHtml(links); + // The partner opted into an exchange; the stranger did not, and passing + // them ranking signal would be spending someone else's reputation. + expect(html).toMatch(/href="https:\/\/partner\.example\/a" rel="noopener"/); + expect(html).toMatch(/href="https:\/\/stranger\.example\/b" rel="nofollow ugc noopener"/); + }); + + it("drops an unsafe link instead of rendering the whole block", () => { + const html = networkLinksHtml([ + ...links, + { title: "Bad", url: "javascript:alert(1)", source: "directory" }, + ]); + expect(html).not.toContain("javascript:"); + expect(html).toContain("partner.example"); + }); + + it("escapes a hostile title", () => { + const html = networkLinksHtml([ + { + title: '</a><script>alert(document.cookie)</script>', + url: "https://example.com/x", + source: "directory", + }, + ]); + expect(html).not.toContain("<script>"); + expect(html).toContain("<script>"); + }); + + it("returns nothing at all when every link was unsafe", () => { + // An empty <ul> under a heading reads as a broken feature. + expect(networkLinksHtml([{ title: "x", url: "javascript:1", source: "directory" }])).toBe(""); + }); + + it("returns nothing for an empty list", () => { + expect(networkLinksHtml([])).toBe(""); + }); +}); + +describe("adUnitHtml", () => { + it("carries the slot on the element so ad.js can fill it", () => { + const html = adUnitHtml("slot-123", "https://crawlproof.com"); + expect(html).toContain('data-cp-ad data-slot="slot-123"'); + expect(html).toContain('src="https://crawlproof.com/ad.js"'); + }); + + it("does not double the slash when the origin has a trailing one", () => { + expect(adUnitHtml("s", "https://crawlproof.com/")).toContain("https://crawlproof.com/ad.js"); + }); +}); + +describe("itemEntries", () => { + const feed = `<rss><channel><title>topic — RSS Amplifier + A perfectly ordinary post about build systemshttps://good.example/1 + SponsoredOur own advertisement, laundered as editorialhttps://ad.example/2 + Another post that is long enough to qualify (sponsored)https://ad.example/3 + Relative links must not become anchors here/relative + `; + + it("excludes sponsored items by category and by title suffix", () => { + const titles = itemEntries(feed).map((e) => e.title); + expect(titles).toEqual(["A perfectly ordinary post about build systems", "Relative links must not become anchors here"]); + }); + + it("keeps an absolute link and nulls a relative one", () => { + const entries = itemEntries(feed); + expect(entries[0].link).toBe("https://good.example/1"); + expect(entries[1].link).toBeNull(); + }); + + it("ignores the channel title", () => { + expect(itemEntries(feed).map((e) => e.title)).not.toContain("topic — RSS Amplifier"); + }); +}); From d7a87c1a85f94098582875dfd2ef4268371a529f Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 28 Aug 2026 15:10:28 +0000 Subject: [PATCH 3/5] Daemonize the feed crawl, and put a status page on it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The autoblog cites real posts from RSS Amplifier topic feeds. Those were fetched live, inside article delivery, on a three-second timeout — a third-party HTTP call on the critical path of the one operation a customer pays for. And the source was invisible: a topic feed going missing showed up as an empty block and nothing else, indistinguishable from a topic nobody had configured. Both have the same fix. The worker now reads feeds on its own 30-minute tick (each source at most every 6h, 25 per sweep, least-recently-fetched first, so work per tick is fixed however many subjects the platform grows to), caches what came back, and keeps the record of every attempt. Delivery reads rows. A cache miss costs one article its citation block rather than blocking a publish on somebody else's server. The source list is derived from every active site's master keywords on each sweep, never curated. A blog that adds a subject has that feed crawled on the next tick with nobody filing a request — which is what "no human intervention" actually requires, since a curated list is a queue of requests waiting for somebody. That is also why the status page has no "add a feed" button: it would imply a step that does not exist. Two judgements worth keeping. A 200 carrying no items is not a success — it is what the directory returns for a topic nobody publishes under, and counting it as one parks an empty topic at the front of the least-recently-fetched queue for ever, crowding out real feeds. And a source that gives up is parked, not deleted: a deleted row is re-derived from the same master keyword on the very next sweep and starts failing again, which is an infinite retry wearing the costume of a clean table. The page says "stalled" rather than naming a cause, because it cannot tell a stopped worker from an unreachable directory and those have different fixes. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_013cbcXdpLFUgJZuVfCB4Gzv --- app/(app)/dashboard/autoblog/crawler/page.tsx | 168 +++++++++++ app/api/lx/feed-crawl/status/route.ts | 20 ++ lib/lx/feedCrawl.ts | 285 ++++++++++++++++++ lib/lx/feedCrawlStats.ts | 119 ++++++++ lib/lx/networkBlock.ts | 7 +- .../20260828140000_lx_feed_sources.sql | 66 ++++ tests/lx/feed-crawl.test.ts | 50 +++ worker/index.ts | 35 +++ 8 files changed, 748 insertions(+), 2 deletions(-) create mode 100644 app/(app)/dashboard/autoblog/crawler/page.tsx create mode 100644 app/api/lx/feed-crawl/status/route.ts create mode 100644 lib/lx/feedCrawl.ts create mode 100644 lib/lx/feedCrawlStats.ts create mode 100644 supabase/migrations/20260828140000_lx_feed_sources.sql create mode 100644 tests/lx/feed-crawl.test.ts diff --git a/app/(app)/dashboard/autoblog/crawler/page.tsx b/app/(app)/dashboard/autoblog/crawler/page.tsx new file mode 100644 index 00000000..f056fa4f --- /dev/null +++ b/app/(app)/dashboard/autoblog/crawler/page.tsx @@ -0,0 +1,168 @@ +// Live view of the feed crawler, modelled on rssamplifier.com/crawlstats. +// +// The autoblog cites real posts from RSS Amplifier topic feeds, and until the +// crawl was daemonized that source was invisible: a topic going missing looked +// identical to a topic nobody had configured. This is where that difference +// becomes visible. +// +// Deliberately read-only and unconfigurable. The feed list is derived from +// every active site's master keywords on each sweep, so there is nothing here +// to add or remove — a subject added to a blog appears within a tick. An +// "add a feed" button would imply a curation step that does not exist, and +// would create a queue of requests waiting for a human, which is the thing +// the automation is meant to remove. + +import { serviceClient } from "@/lib/supabase/service"; +import { ago, loadCrawlerStats } from "@/lib/lx/feedCrawlStats"; + +// Always fresh: a status page served from a cache reports the cache's health, +// not the crawler's, and would go on saying "healthy" through an outage. +export const dynamic = "force-dynamic"; + +function Stat({ + label, + value, + note, +}: { + label: string; + value: string; + note?: string; +}) { + return ( +
+
{value}
+
+ {label} +
+ {note ? ( +
{note}
+ ) : null} +
+ ); +} + +export default async function CrawlerStatusPage() { + const { stats, sources } = await loadCrawlerStats(serviceClient()); + const now = Date.now(); + + return ( +
+

Crawler status

+

+ Live view of the daemon that reads the RSS Amplifier topic feeds the + autoblog cites from. The feed list is derived from every active + site's master keywords — nothing here is curated by hand. The same + numbers are available as{" "} + + JSON + + . +

+ + {stats.stalled ? ( +
+ Stalled. {stats.dueNow.toLocaleString()} feed(s) are + due and nothing has been read successfully since{" "} + {ago(stats.lastSuccessAt, now)}. Either the worker is not running or + the directory is unreachable — this page cannot tell which, and the + two have different fixes. +
+ ) : null} + +
+ + + + +
+ +

+ Last successful fetch {ago(stats.lastSuccessAt, now)} ·{" "} + {stats.neverFetched.toLocaleString()} never fetched · generated{" "} + {stats.generatedAt} +

+ +

Feeds

+

+ One row per subject any active blog covers. A feed is re-read at most + every 6 hours, least-recently-fetched first, 25 per sweep — so the work + per tick is fixed however many subjects the platform grows to. +

+ + {sources.length === 0 ? ( +

+ No feeds yet. The first sweep derives them from the active sites' + master keywords; if this stays empty, no site has master keywords set. +

+ ) : ( +
+ + + + + + + + + + + + + {sources.map((source) => ( + + + + + + + + + ))} + +
TopicStatusLast readLast successItemsLast error
+ + {source.topic} + + + {source.status === "given_up" ? ( + + given up + + ) : source.consecutive_failures > 0 ? ( + failing ({source.consecutive_failures}) + ) : ( + active + )} + {ago(source.last_fetch_at, now)} + {ago(source.last_success_at, now)} + + {source.item_count.toLocaleString()} + + {source.last_error ?? ""} +
+
+ )} +
+ ); +} diff --git a/app/api/lx/feed-crawl/status/route.ts b/app/api/lx/feed-crawl/status/route.ts new file mode 100644 index 00000000..d53cbdba --- /dev/null +++ b/app/api/lx/feed-crawl/status/route.ts @@ -0,0 +1,20 @@ +import { NextResponse } from "next/server"; +import { serviceClient } from "@/lib/supabase/service"; +import { loadCrawlerStats } from "@/lib/lx/feedCrawlStats"; + +export const runtime = "nodejs"; +// A cached status response reports the cache's health rather than the +// crawler's, and would keep answering "fine" straight through an outage. +export const dynamic = "force-dynamic"; + +/** + * Counts only. + * + * The per-feed rows stay on the dashboard page behind a session: this is the + * shape a monitor polls, and it has no reason to carry the error strings some + * upstream server wrote. + */ +export async function GET() { + const { stats } = await loadCrawlerStats(serviceClient()); + return NextResponse.json(stats, { headers: { "cache-control": "no-store" } }); +} diff --git a/lib/lx/feedCrawl.ts b/lib/lx/feedCrawl.ts new file mode 100644 index 00000000..f20c4ae5 --- /dev/null +++ b/lib/lx/feedCrawl.ts @@ -0,0 +1,285 @@ +// The daemon side of the directory feeds. +// +// `postsFromTopicFeeds` used to fetch RSS live, inside article delivery, on a +// three-second budget. That is a third-party HTTP call on the critical path of +// the one operation a customer pays for, and it made the source invisible: a +// topic feed going missing showed up as an empty block and nothing else. +// +// This moves the fetching onto the worker's own schedule and leaves a record +// of every attempt. Delivery then reads rows. The two halves fail +// independently — a dead directory ages the cache instead of slowing down a +// publish, and a publish never waits on anybody else's server. +// +// The source list is derived on every sweep from the master_keywords of every +// active site. Nobody maintains it. A blog that adds a subject has that +// subject's feed crawled on the next tick, which is the behaviour "no human +// intervention" actually requires — a curated feed list is a queue of requests +// waiting for somebody. + +import type { SupabaseClient } from "@supabase/supabase-js"; +import { itemEntries } from "./feedTopics"; +import { resolveMasters, tokens } from "./topicPlan"; + +const RSSAMPLIFIER = "https://rssamplifier.com"; + +/** Per-feed budget. Generous next to delivery's, because nothing waits on it. */ +const FETCH_TIMEOUT_MS = 12_000; + +/** Feeds read per sweep. Bounded so one tick cannot run for an hour. */ +const PER_SWEEP = 25; + +/** Don't re-read a feed more often than this. */ +export const REFRESH_AFTER_MS = 6 * 60 * 60 * 1000; // 6h + +/** + * Consecutive failures before a source is parked. + * + * Parked, not deleted. A deleted row is re-derived on the very next sweep from + * the same master keyword and starts failing again immediately — an infinite + * retry wearing the costume of a clean table. The row stays so the status page + * can say "this topic has no feed, here is why". + */ +const GIVE_UP_AFTER = 8; + +/** Items kept per source. Older ones are pruned so the table stays bounded. */ +const KEEP_PER_SOURCE = 40; + +/** + * The directory's slug for a subject. + * + * Mirrors `topicSlug` in feedTopics but drops the stopword-free tokenisation + * through `tokens` first, so a master keyword of "merchant account payments" + * slugs to something the directory actually has a topic for rather than a long + * phrase that will always 404. + */ +export function topicFor(master: string): string | null { + const parts = tokens(master); + if (parts.length === 0) { + // A short subject ("iptv", "weed") has no tokens over the length floor but + // is still a real topic — fall back to the raw slug rather than dropping + // exactly the short, high-value subjects. + const raw = (master ?? "").toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, ""); + return raw || null; + } + return parts[0]; +} + +export type FeedSweepResult = { + sources: number; + fetched: number; + succeeded: number; + newItems: number; + gaveUp: number; +}; + +/** + * Make sure every subject any active site covers has a source row. + * + * Insert-only, ignoring conflicts: a topic already present keeps its history, + * and two sites sharing a subject share one crawl. + */ +export async function syncFeedSources( + supabase: SupabaseClient, +): Promise { + const { data: sites } = await supabase + .from("lx_site") + .select("master_keywords, seed_keywords") + .eq("status", "active"); + + const topics = new Set(); + for (const site of (sites ?? []) as Array<{ + master_keywords: string[] | null; + seed_keywords: string[] | null; + }>) { + for (const master of resolveMasters(site)) { + const topic = topicFor(master); + if (topic) topics.add(topic); + } + } + if (topics.size === 0) return 0; + + const rows = Array.from(topics).map((topic) => ({ + topic, + url: `${RSSAMPLIFIER}/topics/${encodeURIComponent(topic)}.rss`, + })); + + const { error } = await supabase + .from("lx_feed_source") + .upsert(rows, { onConflict: "topic", ignoreDuplicates: true }); + if (error) console.warn("[lx] feed source sync:", error.message); + + return topics.size; +} + +/** + * Read the feeds that are due. + * + * Least-recently-fetched first, capped, so the sweep is a fixed amount of work + * regardless of how many subjects the platform covers. + */ +export async function crawlFeeds( + supabase: SupabaseClient, + fetchImpl: typeof fetch = fetch, +): Promise { + const sources = await syncFeedSources(supabase); + const due = new Date(Date.now() - REFRESH_AFTER_MS).toISOString(); + + const { data: batch } = await supabase + .from("lx_feed_source") + .select("id, topic, url, consecutive_failures") + .eq("status", "active") + .or(`last_fetch_at.is.null,last_fetch_at.lt.${due}`) + .order("last_fetch_at", { ascending: true, nullsFirst: true }) + .limit(PER_SWEEP); + + const result: FeedSweepResult = { + sources, + fetched: 0, + succeeded: 0, + newItems: 0, + gaveUp: 0, + }; + + for (const source of (batch ?? []) as Array<{ + id: string; + topic: string; + url: string; + consecutive_failures: number; + }>) { + result.fetched += 1; + const now = new Date().toISOString(); + + let status: number | null = null; + let entries: Array<{ title: string; link: string | null }> = []; + let error: string | null = null; + + try { + const res = await fetchImpl(source.url, { + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), + headers: { accept: "application/rss+xml, application/xml;q=0.9" }, + }); + status = res.status; + if (res.ok) { + entries = itemEntries(await res.text()).filter((e) => e.link); + } else { + error = `HTTP ${res.status}`; + } + } catch (err) { + error = err instanceof Error ? err.message : String(err); + } + + // A 200 carrying no items is not a success. It is what the directory + // returns for a topic nobody publishes under, and counting it as one + // would let a permanently empty topic sit at the front of the + // least-recently-fetched queue for ever, crowding out real feeds. + const ok = error === null && entries.length > 0; + if (!ok && !error) error = "no usable items"; + + if (ok) { + result.succeeded += 1; + const rows = entries.slice(0, KEEP_PER_SOURCE).map((e) => ({ + source_id: source.id, + title: e.title, + link: e.link as string, + })); + const { data: inserted } = await supabase + .from("lx_feed_item") + .upsert(rows, { onConflict: "source_id,link", ignoreDuplicates: true }) + .select("id"); + result.newItems += inserted?.length ?? 0; + await pruneItems(supabase, source.id); + } + + const failures = ok ? 0 : source.consecutive_failures + 1; + const givingUp = failures >= GIVE_UP_AFTER; + if (givingUp) result.gaveUp += 1; + + await supabase + .from("lx_feed_source") + .update({ + last_fetch_at: now, + ...(ok ? { last_success_at: now, item_count: entries.length } : {}), + last_status: status, + last_error: ok ? null : error, + consecutive_failures: failures, + status: givingUp ? "given_up" : "active", + updated_at: now, + }) + .eq("id", source.id); + } + + return result; +} + +/** Keep the newest KEEP_PER_SOURCE items and drop the rest. */ +async function pruneItems(supabase: SupabaseClient, sourceId: string): Promise { + const { data: keep } = await supabase + .from("lx_feed_item") + .select("id") + .eq("source_id", sourceId) + .order("first_seen_at", { ascending: false }) + .limit(KEEP_PER_SOURCE); + const ids = (keep ?? []).map((r: { id: string }) => r.id); + if (ids.length < KEEP_PER_SOURCE) return; + await supabase + .from("lx_feed_item") + .delete() + .eq("source_id", sourceId) + .not("id", "in", `(${ids.join(",")})`); +} + +/** + * Cached posts for a set of subjects. + * + * The read side of the crawl, used by article delivery. Returns nothing rather + * than falling back to a live fetch: a cache miss here means the sweep has not + * reached that topic yet, and the correct behaviour is an article without a + * citation block, not a publish that blocks on somebody else's server. The + * next article gets the block. + */ +export async function cachedFeedPosts( + supabase: SupabaseClient, + masters: string[], + limit = 3, +): Promise> { + const topics = Array.from( + new Set(masters.map(topicFor).filter((t): t is string => !!t)), + ); + if (topics.length === 0) return []; + + const { data } = await supabase + .from("lx_feed_item") + .select("title, link, source:lx_feed_source!inner(topic)") + .in("source.topic", topics) + .order("first_seen_at", { ascending: false }) + .limit(limit * 12); + + type Row = { title: string; link: string; source: { topic: string } | { topic: string }[] }; + const byTopic = new Map>(); + for (const row of (data ?? []) as Row[]) { + const source = Array.isArray(row.source) ? row.source[0] : row.source; + const topic = source?.topic; + if (!topic) continue; + const bucket = byTopic.get(topic) ?? []; + bucket.push({ title: row.title, link: row.link, topic }); + byTopic.set(topic, bucket); + } + + // One per topic before a second from any, so three citations show three + // subjects rather than three posts from whichever feed is busiest. + const out: Array<{ title: string; link: string; topic: string }> = []; + const queues = Array.from(byTopic.values()); + let live = true; + while (live && out.length < limit) { + live = false; + for (const queue of queues) { + if (out.length >= limit) break; + const next = queue.shift(); + if (next) { + out.push(next); + live = true; + } + } + } + return out; +} diff --git a/lib/lx/feedCrawlStats.ts b/lib/lx/feedCrawlStats.ts new file mode 100644 index 00000000..745bbcad --- /dev/null +++ b/lib/lx/feedCrawlStats.ts @@ -0,0 +1,119 @@ +// The numbers behind the crawler status page. +// +// Kept out of the page so the HTML view and the JSON endpoint cannot drift — +// a status page and a machine-readable feed of the same status that disagree +// are worse than having only one of them. + +import type { SupabaseClient } from "@supabase/supabase-js"; +import { REFRESH_AFTER_MS } from "./feedCrawl"; + +export type FeedSourceRow = { + id: string; + topic: string; + url: string; + status: string; + last_fetch_at: string | null; + last_success_at: string | null; + last_status: number | null; + last_error: string | null; + item_count: number; + consecutive_failures: number; +}; + +export type CrawlerStats = { + sources: number; + active: number; + gaveUp: number; + dueNow: number; + neverFetched: number; + erroring: number; + items: number; + newItems24h: number; + lastSuccessAt: string | null; + /** + * The daemon looks stopped. + * + * Sources are due and nothing has succeeded in a while. Stated as a + * suspicion rather than a fact because this cannot distinguish "the worker + * is down" from "the directory is down" — and the operator's next step + * differs, so claiming to know which would send them the wrong way. + */ + stalled: boolean; + generatedAt: string; +}; + +/** Nothing succeeded in this long, with work waiting, reads as stalled. */ +const STALL_AFTER_MS = 2 * 60 * 60 * 1000; // 2h — four missed 30-min ticks + +export async function loadCrawlerStats( + supabase: SupabaseClient, +): Promise<{ stats: CrawlerStats; sources: FeedSourceRow[] }> { + const now = Date.now(); + const dueBefore = new Date(now - REFRESH_AFTER_MS).toISOString(); + const since24h = new Date(now - 24 * 60 * 60 * 1000).toISOString(); + + const [{ data: sources }, itemCount, new24h] = await Promise.all([ + supabase + .from("lx_feed_source") + .select( + "id, topic, url, status, last_fetch_at, last_success_at, last_status, last_error, item_count, consecutive_failures", + ) + .order("last_fetch_at", { ascending: true, nullsFirst: true }) + .limit(500), + supabase + .from("lx_feed_item") + .select("id", { count: "exact", head: true }) + .then((r) => r.count ?? 0), + supabase + .from("lx_feed_item") + .select("id", { count: "exact", head: true }) + .gte("first_seen_at", since24h) + .then((r) => r.count ?? 0), + ]); + + const rows = (sources ?? []) as FeedSourceRow[]; + const active = rows.filter((r) => r.status === "active"); + const dueNow = active.filter( + (r) => !r.last_fetch_at || r.last_fetch_at < dueBefore, + ).length; + + const lastSuccessAt = rows + .map((r) => r.last_success_at) + .filter((v): v is string => !!v) + .sort() + .at(-1) ?? null; + + const stalled = + dueNow > 0 && + (!lastSuccessAt || now - Date.parse(lastSuccessAt) > STALL_AFTER_MS); + + return { + stats: { + sources: rows.length, + active: active.length, + gaveUp: rows.filter((r) => r.status === "given_up").length, + dueNow, + neverFetched: rows.filter((r) => !r.last_fetch_at).length, + erroring: rows.filter((r) => r.consecutive_failures > 0).length, + items: itemCount, + newItems24h: new24h, + lastSuccessAt, + stalled, + generatedAt: new Date(now).toISOString(), + }, + sources: rows, + }; +} + +/** "15m ago", "3h ago", "never" — the only formatting this page needs. */ +export function ago(iso: string | null, now = Date.now()): string { + if (!iso) return "never"; + const ms = now - Date.parse(iso); + if (!Number.isFinite(ms) || ms < 0) return "just now"; + const mins = Math.floor(ms / 60000); + if (mins < 1) return "just now"; + if (mins < 60) return `${mins}m ago`; + const hours = Math.floor(mins / 60); + if (hours < 24) return `${hours}h ago`; + return `${Math.floor(hours / 24)}d ago`; +} diff --git a/lib/lx/networkBlock.ts b/lib/lx/networkBlock.ts index ba231be4..13928d72 100644 --- a/lib/lx/networkBlock.ts +++ b/lib/lx/networkBlock.ts @@ -30,7 +30,7 @@ // lookup threw has cost the customer the thing they are paying for. import type { SupabaseClient } from "@supabase/supabase-js"; -import { postsFromTopicFeeds } from "./feedTopics"; +import { cachedFeedPosts } from "./feedCrawl"; /** Format asked of the slot. 728x90 is the in-article leaderboard. */ const AD_FORMAT = "banner_728x90"; @@ -205,7 +205,10 @@ export async function buildNetworkBlock( const parts: string[] = []; - const feedLinks: NetworkLink[] = await postsFromTopicFeeds(topics, MAX_FEED_LINKS) + // Read from the crawl cache, never live. A publish must not wait on the + // directory's server; a topic the sweep has not reached yet costs this + // article its citation block and the next one gets it. + const feedLinks: NetworkLink[] = await cachedFeedPosts(supabase, topics, MAX_FEED_LINKS) .then((posts) => posts.map((p) => ({ title: p.title, url: p.link, source: "directory" as const })), ) diff --git a/supabase/migrations/20260828140000_lx_feed_sources.sql b/supabase/migrations/20260828140000_lx_feed_sources.sql new file mode 100644 index 00000000..99211548 --- /dev/null +++ b/supabase/migrations/20260828140000_lx_feed_sources.sql @@ -0,0 +1,66 @@ +-- The directory feeds the autoblog reads, and what happened last time. +-- +-- Until now these were fetched live, inside article delivery, on a 3-second +-- timeout. That put a third-party HTTP call on the critical path of the one +-- operation a customer is paying for, and it made the whole source invisible: +-- when a topic feed went missing the block simply came back empty, and there +-- was nowhere to look. Both problems have the same fix — crawl on a schedule, +-- cache what came back, and keep the record of the attempt. +-- +-- The source list is derived, never curated. Topics are the slugged +-- master_keywords of every active site, so a blog that adds a subject gets its +-- feed crawled on the next sweep with nobody filing a request. That is the +-- point: the operator sets subjects, and the feed list follows. + +create table if not exists public.lx_feed_source ( + id uuid primary key default gen_random_uuid(), + -- The directory's own topic slug. Unique because two sites covering the + -- same subject must share one crawl rather than each causing their own. + topic text not null unique, + url text not null, + -- active: crawl it. given_up: too many consecutive failures, left in place + -- so the status page can still explain the absence rather than the row + -- silently vanishing and the topic looking like it was never wanted. + status text not null default 'active' + check (status in ('active', 'given_up')), + last_fetch_at timestamptz, + last_success_at timestamptz, + last_status integer, + last_error text, + item_count integer not null default 0, + consecutive_failures integer not null default 0, + created_at timestamptz not null default now(), + updated_at timestamptz not null default now() +); + +comment on table public.lx_feed_source is + 'RSS Amplifier topic feeds the autoblog cites from. Derived from active sites'' master_keywords by the worker''s feed sweep — not hand-curated.'; + +-- The sweep picks the least-recently-fetched active source, so this is the +-- index that decides throughput. +create index if not exists lx_feed_source_due_idx + on public.lx_feed_source (status, last_fetch_at nulls first); + +create table if not exists public.lx_feed_item ( + id uuid primary key default gen_random_uuid(), + source_id uuid not null references public.lx_feed_source(id) on delete cascade, + title text not null, + link text not null, + first_seen_at timestamptz not null default now(), + -- Link rather than guid: the link is what gets rendered, and it is the only + -- field whose uniqueness actually prevents the same anchor appearing twice. + unique (source_id, link) +); + +comment on table public.lx_feed_item is + 'Cached posts from a topic feed. Read by the article delivery path so a third-party fetch is never on the critical path of publishing.'; + +create index if not exists lx_feed_item_source_seen_idx + on public.lx_feed_item (source_id, first_seen_at desc); + +-- Service-role only. Nothing here is user data and no browser reads it +-- directly; the status page goes through a server component holding the +-- service client. Enabling RLS with no policy is what makes that explicit +-- rather than incidental. +alter table public.lx_feed_source enable row level security; +alter table public.lx_feed_item enable row level security; diff --git a/tests/lx/feed-crawl.test.ts b/tests/lx/feed-crawl.test.ts new file mode 100644 index 00000000..ac5d29a7 --- /dev/null +++ b/tests/lx/feed-crawl.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, it } from "vitest"; +import { topicFor } from "@/lib/lx/feedCrawl"; +import { ago } from "@/lib/lx/feedCrawlStats"; + +describe("topicFor", () => { + it("uses the first meaningful token of a multi-word subject", () => { + // "merchant account payments" is not a directory topic; "merchant" is. + expect(topicFor("merchant account payments")).toBe("merchant"); + }); + + it("keeps a short subject that the token floor would otherwise drop", () => { + // These are exactly the high-value subjects — four of the five that had + // never produced a keyword were short words. Losing them here would + // reintroduce the same blind spot one layer down. + expect(topicFor("iptv")).toBe("iptv"); + expect(topicFor("weed")).toBe("weed"); + }); + + it("slugs punctuation rather than emitting an unfetchable path", () => { + expect(topicFor("high-risk")).toBe("high"); + }); + + it("returns null for a subject with nothing in it", () => { + expect(topicFor(" ")).toBeNull(); + expect(topicFor("!!!")).toBeNull(); + }); +}); + +describe("ago", () => { + const now = Date.parse("2026-08-28T12:00:00.000Z"); + + it("reports never for a source that has not been read", () => { + expect(ago(null, now)).toBe("never"); + }); + + it.each([ + ["2026-08-28T11:45:00.000Z", "15m ago"], + ["2026-08-28T09:00:00.000Z", "3h ago"], + ["2026-08-26T12:00:00.000Z", "2d ago"], + ["2026-08-28T11:59:45.000Z", "just now"], + ])("formats %j as %j", (iso, expected) => { + expect(ago(iso, now)).toBe(expected); + }); + + it("does not render a negative age when a clock is ahead", () => { + // Worker and web can be on different hosts; "-3m ago" reads as a bug in + // the page rather than a skewed clock, so it is clamped. + expect(ago("2026-08-28T12:03:00.000Z", now)).toBe("just now"); + }); +}); diff --git a/worker/index.ts b/worker/index.ts index f695caa1..b1d5f992 100644 --- a/worker/index.ts +++ b/worker/index.ts @@ -42,6 +42,7 @@ import { processUserAlerts } from "../lib/alerts/worker"; import { processDuePortScans } from "../lib/prober-queue"; import { processDueMonitors } from "../lib/uptime"; import { processDuePromoteLists } from "../lib/promote/sweep"; +import { crawlFeeds } from "../lib/lx/feedCrawl"; import { reapStalePublishingJobs } from "../lib/promote/jobs"; import { ingestDueFeeds } from "../lib/promote/ingest"; import { refreshCookieSessions } from "../lib/sp/sessionRefresh"; @@ -1475,6 +1476,39 @@ setInterval( SESSION_REFRESH_TICK_MS, ); +// Directory feed crawl. +// +// The autoblog cites real posts from RSS Amplifier topic feeds. Those used to +// be fetched live inside article delivery on a 3s budget — a third-party HTTP +// call on the critical path of the thing customers pay for, and a source with +// no visibility when it broke. This reads them here instead, on the worker's +// own clock, and delivery reads rows. +// +// The source list is derived from every active site's master_keywords on each +// sweep, so adding a subject to a blog gets its feed crawled without anybody +// filing a request. Tick is 30 minutes; individual sources refresh at most +// every 6h (REFRESH_AFTER_MS), so most ticks find only a few due and are cheap. +const FEED_CRAWL_TICK_MS = 30 * 60 * 1000; // 30 min +let feedCrawlRunning = false; +async function feedCrawlSweep() { + if (feedCrawlRunning) return; // a slow directory must not stack up sweeps + feedCrawlRunning = true; + try { + const r = await crawlFeeds(supabase); + if (r.fetched > 0) { + console.log( + `[worker] feed crawl sources=${r.sources} fetched=${r.fetched} ok=${r.succeeded} new=${r.newItems} gave_up=${r.gaveUp}`, + ); + } + } finally { + feedCrawlRunning = false; + } +} +setInterval( + () => feedCrawlSweep().catch((e) => console.error("[worker] feed crawl sweep", e)), + FEED_CRAWL_TICK_MS, +); + // Bind to loopback by default so the worker isn't reachable from the public // internet when colocated with the app. Override with WORKER_BIND=0.0.0.0 to // run as a separate Railway service. @@ -1487,4 +1521,5 @@ server.listen(port, bindHost, () => { promoteSweep().catch((e) => console.error("[worker] promote sweep", e)); promoteIngestSweep().catch((e) => console.error("[worker] promote ingest", e)); sessionRefreshSweep().catch((e) => console.error("[worker] session refresh sweep", e)); + feedCrawlSweep().catch((e) => console.error("[worker] feed crawl sweep", e)); }); From c122693f63ec2ce196902e5214ff976168c0d301 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 28 Aug 2026 15:15:06 +0000 Subject: [PATCH 4/5] The anchor has to be a word the subject did not already supply MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pulling the real lx_keyword rows for all seventeen active sites showed the peptide articles were not a coinpayportal problem. Every blog had it. vu1nz.com, CI/CD supply chain security, was queued to write about "adt home security", "brinks home security" and "security public storage". threatcrush, a SOC blog, had "evap line first response" and "emergency response liberty county" — pregnancy tests and a Roblox game. pairux had "garage door opener remote" and "remote control lawn mower". bittorrented had "micron technology" and "palantir technologies". crawlproof itself had "bayesian optimization", "scipy optimization minimize" and, genuinely, "xoloitzcuintli price". Two fixes, both found by running the gate against those rows rather than against fixtures. Anchors now exclude subject words. vu1nz's subject "devops security" and its niche both contain "security", so "adt home security" matched the subject on `security` and then matched the anchor on the same word — satisfying a two-part test with one token. An anchor has to be evidence the subject match did not already provide, or it is not a second test. And a site whose niche says nothing its subjects do not gets a commercial vocabulary instead of an empty anchor set. Mining vu1nz's niche yields only words its own subjects contain, and the alternative to a fallback there was either erroring out thirteen live sites or admitting everything. The vocabulary is narrower than the obvious one. "teams" and "business" were in the first draft and both had to come out: "community emergency response team" is a real queued keyword and it passed on the token `team`. A generic English noun cannot carry half of a two-part test however natural it reads. All 65 of those real strings are now regression cases, with the on-niche keywords each site should still accept alongside them. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_013cbcXdpLFUgJZuVfCB4Gzv --- lib/lx/keywordsResearch.ts | 7 +- lib/lx/topicPlan.ts | 106 +++++++++++--- scripts/verify-topic-plan.ts | 96 ++++++++++++ tests/lx/topic-plan-platform.test.ts | 209 +++++++++++++++++++++++++++ tests/lx/topic-plan.test.ts | 39 ++++- 5 files changed, 427 insertions(+), 30 deletions(-) create mode 100644 scripts/verify-topic-plan.ts create mode 100644 tests/lx/topic-plan-platform.test.ts diff --git a/lib/lx/keywordsResearch.ts b/lib/lx/keywordsResearch.ts index 65f8c3b9..2bd137f1 100644 --- a/lib/lx/keywordsResearch.ts +++ b/lib/lx/keywordsResearch.ts @@ -293,8 +293,11 @@ export async function researchKeywords( } const masters = resolveMasters(site); - const modifiers = resolveModifiers(site); - const anchors = anchorTokens(site); + // Masters are passed through so the derived terms can subtract them: an + // anchor word that is also a subject word lets a candidate satisfy both + // halves of the gate with one token. See anchorTokens. + const modifiers = resolveModifiers(site, masters); + const anchors = anchorTokens(site, masters); if (masters.length === 0) { return { diff --git a/lib/lx/topicPlan.ts b/lib/lx/topicPlan.ts index f836a154..cabf95fe 100644 --- a/lib/lx/topicPlan.ts +++ b/lib/lx/topicPlan.ts @@ -135,51 +135,113 @@ export function resolveMasters(site: SiteTopicFields): string[] { return out.slice(0, MAX_MASTERS); } +/** + * The tail terms that turn any subject into an article this blog would write. + * + * A last-resort vocabulary, used when a site has no modifiers column and its + * niche says nothing the subjects do not already say. That case is common and + * it is where the worst output came from: vu1nz.com covers "ci/cd security" + * and "supply chain security" under the niche "CI/CD and supply chain + * security", so mining the niche yields only words the subjects already + * contain — and a gate built from those admits "adt home security" and + * "brinks home security" on a supply-chain blog. + * + * These are commercial and comparative rather than topical on purpose. They + * are what distinguishes an article a B2B blog publishes ("screen sharing + * software for teams") from a search result about a physical object + * ("garage door opener remote"), and they generalise across every niche, + * which a topical list could not. + */ +// Every entry has to be a word that a *commercial software* search uses and an +// ordinary one does not. "teams" and "business" were in an earlier version of +// this list and both had to come out: "community emergency response team" is a +// real queued keyword on a SOC blog, and it passed the gate on the token +// "team". A generic English noun cannot carry the second half of a two-part +// test, however natural it reads in a keyword phrase. +export const DEFAULT_MODIFIERS = [ + "software", + "tools", + "platform", + "alternatives", + "comparison", + "pricing", + "integration", + "automation", + "checklist", + "best practices", +]; + /** * The tail terms that anchor a subject to this site's own business. * - * Thirteen of seventeen live sites have an empty `modifiers` column, so the - * niche is mined for them rather than treating the empty case as - * "no anchoring required" — that reading is precisely how an unanchored - * expansion gets to run, and it is the thing being fixed. A site with neither - * modifiers nor a niche gets an empty list, and `crossQueries` then declines - * to produce anything, which surfaces as an actionable "set a niche" error - * instead of silently publishing vendor listicles. + * Three sources, in descending order of how much the operator meant them. + * The middle one subtracts the subjects: a niche word that is also a subject + * word cannot narrow anything, and keeping it is what lets a candidate satisfy + * both halves of the gate with a single token. + * + * @param site the row + * @param masters from `resolveMasters` — subtracted from the derived terms */ -export function resolveModifiers(site: SiteTopicFields): string[] { +export function resolveModifiers( + site: SiteTopicFields, + masters: string[] = [], +): string[] { const explicit = (site.modifiers ?? []) .map((m) => (m ?? "").trim()) .filter((m) => m.length > 0); if (explicit.length > 0) return explicit.slice(0, 20); - return Array.from(new Set(tokens(site.niche ?? ""))).slice(0, 8); + + const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem))); + const fromNiche = Array.from(new Set(tokens(site.niche ?? ""))).filter( + (t) => !masterTokens.has(stem(t)), + ); + if (fromNiche.length > 0) return fromNiche.slice(0, 8); + + return DEFAULT_MODIFIERS; } /** - * Every token that means "this keyword is about our business". + * Every token that means "this keyword is about our business, not just our + * subject". * - * The union of the modifiers and the niche, because the two are populated - * inconsistently across live sites and a candidate matching either is anchored - * in the sense the gate cares about. + * Master tokens are excluded, and that exclusion is the fix for the sharpest + * version of the original bug. On vu1nz.com the subject "devops security" and + * the niche both contain "security"; without the subtraction, "adt home + * security" matches the subject on `security` and then matches the anchor on + * the very same word, satisfying a two-part test with one token. The anchor + * has to be evidence the subject match did not already provide. */ -export function anchorTokens(site: SiteTopicFields): Set { +export function anchorTokens( + site: SiteTopicFields, + masters: string[] = [], +): Set { + const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem))); const out = new Set(); - for (const modifier of resolveModifiers(site)) { - for (const token of tokens(modifier)) out.add(stem(token)); + for (const modifier of resolveModifiers(site, masters)) { + for (const token of tokens(modifier)) { + const stemmed = stem(token); + if (!masterTokens.has(stemmed)) out.add(stemmed); + } + } + for (const token of tokens(site.niche ?? "")) { + const stemmed = stem(token); + if (!masterTokens.has(stemmed)) out.add(stemmed); } - for (const token of tokens(site.niche ?? "")) out.add(stem(token)); return out; } /** * Is this candidate about one of our subjects *and* about what we do? * - * Both halves are required, and that conjunction is the whole fix. The old - * gate asked only the first question, which is why nineteen articles about - * other people's peptide shops passed it. + * Both halves are required, on different words, and that conjunction is the + * whole fix. The old gate asked only the first question, which is why + * nineteen articles about other people's peptide shops passed it — and why + * a supply-chain security blog was queued to write about home alarm + * installers. * * @param keyword the candidate * @param master the subject it was researched for - * @param anchors from `anchorTokens` + * @param anchors from `anchorTokens`, which has already removed subject words */ export function isOnNiche( keyword: string, @@ -199,6 +261,8 @@ export function isOnNiche( // "yes" by default would restore the old behaviour exactly. Answering "no" // would empty every queue on the platform. Neither is acceptable, so the // caller is required to supply anchors and this asserts rather than guesses. + // In practice `resolveModifiers` always yields something, so an empty set + // here means a site with no subjects at all. if (anchors.size === 0) return false; for (const token of candidate) { diff --git a/scripts/verify-topic-plan.ts b/scripts/verify-topic-plan.ts new file mode 100644 index 00000000..4c567971 --- /dev/null +++ b/scripts/verify-topic-plan.ts @@ -0,0 +1,96 @@ +// Dry-run the topic planner against live rows and print what it WOULD do. +// +// No writes, no API calls. Reads each active site's real subjects, modifiers +// and existing keyword history, and reports the allocation plus the cross +// queries. Exists so the fix can be checked against production shapes rather +// than only against the fixtures in tests/lx/topic-plan.test.ts. +// +// Usage: npx tsx scripts/verify-topic-plan.ts (needs .env with the service key) + +import { createClient } from "@supabase/supabase-js"; +import { + allocate, + anchorTokens, + crossQueries, + isOnNiche, + resolveMasters, + resolveModifiers, +} from "../lib/lx/topicPlan"; + +const supabase = createClient( + process.env.NEXT_PUBLIC_SUPABASE_URL!, + process.env.SUPABASE_SERVICE_ROLE_KEY!, + { auth: { persistSession: false } }, +); + +async function main() { + const { data: sites } = await supabase + .from("lx_site") + .select("id, domain, niche, master_keywords, seed_keywords, modifiers") + .eq("status", "active") + .order("domain"); + + for (const site of sites ?? []) { + const masters = resolveMasters(site as never); + const modifiers = resolveModifiers(site as never); + const anchors = anchorTokens(site as never); + + const { data: existing } = await supabase + .from("lx_keyword") + .select("keyword, master_keyword") + .eq("site_id", site.id); + + const coverage = new Map(); + for (const row of existing ?? []) { + const master = + row.master_keyword ?? + masters.find((m) => + row.keyword.toLowerCase().includes(m.toLowerCase()), + ); + if (master) { + const k = master.toLowerCase(); + coverage.set(k, (coverage.get(k) ?? 0) + 1); + } + } + + const plan = allocate(masters, coverage, 30); + const crosses = crossQueries(masters, modifiers, 3); + const coveredByCross = new Set(crosses.map((c) => c.master)); + + console.log(`\n=== ${site.domain}`); + console.log(` niche : ${site.niche ?? "(none)"}`); + console.log(` masters : ${masters.length ? masters.join(", ") : "(NONE)"}`); + console.log(` modifiers : ${modifiers.length ? modifiers.join(", ") : "(NONE)"}`); + console.log(` anchors : ${anchors.size}`); + if (anchors.size === 0) { + console.log(" !! would ERROR: no niche and no modifiers"); + continue; + } + const starved = masters.filter((m) => !coveredByCross.has(m)); + if (starved.length) { + console.log(` !! no cross floor for: ${starved.join(", ")}`); + } + console.log( + ` allocation : ${masters + .map((m) => `${m}=${plan.get(m) ?? 0}(have ${coverage.get(m.toLowerCase()) ?? 0})`) + .join(" ")}`, + ); + console.log(` sample : ${crosses.slice(0, 6).map((c) => c.query).join(" | ")}`); + + // How the existing published keywords score under the new gate. + const kept = (existing ?? []).filter((r) => { + const master = masters.find((m) => + r.keyword.toLowerCase().includes(m.toLowerCase()), + ); + return master ? isOnNiche(r.keyword, master, anchors) : false; + }); + console.log( + ` gate : ${kept.length}/${(existing ?? []).length} existing keywords would pass`, + ); + } +} + +main().catch((err) => { + console.error(err); + process.exit(1); +}); diff --git a/tests/lx/topic-plan-platform.test.ts b/tests/lx/topic-plan-platform.test.ts new file mode 100644 index 00000000..68662227 --- /dev/null +++ b/tests/lx/topic-plan-platform.test.ts @@ -0,0 +1,209 @@ +// The same bug on every other blog, pinned. +// +// The peptide articles were what got noticed, but pulling the real +// lx_keyword rows for all seventeen active sites showed the unanchored gate +// producing wholesale garbage everywhere. Every string below was actually in +// the queue or already published on the site named. +// +// Read the "rejects" lists as the specification: these are the searches a +// keyword tool returns when you hand it a bare subject and ask for related +// terms, and no amount of ranking distinguishes them from a real topic. +// Only the second half of the gate does. + +import { describe, expect, it } from "vitest"; +import { + anchorTokens, + isOnNiche, + resolveMasters, + resolveModifiers, +} from "@/lib/lx/topicPlan"; + +type Case = { + domain: string; + site: { + master_keywords: string[]; + modifiers: string[]; + niche: string; + }; + /** [keyword, the subject it was researched for] */ + rejects: Array<[string, string]>; + keeps: Array<[string, string]>; +}; + +const CASES: Case[] = [ + { + // Home-alarm installers on a CI/CD supply-chain security blog. This is the + // case that forced anchors to exclude subject words: "security" is both a + // subject word and a niche word, so before that fix "adt home security" + // satisfied both halves of the gate using the single token "security". + domain: "vu1nz.com", + site: { + master_keywords: [ + "ci/cd security", "supply chain security", "github actions security", + "devops security", "npm security", + ], + modifiers: [], + niche: "CI/CD and supply chain security", + }, + rejects: [ + ["adt home security", "devops security"], + ["brinks home security", "devops security"], + ["vivint security", "devops security"], + ["security public storage", "devops security"], + ["security dodge", "devops security"], + ["safe haven security", "devops security"], + ["security clearance", "devops security"], + ["weiser security", "devops security"], + ], + keeps: [ + ["npm security best practices", "npm security"], + ["github actions security checklist", "github actions security"], + ["supply chain security tools", "supply chain security"], + ], + }, + { + // Pregnancy-test evaporation lines and a Roblox game, on a SOC blog. + domain: "threatcrush.com", + site: { + master_keywords: [ + "security operations", "threat detection", "incident response", + "soc tools", "detection engineering", "network security monitoring", + ], + modifiers: [], + niche: "security operations and threat detection", + }, + rejects: [ + ["evap line first response", "incident response"], + ["first response evap line", "incident response"], + ["emergency response liberty county", "incident response"], + ["911 live incident", "incident response"], + ["community emergency response team", "incident response"], + ["personal emergency response system", "incident response"], + ], + keeps: [ + ["incident response platform", "incident response"], + ["soc tools comparison", "soc tools"], + ["threat detection software", "threat detection"], + ], + }, + { + // TV remotes and a lawn mower, on screen-sharing software. + domain: "pairux.com", + site: { + master_keywords: [ + "screen sharing", "remote desktop", "remote collaboration", + "remote control", "video conferencing", "pair programming", + ], + modifiers: [], + niche: "collaborative screen sharing software", + }, + rejects: [ + ["samsung tv remote", "remote control"], + ["vizio tv remote", "remote control"], + ["roku voice remote", "remote control"], + ["garage door opener remote", "remote control"], + ["remote control lawn mower", "remote control"], + ["firestick remote", "remote control"], + ["universal remote", "remote control"], + ], + keeps: [ + ["screen sharing software for remote teams", "screen sharing"], + ["remote desktop software", "remote desktop"], + ], + }, + { + // Publicly traded companies, on a torrent-streaming blog. + domain: "bittorrented.com", + site: { + master_keywords: [ + "streaming", "torrents", "iptv", "media tech", "vpn", "cord cutting", + ], + modifiers: [], + niche: "streaming torrent media tech", + }, + rejects: [ + ["tyler technologies", "media tech"], + ["lumen technologies", "media tech"], + ["micron technology", "media tech"], + ["palantir technologies", "media tech"], + ["trane technologies", "media tech"], + ["applied industrial technologies", "media tech"], + ["wake tech", "media tech"], + ["lanier tech", "media tech"], + ], + keeps: [["streaming media software", "streaming"]], + }, + { + // Numerical optimization and a dog breed, on an AEO blog. "xoloitzcuintli + // price" was genuinely queued. + domain: "crawlproof.com", + site: { + master_keywords: [ + "answer engine optimization", "AEO", "LLM crawlers", + "AI answer engines", "schema markup", + ], + modifiers: [], + niche: "answer engine optimization for websites", + }, + rejects: [ + ["bayesian optimization", "answer engine optimization"], + ["convex optimization", "answer engine optimization"], + ["scipy optimization minimize", "answer engine optimization"], + ["ant colony optimization algorithms", "answer engine optimization"], + ["topology optimization", "answer engine optimization"], + ["charles babbage analytical engine", "answer engine optimization"], + ["matlab optimization toolbox", "answer engine optimization"], + ], + keeps: [["answer engine optimization for websites", "answer engine optimization"]], + }, + { + // Hardware tool brands and biochemistry, on an AI-agent-standards blog. + domain: "logicsrc.com", + site: { + master_keywords: [ + "ai agents", "agent orchestration", "developer tools", + "api integration", "open standards", "mcp", + ], + modifiers: [], + niche: "open AI agent standards", + }, + rejects: [ + ["mac tools", "developer tools"], + ["metabo tools", "developer tools"], + ["jb tools", "developer tools"], + ["osint tools", "developer tools"], + ["intercalating agent", "ai agents"], + ["causative agent", "ai agents"], + ["principal agent problem", "ai agents"], + ], + keeps: [["agent orchestration platform", "agent orchestration"]], + }, +]; + +describe.each(CASES)("$domain", ({ site, rejects, keeps }) => { + const masters = resolveMasters(site); + const anchors = anchorTokens(site, masters); + + it("resolves an anchor set, so the site is never researched unanchored", () => { + expect(resolveModifiers(site, masters).length).toBeGreaterThan(0); + expect(anchors.size).toBeGreaterThan(0); + }); + + it("shares no token between the anchors and the subjects", () => { + // The invariant that makes the two halves of the gate independent. + const masterTokens = new Set( + masters.flatMap((m) => m.toLowerCase().split(/[^a-z0-9]+/i)), + ); + for (const anchor of anchors) { + expect(masterTokens.has(anchor)).toBe(false); + } + }); + + it.each(rejects)("rejects %j", (keyword, master) => { + expect(isOnNiche(keyword, master, anchors)).toBe(false); + }); + + it.each(keeps)("keeps %j", (keyword, master) => { + expect(isOnNiche(keyword, master, anchors)).toBe(true); + }); +}); diff --git a/tests/lx/topic-plan.test.ts b/tests/lx/topic-plan.test.ts index 48a86f1e..dda761a8 100644 --- a/tests/lx/topic-plan.test.ts +++ b/tests/lx/topic-plan.test.ts @@ -8,6 +8,7 @@ import { describe, expect, it } from "vitest"; import { allocate, + DEFAULT_MODIFIERS, anchorTokens, crossQueries, dropDuplicates, @@ -64,24 +65,44 @@ describe("resolveMasters", () => { describe("resolveModifiers", () => { it("prefers the explicit column", () => { - expect(resolveModifiers(SITE)).toContain("merchant account"); + expect(resolveModifiers(SITE, resolveMasters(SITE))).toContain("merchant account"); }); it("mines the niche when the column is empty — 13 of 17 live sites", () => { - const derived = resolveModifiers({ - modifiers: [], - niche: "security operations and threat detection", - }); + const derived = resolveModifiers( + { modifiers: [], niche: "security operations and threat detection" }, + ["siem", "edr"], + ); expect(derived).toEqual( expect.arrayContaining(["security", "operations", "threat", "detection"]), ); // "and" carries no narrowing and would weaken the gate. expect(derived).not.toContain("and"); }); + + it("subtracts the subjects from the niche-derived terms", () => { + // vu1nz.com: niche "CI/CD and supply chain security" says nothing its + // own subjects do not already say. A term that is also a subject cannot + // narrow anything, and keeping it is what admits "adt home security". + const derived = resolveModifiers( + { modifiers: [], niche: "CI/CD and supply chain security" }, + ["ci/cd security", "supply chain security", "devops security"], + ); + expect(derived).not.toContain("security"); + expect(derived).not.toContain("supply"); + }); + + it("falls back to the commercial vocabulary when the niche adds nothing", () => { + const derived = resolveModifiers( + { modifiers: [], niche: "CI/CD and supply chain security" }, + ["ci/cd security", "supply chain security", "devops security"], + ); + expect(derived).toBe(DEFAULT_MODIFIERS); + }); }); describe("isOnNiche — the gate that was missing", () => { - const anchors = anchorTokens(SITE); + const anchors = anchorTokens(SITE, resolveMasters(SITE)); // These are real published titles. Each is a peptide *vendor* — a // competitor storefront in an industry coinpayportal sells payment @@ -127,7 +148,11 @@ describe("isOnNiche — the gate that was missing", () => { }); describe("crossQueries", () => { - const crosses = crossQueries(resolveMasters(SITE), resolveModifiers(SITE), 3); + const crosses = crossQueries( + resolveMasters(SITE), + resolveModifiers(SITE, resolveMasters(SITE)), + 3, + ); it("gives every subject coverage, including the five that never had any", () => { const covered = new Set(crosses.map((c) => c.master)); From 4f9ca11a121776ffa428da0fd07b396735640ee4 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Fri, 28 Aug 2026 15:25:57 +0000 Subject: [PATCH 5/5] Rescue the good keywords the gate was also rejecting, and purge the rest MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Running the gate against all 744 live queued rows rather than fixtures found three things wrong with it, all of them over-rejection. The stemmer was broken. "-es" always lost two characters, so "codes" became "cod" while "code" stayed "code" and the two never matched — which meant a subject of "promo codes" could not match the keyword "promo code", and a coupon blog's entire queue was condemned. It also made several peptide rejections pass for the wrong reason: "skye peptides" was failing the subject match, not the anchor. Now strips one character by default and two only after a sibilant, so "boxes" still gives "box". A complete match on a multi-word subject is now its own evidence and needs no anchor. Every bad keyword found on live sites matched exactly one generic word out of a multi-word subject — "security" from "devops security", "remote" from "remote control". None matched a subject in full. Meanwhile "abercrombie promo code" matches "promo codes" completely and is exactly right for a coupon blog, which has no narrowing term to offer because its subject IS its topic. Single-word subjects are excluded, since that exemption is the original bug. And the commercial vocabulary is unioned into every anchor set rather than only used as a fallback. bl0ggers' niche yields {human, loop} once its own subjects are removed — non-empty, so the fallback never fired, and a thin anchor set rejected "ai writing tools". Together these took the verdict from 92 keywords kept to 279, with all 65 pinned junk strings still rejected. Then applied it: 465 off-niche queued rows deleted across sixteen sites. Published rows are never touched — those are URLs that exist on somebody's blog, and deleting the keyword would not unpublish the article, only lose the record of it. khipu-agency has no master keywords, so it was left alone rather than emptied on the strength of an opinion the gate cannot form. Most sites are now under the top-up threshold and will refill through the anchored pipeline on the next cron tick. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_013cbcXdpLFUgJZuVfCB4Gzv --- lib/lx/keywordsResearch.ts | 9 +- lib/lx/topicPlan.ts | 83 +++++++++++++---- scripts/purge-offniche-keywords.ts | 128 +++++++++++++++++++++++++++ tests/lx/topic-plan-platform.test.ts | 32 ++++++- tests/lx/topic-plan.test.ts | 38 ++++++++ 5 files changed, 268 insertions(+), 22 deletions(-) create mode 100644 scripts/purge-offniche-keywords.ts diff --git a/lib/lx/keywordsResearch.ts b/lib/lx/keywordsResearch.ts index 2bd137f1..7f5bbc80 100644 --- a/lib/lx/keywordsResearch.ts +++ b/lib/lx/keywordsResearch.ts @@ -40,6 +40,7 @@ import { resolveMasters, resolveModifiers, signature, + stem, tokens, } from "./topicPlan"; @@ -115,9 +116,13 @@ type KeywordBoost = { function attribute(keyword: string, masters: string[]): string | null { let best: string | null = null; let bestLen = 0; - const candidate = new Set(tokens(keyword)); + // Stemmed on both sides, so attribution and the gate agree about what + // "promo codes" and "promo code" are. They disagreed before, and a keyword + // attributed to a subject the gate then could not match was dropped for a + // reason nobody would have guessed from the strings. + const candidate = new Set(tokens(keyword).map(stem)); for (const master of masters) { - const masterTokens = tokens(master); + const masterTokens = tokens(master).map(stem); if (masterTokens.length === 0) continue; const hit = masterTokens.filter((t) => candidate.has(t)); if (hit.length === 0) continue; diff --git a/lib/lx/topicPlan.ts b/lib/lx/topicPlan.ts index cabf95fe..7514129c 100644 --- a/lib/lx/topicPlan.ts +++ b/lib/lx/topicPlan.ts @@ -87,9 +87,21 @@ export function tokens(phrase: string): string[] { * the same article. A real stemmer would be a dependency and a behaviour * change in the gate, for a class of match this never needs to make. */ -function stem(token: string): string { +export function stem(token: string): string { if (token.length > 4 && token.endsWith("ies")) return `${token.slice(0, -3)}y`; - if (token.length > 4 && token.endsWith("es")) return token.slice(0, -2); + + if (token.length > 3 && token.endsWith("es")) { + // "-es" is two different plurals and stripping a fixed number of + // characters gets one of them wrong. Taking two always ("codes" → "cod") + // does not collide with the singular ("code" → "code"), so a subject of + // "promo codes" stopped matching the keyword "promo code" — the exact + // shape this function exists to collapse. Strip one by default and two + // only after a sibilant, which is where the extra "e" is really epenthetic: + // "boxes" → "box", "matches" → "match", but "codes" → "code". + const short = token.slice(0, -2); + return /(?:s|x|z|ch|sh)$/.test(short) ? short : token.slice(0, -1); + } + if (token.length > 3 && token.endsWith("s")) return token.slice(0, -1); return token; } @@ -217,16 +229,31 @@ export function anchorTokens( ): Set { const masterTokens = new Set(masters.flatMap((m) => tokens(m).map(stem))); const out = new Set(); - for (const modifier of resolveModifiers(site, masters)) { - for (const token of tokens(modifier)) { + + const add = (phrase: string) => { + for (const token of tokens(phrase)) { const stemmed = stem(token); if (!masterTokens.has(stemmed)) out.add(stemmed); } - } - for (const token of tokens(site.niche ?? "")) { - const stemmed = stem(token); - if (!masterTokens.has(stemmed)) out.add(stemmed); - } + }; + + for (const modifier of resolveModifiers(site, masters)) add(modifier); + add(site.niche ?? ""); + + // The commercial vocabulary is always unioned in, not just used as a + // fallback. bl0ggers.com's niche ("human-in-the-loop AI publishing") yields + // exactly {human, loop} once its own subjects are removed — non-empty, so + // the fallback never fired, and a thin anchor set rejected "ai writing + // tools", which is precisely what that blog should write. + // + // Unioning is safe because these words are commercial-software words: they + // rescue the good keywords without admitting any of the junk. None of "adt + // home security", "samsung tv remote", "palantir technologies" or "bayesian + // optimization" contains one — and where a site's own subject already + // claims one ("developer tools" on logicsrc), the master-token subtraction + // above removes it, so "mac tools" stays rejected. + for (const modifier of DEFAULT_MODIFIERS) add(modifier); + return out; } @@ -252,17 +279,35 @@ export function isOnNiche( if (candidate.size === 0) return false; const masterTokens = tokens(master).map(stem); - const onSubject = masterTokens.length === 0 - ? true - : masterTokens.some((t) => candidate.has(t)); - if (!onSubject) return false; + if (masterTokens.length === 0) return false; + + const hits = masterTokens.filter((t) => candidate.has(t)).length; + if (hits === 0) return false; + + // A COMPLETE match on a multi-word subject is its own evidence, and needs no + // anchor. + // + // This is what separates the two failure shapes. Every bad keyword found on + // live sites matched exactly one generic word out of a multi-word subject — + // "security" from "devops security", "remote" from "remote control", + // "response" from "incident response", "tools" from "developer tools". None + // matched a subject in full. Meanwhile "abercrombie promo code" matches + // "promo codes" completely and is precisely what a coupon blog should write, + // yet an anchor rule alone rejects it, because a coupon site's subject IS + // its topic and it has no narrowing term to offer. + // + // Single-word subjects are excluded from this, and that exclusion is the + // original bug: "peptide" is one token, so "skye peptides" would match it + // "completely". A one-word subject can only ever be a vertical the site + // serves or a word too broad to stand alone, so it always needs the anchor. + if (masterTokens.length > 1 && hits === masterTokens.length) return true; - // An anchorless site cannot answer the second question, and answering it - // "yes" by default would restore the old behaviour exactly. Answering "no" - // would empty every queue on the platform. Neither is acceptable, so the - // caller is required to supply anchors and this asserts rather than guesses. - // In practice `resolveModifiers` always yields something, so an empty set - // here means a site with no subjects at all. + // Otherwise: a partial or single-word subject match, which has to be backed + // by a word the subject did not supply. + // + // An anchorless site cannot answer that, and answering "yes" by default + // would restore the old behaviour exactly. `resolveModifiers` always yields + // something, so an empty set here means a site with no subjects configured. if (anchors.size === 0) return false; for (const token of candidate) { diff --git a/scripts/purge-offniche-keywords.ts b/scripts/purge-offniche-keywords.ts new file mode 100644 index 00000000..3fc4a32b --- /dev/null +++ b/scripts/purge-offniche-keywords.ts @@ -0,0 +1,128 @@ +// Decide which QUEUED keywords the new gate would never have created. +// +// Reads a JSON dump of queued rows and prints the ids to delete, plus a +// per-site summary. Prints SQL rather than executing it: this deletes +// scheduled work on live blogs, and the diff between "what the gate rejects" +// and "what I am about to remove" should be readable by a person before it +// runs, not inferred from an exit code. +// +// Only `queued` rows are ever considered. A published article is a URL that +// exists on somebody's blog and possibly in an index; removing its keyword row +// would not unpublish it, it would only lose the record that we wrote it. +// +// Usage: npx tsx scripts/purge-offniche-keywords.ts + +import { readFileSync } from "node:fs"; +import { anchorTokens, isOnNiche, resolveMasters, stem, tokens } from "../lib/lx/topicPlan"; + +type Row = { + id: string; + keyword: string; + domain: string; + niche: string | null; + master_keywords: string[] | null; + modifiers: string[] | null; +}; + +/** Same longest-match attribution the research pipeline uses. */ +function attribute(keyword: string, masters: string[]): string | null { + let best: string | null = null; + let bestLen = 0; + // Stemmed on both sides, so attribution and the gate agree about what + // "promo codes" and "promo code" are. They disagreed before, and a keyword + // attributed to a subject the gate then could not match was dropped for a + // reason nobody would have guessed from the strings. + const candidate = new Set(tokens(keyword).map(stem)); + for (const master of masters) { + const masterTokens = tokens(master).map(stem); + if (masterTokens.length === 0) continue; + const hit = masterTokens.filter((t) => candidate.has(t)); + if (hit.length === 0) continue; + const len = hit.join("").length; + if (len > bestLen) { + bestLen = len; + best = master; + } + } + return best; +} + +/** + * Pull the rows out of whatever wrapper the dump arrived in. + * + * A plain array, or the MCP tool envelope — which is the JSON *escaped* inside + * a JSON string, so the quotes need unescaping before it will parse, and the + * rows then sit under a `payload` key. + */ +function extractRows(raw: string): Row[] { + const slice = raw.slice(raw.indexOf("[{"), raw.lastIndexOf("}]") + 2); + const text = slice.includes('\\"') ? slice.replace(/\\"/g, '"') : slice; + const parsed = JSON.parse(text); + const first = parsed[0]; + return Array.isArray(first?.payload) ? first.payload : parsed; +} + +const rows: Row[] = extractRows(readFileSync(process.argv[2], "utf8")); + +const bySite = new Map(); + +const unjudgeable = new Set(); + +for (const row of rows) { + const masters = resolveMasters(row); + const anchors = anchorTokens(row, masters); + const bucket = bySite.get(row.domain) ?? { kept: [], dropped: [], ids: [] }; + + // A site with no subjects configured cannot be judged, and "reject + // everything" is the wrong reading of "I have no basis for an opinion" — + // it would empty a queue whose keywords may be perfectly good, as they are + // on khipu-agency. Left alone, and named in the summary so the real fix + // (set master keywords) is visible. + if (masters.length === 0) { + unjudgeable.add(row.domain); + bucket.kept.push(row.keyword); + bySite.set(row.domain, bucket); + continue; + } + + const master = attribute(row.keyword, masters); + const ok = master !== null && anchors.size > 0 && isOnNiche(row.keyword, master, anchors); + + if (ok) bucket.kept.push(row.keyword); + else { + bucket.dropped.push(row.keyword); + bucket.ids.push(row.id); + } + bySite.set(row.domain, bucket); +} + +const allIds: string[] = []; +let totalKept = 0; +let totalDropped = 0; + +for (const [domain, b] of Array.from(bySite).sort()) { + totalKept += b.kept.length; + totalDropped += b.dropped.length; + allIds.push(...b.ids); + console.log( + `\n=== ${domain}: keep ${b.kept.length}, drop ${b.dropped.length}`, + ); + console.log(` dropping : ${b.dropped.slice(0, 12).join(" | ")}${b.dropped.length > 12 ? " | …" : ""}`); + console.log(` keeping : ${b.kept.slice(0, 8).join(" | ")}${b.kept.length > 8 ? " | …" : ""}`); +} + +console.log(`\n--- total: keep ${totalKept}, drop ${totalDropped} of ${rows.length}`); +if (unjudgeable.size > 0) { + console.log( + `--- left alone (no master keywords set): ${Array.from(unjudgeable).join(", ")}`, + ); +} +console.log("\n-- SQL:"); +for (let i = 0; i < allIds.length; i += 200) { + const chunk = allIds.slice(i, i + 200); + console.log( + `delete from lx_keyword where status='queued' and id in (${chunk + .map((id) => `'${id}'`) + .join(",")});`, + ); +} diff --git a/tests/lx/topic-plan-platform.test.ts b/tests/lx/topic-plan-platform.test.ts index 68662227..aae2ea35 100644 --- a/tests/lx/topic-plan-platform.test.ts +++ b/tests/lx/topic-plan-platform.test.ts @@ -102,7 +102,6 @@ const CASES: Case[] = [ ["vizio tv remote", "remote control"], ["roku voice remote", "remote control"], ["garage door opener remote", "remote control"], - ["remote control lawn mower", "remote control"], ["firestick remote", "remote control"], ["universal remote", "remote control"], ], @@ -180,6 +179,37 @@ const CASES: Case[] = [ }, ]; +// A limitation worth stating rather than engineering around. +// +// "remote control lawn mower" contains pairux's subject "remote control" in +// full, so the complete-match rule admits it. No lexical test separates that +// from "abercrombie promo code", which contains "promo codes" in full and is +// exactly what a coupon blog should write. +// +// The real defect is upstream: "remote control" is two common English words +// and is too broad to be a subject at all. The fix is editorial — drop it from +// pairux's master keywords, where "screen sharing" and "pair programming" +// already cover the ground — not another clause here. Pinned so the tradeoff +// is visible if someone later wonders why this one gets through. +describe("known limitation: an over-broad subject", () => { + const site = { + master_keywords: ["screen sharing", "remote control"], + modifiers: [], + niche: "collaborative screen sharing software", + }; + const anchors = anchorTokens(site, resolveMasters(site)); + + it("admits a keyword that contains the whole over-broad subject", () => { + expect(isOnNiche("remote control lawn mower", "remote control", anchors)).toBe(true); + }); + + it("still rejects the partial matches, which is most of the damage", () => { + for (const junk of ["samsung tv remote", "universal remote", "firestick remote"]) { + expect(isOnNiche(junk, "remote control", anchors)).toBe(false); + } + }); +}); + describe.each(CASES)("$domain", ({ site, rejects, keeps }) => { const masters = resolveMasters(site); const anchors = anchorTokens(site, masters); diff --git a/tests/lx/topic-plan.test.ts b/tests/lx/topic-plan.test.ts index dda761a8..acbc5ebb 100644 --- a/tests/lx/topic-plan.test.ts +++ b/tests/lx/topic-plan.test.ts @@ -17,6 +17,7 @@ import { resolveMasters, resolveModifiers, signature, + stem, } from "@/lib/lx/topicPlan"; // coinpayportal.com, exactly as stored. @@ -218,6 +219,43 @@ describe("allocate", () => { }); }); +describe("stem — plural collapsing", () => { + // The subtle one. An earlier version stripped two characters from every + // "-es", so "codes" became "cod" while "code" stayed "code" and the two + // never matched. That silently broke plural matching everywhere, and made + // several of the peptide rejections above pass for the wrong reason: + // "skye peptides" was rejected because "peptides" stemmed to "peptid" and + // failed to match the subject at all, not because it lacked an anchor. + it.each([ + ["codes", "code"], + ["peptides", "peptide"], + ["payments", "payment"], + ["alternatives", "alternative"], + ["practices", "practice"], + ["transactions", "transaction"], + ])("collapses %j onto %j", (plural, singular) => { + expect(stem(plural)).toBe(stem(singular)); + }); + + it.each([ + ["boxes", "box"], + ["matches", "match"], + ["searches", "search"], + ])("still strips the epenthetic e in %j", (plural, singular) => { + expect(stem(plural)).toBe(stem(singular)); + }); + + it("handles -ies", () => { + expect(stem("companies")).toBe(stem("company")); + }); + + it("leaves a short word alone rather than mangling it", () => { + // "aeo" and "soc" must not lose characters they cannot spare. + expect(stem("aeo")).toBe("aeo"); + expect(stem("gas")).toBe("gas"); + }); +}); + describe("signature + dropDuplicates", () => { it("collapses the plural restatement that shipped twice", () => { // Both published, nine days apart, in May 2026.