From d8b8d79636246f8df6f24e058752dc361881c74c Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Thu, 13 Aug 2026 13:31:51 +0000 Subject: [PATCH] feat(autoblog): gate drafts on quality, repair instead of discard, ship E-E-A-T fields MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit generateArticle() validated exactly one thing — that the internal links the model claimed to place were physically present — and any failure marked the keyword 'failed', throwing away a finished 4,000-word draft. Everything else about the post shipped unread. Meanwhile this repo already ships a slop detector we point at other people's sites, and the autoblog SDK ships the heuristic gate a receiver applies to posts we deliver. Neither was ever aimed at our own output. - lib/lx/qualityGate.ts: runs the in-repo slop checks (filler, placeholders, misspellings, first-party evidence) plus the SDK's receiver heuristics against a draft, and adds the cross-article checks a single-page scan structurally cannot do — near-duplicate and repeated-opening detection against the site's recent posts. Deterministic, no LLM call. - articleGen: link validation and the gate now produce model-readable violations instead of terminal failures, and a rejected draft is regenerated once with those violations appended to the brief. Gating runs before image generation, so a rejected draft never pays for four gpt-image-2 calls. The accepted draft's score is stored on the row. - webhookDeliver: posts went out with author: null, no structured data, and both dates stamped with the delivery time — so a retry moved the article's publication date, and our own audit would flag the content we deliver for missing attribution. Now carries a configurable byline, BlogPosting JSON-LD, and dates from the row. Only the near-duplicate threshold differs from the audit's (0.55 vs 0.70): we are judging our own generator, where that much overlap already means the post competes with one we published last week. Typecheck clean; 1451 tests pass, 15 of them new. Co-Authored-By: Claude Opus 5 (1M context) --- lib/audit/checks/slop.ts | 14 +- lib/lx/articleGen.ts | 298 ++++++++++++------ lib/lx/qualityGate.ts | 258 +++++++++++++++ lib/lx/webhookDeliver.ts | 100 +++++- ...260813120000_autoblog_quality_and_eeat.sql | 41 +++ tests/autoblog-quality-gate.test.ts | 199 ++++++++++++ 6 files changed, 812 insertions(+), 98 deletions(-) create mode 100644 lib/lx/qualityGate.ts create mode 100644 supabase/migrations/20260813120000_autoblog_quality_and_eeat.sql create mode 100644 tests/autoblog-quality-gate.test.ts diff --git a/lib/audit/checks/slop.ts b/lib/audit/checks/slop.ts index c3c4d601..09ef2eab 100644 --- a/lib/audit/checks/slop.ts +++ b/lib/audit/checks/slop.ts @@ -717,8 +717,15 @@ function safeHost(url: string): string | null { // Cross-page analysis // --------------------------------------------------------------------------- -/** 5-word shingle set, capped so a huge page can't blow up memory. */ -function shingles(text: string): Set { +/** + * 5-word shingle set, capped so a huge page can't blow up memory. + * + * Exported so the autoblog pre-publish gate (lib/lx/qualityGate.ts) can run + * the same near-duplicate comparison against previously generated articles. + * Keeping one implementation means a draft we accept is scored by exactly the + * measure the site audit would later use against it. + */ +export function shingles(text: string): Set { const words = text.toLowerCase().replace(/[^a-z0-9\s]/g, " ").split(/\s+/).filter(Boolean); const out = new Set(); for (let i = 0; i + 5 <= words.length && out.size < 4000; i++) { @@ -727,7 +734,8 @@ function shingles(text: string): Set { return out; } -function jaccard(a: Set, b: Set): number { +/** Shingle-set overlap, 0–1. Exported alongside `shingles` for the autoblog gate. */ +export function jaccard(a: Set, b: Set): number { if (a.size === 0 || b.size === 0) return 0; const [small, large] = a.size <= b.size ? [a, b] : [b, a]; let shared = 0; diff --git a/lib/lx/articleGen.ts b/lib/lx/articleGen.ts index ae93e5f8..ef474719 100644 --- a/lib/lx/articleGen.ts +++ b/lib/lx/articleGen.ts @@ -27,6 +27,11 @@ import { type ExchangeCandidate, } from "./exchangeMatcher"; import { generateStructuredOutput } from "./backendAi"; +import { + MAX_PRIOR_BODIES, + runQualityGate, + type QualityGateResult, +} from "./qualityGate"; import { consumeArticleGenerationCharge, refundArticleGenerationCharge, @@ -920,6 +925,85 @@ export function validateInternalLinks( return missing.length === 0 ? { ok: true } : { ok: false, missing }; } +// How many times we ask the model for a usable draft. Attempt 2 is a repair +// pass that receives the previous attempt's violations. Two is deliberate: +// the common failure is the model dropping one link or leaning on a filler +// phrase, which one round of specific feedback fixes. Beyond that the draft +// is usually wrong in a way more sampling will not fix, and each attempt +// costs a full long-form generation. +const MAX_GENERATION_ATTEMPTS = 2; + +/** + * Link-placement violations, phrased for the model rather than for a log line. + * + * Previously each of these called failKeyword() directly, which threw away a + * finished 4,000-word draft because one URL was missing from the body. They + * are now collected so the repair pass can fix them in place. + */ +export function collectLinkViolations( + article: ArticleOutput, + offeredExchangeUrls: Set, +): string[] { + const violations: string[] = []; + + const internal = validateInternalLinks( + article.markdown_body, + article.used_internal_link_urls, + ); + if (!internal.ok) { + violations.push( + `You listed these URLs in used_internal_link_urls but did not actually place them in markdown_body: ${internal.missing.join(", ")}. Insert each one inline as [anchor](url) in a sentence genuinely about that page, or drop it from used_internal_link_urls.`, + ); + } + + const exchangeUsed = article.used_exchange_link_urls ?? []; + const phantom = exchangeUsed.filter((u) => !offeredExchangeUrls.has(u)); + if (phantom.length > 0) { + violations.push( + `These URLs are not in the partner-blog candidate list you were given and must not appear: ${phantom.join(", ")}. Use only the exact URLs provided.`, + ); + } + + const exchange = validateInternalLinks(article.markdown_body, exchangeUsed); + if (!exchange.ok) { + violations.push( + `You listed these URLs in used_exchange_link_urls but did not place them in markdown_body: ${exchange.missing.join(", ")}. Insert each inline, or drop it from used_exchange_link_urls.`, + ); + } + + return violations; +} + +/** Recent article bodies on this site, for the gate's duplication checks. */ +async function fetchPriorBodies( + supabase: SupabaseClient, + siteId: string, +): Promise> { + const { data } = await supabase + .from("lx_article") + .select("slug, content_markdown") + .eq("site_id", siteId) + .in("status", ["ready", "publishing", "published"]) + .order("created_at", { ascending: false }) + .limit(MAX_PRIOR_BODIES); + return ((data as Array<{ slug: string; content_markdown: string }> | null) ?? []).map( + (r) => ({ slug: r.slug, body: r.content_markdown ?? "" }), + ); +} + +/** Append the previous attempt's violations to the user prompt as a fix list. */ +export function buildRepairPrompt(basePrompt: string, violations: string[]): string { + return [ + basePrompt, + "", + "---", + "", + "IMPORTANT — your previous attempt at this article was rejected by the pre-publish quality gate. Write the article again from scratch, keeping everything that was working, and fix every item below. Do not acknowledge this instruction in the output.", + "", + ...violations.map((v, i) => `${i + 1}. ${v}`), + ].join("\n"); +} + export type GenerateArticleResult = { ok: boolean; articleId?: string; @@ -1103,111 +1187,144 @@ export async function generateArticle( ); } - // Generate the article body. - let article: ArticleOutput; - try { - const generated = await generateStructuredOutput({ - name: "lx_article", - schema: ArticleSchema, - system: buildSystemPrompt(), - user: buildUserPrompt({ - site: typedSite, - keyword: keyword.keyword, - keywordMeta: { - articleType: keyword.article_type, - customInstructions: keyword.custom_instructions, - }, - candidates, - linkSlots, - exchangeCandidates, - exchangeSlots, - exchangeRelaxed, - }), - anthropic, - openai, - anthropicModel: CLAUDE_MODEL, - // 3,200–4,500 words ≈ ~18k–25k output tokens. JSON escape overhead - // can push that another 30%. 48k gives meaningful headroom. - maxTokens: 48000, - anthropicCacheSystemPrompt: true, - }); - article = normalizeArticleOutput(generated.output); - } catch (err) { - const errMsg = err instanceof Error ? err.message : String(err); - // Provider monthly caps (e.g. Anthropic HTTP 400 "specified API usage - // limits") and generic rate-limits are transient. Marking the keyword - // 'failed' here permanently consumes it — when the cap resets the - // user is stuck with stranded rows and a dedup'd research path that - // never re-inserts them. Requeue instead so the keyword retries on - // the next worker tick once the upstream is back. - // eslint-disable-next-line @typescript-eslint/no-explicit-any - const status = (err as any)?.status; - const isTransient = - status === 429 || - /specified API usage limits|usage limits|rate.?limit/i.test(errMsg); - if (isTransient) { - await supabase - .from("lx_keyword") - .update({ status: "queued" }) - .eq("id", keyword.id); + // Prior bodies for the gate's near-duplicate / repeated-opening checks. + // Fetched once and reused across attempts. + const priorBodies = await fetchPriorBodies(supabase, typedSite.id); + const offeredExchangeUrls = new Set(exchangeCandidates.map((c) => c.url)); + const basePrompt = buildUserPrompt({ + site: typedSite, + keyword: keyword.keyword, + keywordMeta: { + articleType: keyword.article_type, + customInstructions: keyword.custom_instructions, + }, + candidates, + linkSlots, + exchangeCandidates, + exchangeSlots, + exchangeRelaxed, + }); + + // Generate the article body, re-running once with the gate's complaints if + // the first draft is not publishable. Gating happens BEFORE image + // generation so a rejected draft never pays for four gpt-image-2 calls. + let article: ArticleOutput | null = null; + let gate: QualityGateResult | null = null; + let violations: string[] = []; + + for (let attempt = 1; attempt <= MAX_GENERATION_ATTEMPTS; attempt++) { + let candidate: ArticleOutput; + try { + const generated = await generateStructuredOutput({ + name: "lx_article", + schema: ArticleSchema, + system: buildSystemPrompt(), + user: + violations.length > 0 + ? buildRepairPrompt(basePrompt, violations) + : basePrompt, + anthropic, + openai, + anthropicModel: CLAUDE_MODEL, + // 3,200–4,500 words ≈ ~18k–25k output tokens. JSON escape overhead + // can push that another 30%. 48k gives meaningful headroom. + maxTokens: 48000, + anthropicCacheSystemPrompt: true, + }); + candidate = normalizeArticleOutput(generated.output); + } catch (err) { + const errMsg = err instanceof Error ? err.message : String(err); + // Provider monthly caps (e.g. Anthropic HTTP 400 "specified API usage + // limits") and generic rate-limits are transient. Marking the keyword + // 'failed' here permanently consumes it — when the cap resets the + // user is stuck with stranded rows and a dedup'd research path that + // never re-inserts them. Requeue instead so the keyword retries on + // the next worker tick once the upstream is back. + // eslint-disable-next-line @typescript-eslint/no-explicit-any + const status = (err as any)?.status; + const isTransient = + status === 429 || + /specified API usage limits|usage limits|rate.?limit/i.test(errMsg); + if (isTransient) { + await supabase + .from("lx_keyword") + .update({ status: "queued" }) + .eq("id", keyword.id); + await refundArticleGenerationCharge(supabase, charge); + console.warn( + `[lx] keyword ${keyword.id} requeued (transient backend AI error): ${errMsg}`, + ); + return { ok: false, error: "backend AI transient error" }; + } + await failKeyword(supabase, keyword.id, `backend AI error: ${errMsg}`); await refundArticleGenerationCharge(supabase, charge); + return { ok: false, error: "backend AI error" }; + } + + // Link placement, then the content gate. Both produce model-readable + // violations rather than terminal failures, so attempt 2 can fix them. + // + // Exchange links are framed as UP TO (not EXACTLY) in the prompt, so an + // empty used_exchange_link_urls is a judgement call, not a violation — + // only a claimed-but-absent or never-offered URL counts. + const linkViolations = collectLinkViolations(candidate, offeredExchangeUrls); + + // The gate needs rendered HTML. Render the body without images: the + // inline-image markers are comments and the hero is not in the body, so + // the word count, link density, and slop signals are unaffected — and + // rendering here means a rejected draft never reaches image generation. + let gateHtml = ""; + try { + gateHtml = await markdownToHtml(candidate.markdown_body); + } catch (err) { + // A body that will not render is not repairable by re-prompting on + // content grounds, but the next attempt may simply not hit it. console.warn( - `[lx] keyword ${keyword.id} requeued (transient backend AI error): ${errMsg}`, + `[lx] gate render failed on attempt ${attempt}`, + err instanceof Error ? err.message : err, ); - return { ok: false, error: "backend AI transient error" }; } - await failKeyword(supabase, keyword.id, `backend AI error: ${errMsg}`); - await refundArticleGenerationCharge(supabase, charge); - return { ok: false, error: "backend AI error" }; - } - // Validate that all "used" internal links are actually present in the body. - const linkCheck = validateInternalLinks( - article.markdown_body, - article.used_internal_link_urls, - ); - if (!linkCheck.ok) { - await failKeyword( - supabase, - keyword.id, - `model claimed links not present: ${linkCheck.missing.join(", ")}`, + const result = gateHtml + ? runQualityGate({ + html: gateHtml, + title: candidate.title, + metaDescription: candidate.meta_description, + priorBodies, + }) + : null; + + article = candidate; + gate = result; + violations = [...linkViolations, ...(result?.violations ?? [])]; + + if (violations.length === 0) break; + + console.warn( + `[lx] keyword ${keyword.id} attempt ${attempt}/${MAX_GENERATION_ATTEMPTS} rejected (slop=${result?.score ?? "n/a"}): ${violations.join(" | ")}`, ); - await refundArticleGenerationCharge(supabase, charge); - return { ok: false, error: "internal-link validation failed" }; } - // Validate exchange links similarly. The prompt frames these as UP TO - // (not EXACTLY) — a 0 here just means the model judged none fit, not a - // failure. Only flag when the model claims a URL it didn't actually - // place, or claims a URL we never offered as a candidate. - const exchangeUrlsUsed = article.used_exchange_link_urls ?? []; - const offeredExchangeUrls = new Set(exchangeCandidates.map((c) => c.url)); - const phantomExchange = exchangeUrlsUsed.filter( - (u) => !offeredExchangeUrls.has(u), - ); - if (phantomExchange.length > 0) { - await failKeyword( - supabase, - keyword.id, - `model used exchange URLs not in candidate list: ${phantomExchange.join(", ")}`, - ); + if (!article) { + // Unreachable — the loop either assigns or returns — but it keeps the + // narrowing honest for everything below. + await failKeyword(supabase, keyword.id, "no draft produced"); await refundArticleGenerationCharge(supabase, charge); - return { ok: false, error: "exchange-link validation failed" }; + return { ok: false, error: "no draft produced" }; } - const exchangeCheck = validateInternalLinks( - article.markdown_body, - exchangeUrlsUsed, - ); - if (!exchangeCheck.ok) { + + if (violations.length > 0) { await failKeyword( supabase, keyword.id, - `model claimed exchange links not present: ${exchangeCheck.missing.join(", ")}`, + `quality gate failed after ${MAX_GENERATION_ATTEMPTS} attempts (slop=${gate?.score ?? "n/a"}): ${violations.join(" | ")}`, ); await refundArticleGenerationCharge(supabase, charge); - return { ok: false, error: "exchange-link validation failed" }; + return { ok: false, error: "quality gate failed" }; } + const exchangeUrlsUsed = article.used_exchange_link_urls ?? []; + // Slugify + dedupe slug. const baseSlug = article.slug || slugify(article.title); const finalSlug = await uniqueSlug(supabase, typedSite.id, baseSlug); @@ -1332,6 +1449,11 @@ export async function generateArticle( tags: article.tags, internal_links: internalLinksPayload, outbound_links: outboundLinksPayload, + // Gate verdict for the draft we accepted. Stored so a site's quality + // can be tracked over time rather than only enforced at a threshold — + // a blog whose scores are creeping up is drifting before it fails. + slop_score: gate?.score ?? null, + slop_issues: gate ? gate.issues : [], status: "ready", }) .select("id") diff --git a/lib/lx/qualityGate.ts b/lib/lx/qualityGate.ts new file mode 100644 index 00000000..20057db9 --- /dev/null +++ b/lib/lx/qualityGate.ts @@ -0,0 +1,258 @@ +// Pre-publish quality gate for autoblog drafts. +// +// Until now generateArticle() validated exactly one thing — that the internal +// links the model *claimed* to place were physically present — and then wrote +// the row at status='ready'. Everything else about the draft went out unread. +// +// Meanwhile this repo already ships a slop detector that we point at other +// people's sites (lib/audit/checks/slop.ts) and the autoblog SDK ships the +// heuristic gate a receiver applies to posts we deliver +// (@profullstack/autoblog/quality). This module runs both against our own +// output before we spend image money or write the row, so: +// +// 1. We fail on our own terms instead of getting 4xx'd by a receiver. +// 2. A draft that would embarrass us in a Slop Score audit never ships. +// 3. The violations come back as text the model can act on, which lets +// generateArticle() repair a draft rather than burn the keyword. +// +// Deliberately no LLM call here. The SDK's scoreQuality() would add one, but +// every signal below is deterministic, which keeps the gate free, instant, +// and identical between the test suite and production. + +import { scoreHeuristics } from "@profullstack/autoblog/quality"; +import type { Post } from "@profullstack/autoblog"; +import { + analyzePage, + jaccard, + shingles, + slopGrade, + slopScore, + toSlopPage, + type SlopGrade, + type SlopIssue, +} from "../audit/checks/slop"; + +/** + * Issue keys from analyzePage() that are meaningful for a rendered article + * body. The rest of that function inspects full-page concerns — viewport meta, + * deprecated tags, dev-host leakage, inline-style density — which belong to the + * receiver's template, not to our markdown. Scoring them here would blame the + * draft for its host's markup. + */ +const BODY_RELEVANT_ISSUE_KEYS = new Set([ + "content.placeholder", + "content.filler", + "content.no_first_party_evidence", + "content.thin", + "content.stale_copyright", + "content.misspelling", + "code.dead_links", + "design.placeholder_alt", +]); + +/** + * Near-duplicate threshold against our own prior articles. + * + * The site audit flags pages at ≥0.70 because it is judging a stranger's site + * and wants to be sure before making the accusation. Here we are judging our + * own generator, where 0.55 shingle overlap between two posts on the same blog + * already means the second one is mostly a re-run of the first — and it is + * cheaper to regenerate now than to publish a page that competes with one we + * published last week. + */ +const NEAR_DUPLICATE_THRESHOLD = 0.55; + +/** How much of the body counts as "the intro" for repeated-opening detection. */ +const INTRO_CHARS = 200; + +/** Prior articles compared against. Beyond this the shingling cost stops paying. */ +export const MAX_PRIOR_BODIES = 40; + +/** + * Slop score above which a draft is rejected. 25 is the top of the audit's + * "Clean" band — we hold our own output to the grade we would want a customer + * to see on their report. + */ +export const DEFAULT_MAX_SLOP_SCORE = 25; + +export type PriorBody = { + slug: string; + /** Rendered HTML or markdown — only the word sequence matters. */ + body: string; +}; + +export type QualityGateInput = { + html: string; + title: string; + metaDescription: string; + /** Previously generated articles on the same site. */ + priorBodies?: PriorBody[]; + /** Overrides DEFAULT_MAX_SLOP_SCORE. */ + maxSlopScore?: number; +}; + +export type QualityGateResult = { + ok: boolean; + score: number; + grade: SlopGrade; + issues: SlopIssue[]; + /** + * Repair instructions, one per violation, phrased as something the model can + * act on. Empty iff ok. + */ + violations: string[]; + metrics: { + wordCount: number; + linkCount: number; + linkDensity: number; + imageCount: number; + }; +}; + +/** Strip markdown/HTML down to the word sequence the shingler wants. */ +function bodyText(input: string): string { + return input + .replace(//gi, " ") + .replace(//gi, " ") + .replace(/<[^>]+>/g, " ") + .replace(/!\[[^\]]*\]\([^)]*\)/g, " ") + .replace(/\[([^\]]+)\]\([^)]*\)/g, "$1") + .replace(/[#>*_`~|-]+/g, " ") + .replace(/\s+/g, " ") + .trim(); +} + +function introKey(text: string): string { + return text.slice(0, INTRO_CHARS).toLowerCase().replace(/[^a-z0-9 ]/g, "").trim(); +} + +/** + * Wrap a draft in the minimal Post the SDK's heuristic gate needs. The fields + * it does not read are filled with placeholders rather than left undefined so + * a future SDK version that starts reading them fails loudly in tests instead + * of silently scoring undefined. + */ +function draftPost(input: QualityGateInput): Post { + return { + id: "draft", + url: "https://example.invalid/draft", + title: input.title, + slug: "draft", + excerpt: input.metaDescription, + html: input.html, + status: "draft", + published_at: new Date(0).toISOString(), + updated_at: new Date(0).toISOString(), + tags: [], + categories: [], + }; +} + +export function runQualityGate(input: QualityGateInput): QualityGateResult { + const maxScore = input.maxSlopScore ?? DEFAULT_MAX_SLOP_SCORE; + const violations: string[] = []; + + // --- 1. Slop signals on the body itself --------------------------------- + const page = toSlopPage({ url: "https://example.invalid/draft", status: 200, html: input.html }); + // toSlopPage reads /<meta> out of the HTML, which a body fragment has + // no reason to carry. Supply them so the checks that compare title against + // body see the real values. + const analyzed = analyzePage({ ...page, title: input.title, description: input.metaDescription }); + const bodyIssues = analyzed.issues.filter((i) => BODY_RELEVANT_ISSUE_KEYS.has(i.key)); + + for (const issue of bodyIssues) { + violations.push(`${issue.label}. ${issue.fix}`); + } + + // --- 2. The heuristic gate our own receivers will apply ------------------ + // Failing here means the post would be rejected on delivery, so catching it + // now converts a wasted webhook round-trip into a repair. + // + // Note on link density: the SDK already skips in-page anchors (href="#…"), + // so our 40-link table of contents does not count against us, and its + // `linkDensity` metric is already expressed in percent. + const heuristics = scoreHeuristics(draftPost(input)); + for (const failure of heuristics.failed) { + violations.push( + `Receiver heuristic gate would reject this post: ${failure} (words=${heuristics.metrics.wordCount}, links=${heuristics.metrics.linkCount}, link density=${heuristics.metrics.linkDensity.toFixed(2)}%, images=${heuristics.metrics.imageCount}).`, + ); + } + + // --- 3. Cross-article duplication --------------------------------------- + // The single most damning signal an audit can find on a scaled blog, and the + // one a single-page check structurally cannot see. + const siteIssues: SlopIssue[] = []; + const priors = (input.priorBodies ?? []).slice(0, MAX_PRIOR_BODIES); + if (priors.length > 0) { + const draftText = bodyText(input.html); + const draftShingles = shingles(draftText); + + const dupes: Array<{ slug: string; sim: number }> = []; + for (const prior of priors) { + const sim = jaccard(draftShingles, shingles(bodyText(prior.body))); + if (sim >= NEAR_DUPLICATE_THRESHOLD) dupes.push({ slug: prior.slug, sim }); + } + dupes.sort((a, b) => b.sim - a.sim); + + if (dupes.length > 0) { + siteIssues.push({ + key: "content.near_duplicate", + dimension: "content", + label: `Draft is a near-duplicate of ${dupes.length} existing post${dupes.length === 1 ? "" : "s"} on this site`, + fix: "Rewrite to cover something the existing posts do not, or drop the keyword.", + weight: Math.min(14, 4 + dupes.length * 3), + count: dupes.length, + samples: dupes.slice(0, 5).map((d) => `${(d.sim * 100).toFixed(0)}% — ${d.slug}`), + }); + violations.push( + `This draft repeats material already published on this blog: ${dupes + .slice(0, 3) + .map((d) => `"${d.slug}" (${(d.sim * 100).toFixed(0)}% overlap)`) + .join( + ", ", + )}. Rewrite it so the argument, examples, and structure are genuinely different from those posts — do not simply reword the same sections.`, + ); + } + + // Repeated openings. Every post on this blog comes from one system prompt + // that mandates a fixed opening move, so this is the failure mode the + // generator is most prone to. + const draftIntro = introKey(draftText); + if (draftIntro.length >= 60) { + const sharedIntro = priors.filter((p) => introKey(bodyText(p.body)) === draftIntro); + if (sharedIntro.length > 0) { + siteIssues.push({ + key: "content.boilerplate_intro", + dimension: "content", + label: `Draft opens with the same first sentence as ${sharedIntro.length} existing post${sharedIntro.length === 1 ? "" : "s"}`, + fix: "Write an opening specific to this topic.", + weight: Math.min(8, 2 + sharedIntro.length * 2), + count: sharedIntro.length, + samples: sharedIntro.slice(0, 3).map((p) => p.slug), + }); + violations.push( + `The opening paragraph is identical to ${sharedIntro.length} post(s) already on this blog. Write a new opening built from this topic's own specifics.`, + ); + } + } + } + + // --- 4. Score ------------------------------------------------------------ + const scored = { ...analyzed, issues: bodyIssues, points: bodyIssues.reduce((n, i) => n + i.weight, 0) }; + const score = slopScore([scored], siteIssues); + + if (score > maxScore) { + violations.push( + `Overall slop score is ${score}/100 (grade "${slopGrade(score)}"); this blog's ceiling is ${maxScore}. Address the issues above.`, + ); + } + + return { + ok: violations.length === 0, + score, + grade: slopGrade(score), + issues: [...bodyIssues, ...siteIssues], + violations, + metrics: heuristics.metrics, + }; +} diff --git a/lib/lx/webhookDeliver.ts b/lib/lx/webhookDeliver.ts index 26c8603c..49bf3cf8 100644 --- a/lib/lx/webhookDeliver.ts +++ b/lib/lx/webhookDeliver.ts @@ -32,6 +32,8 @@ type ArticleRow = { webhook_delivery_id: string | null; webhook_attempts: number; created_at: string; + published_at: string | null; + updated_at: string | null; }; type SiteRow = { @@ -40,8 +42,65 @@ type SiteRow = { blog_root_url: string; webhook_url: string | null; webhook_secret: string | null; + author_name: string | null; + author_url: string | null; }; +/** + * BlogPosting JSON-LD prepended to the delivered HTML. + * + * crawlproof's own audit marks a site down for shipping articles without an + * author, a published date, or Article markup (see content.author, + * content.date_signal, and the schema recommendations in lib/audit). Until now + * the autoblog delivered exactly that: no byline, no dates in the body, no + * structured data — so our own product handed customers content that our own + * report would then flag on their domain. + * + * Emitting the markup inside `html` rather than as a new payload field is + * deliberate: the CloudEvents `Post` shape is shared by four Profullstack + * consumers and adding a field means every receiver has to learn it, whereas + * a receiver that already renders `html` picks this up with no change at all. + */ +export function buildArticleJsonLd(input: { + url: string; + title: string; + description: string; + imageUrl: string | null; + publishedAt: string; + updatedAt: string; + authorName: string | null; + authorUrl: string | null; + publisherName: string; + tags: string[]; +}): string { + const author = input.authorName + ? { + "@type": "Person", + name: input.authorName, + ...(input.authorUrl ? { url: input.authorUrl, sameAs: [input.authorUrl] } : {}), + } + : { "@type": "Organization", name: input.publisherName }; + + const payload = { + "@context": "https://schema.org", + "@type": "BlogPosting", + headline: input.title, + description: input.description, + mainEntityOfPage: { "@type": "WebPage", "@id": input.url }, + url: input.url, + datePublished: input.publishedAt, + dateModified: input.updatedAt, + author, + publisher: { "@type": "Organization", name: input.publisherName }, + ...(input.imageUrl ? { image: [input.imageUrl] } : {}), + ...(input.tags.length > 0 ? { keywords: input.tags.join(", ") } : {}), + }; + + // JSON inside a <script> must not be able to close the tag early. + const json = JSON.stringify(payload).replace(/</g, "\\u003c"); + return `<script type="application/ld+json">${json}</script>`; +} + export type DeliveryResult = { ok: boolean; status: "published" | "failed"; @@ -53,6 +112,27 @@ export type DeliveryResult = { function articleToPost(article: ArticleRow, site: SiteRow): Post { const blogRoot = site.blog_root_url.replace(/\/$/, ""); const url = `${blogRoot}/${article.slug}`; + + // Dates come from the row, not from `now()`. Stamping both fields with the + // delivery time meant every retry moved the article's publication date, and + // a redelivered post looked freshly written to anything reading the feed. + const now = new Date().toISOString(); + const publishedAt = article.published_at ?? article.created_at ?? now; + const updatedAt = article.updated_at ?? publishedAt; + + const jsonLd = buildArticleJsonLd({ + url, + title: article.title, + description: article.excerpt || article.meta_description || "", + imageUrl: article.image_url, + publishedAt, + updatedAt, + authorName: site.author_name, + authorUrl: site.author_url, + publisherName: site.domain, + tags: article.tags ?? [], + }); + return { id: article.id, url, @@ -60,12 +140,14 @@ function articleToPost(article: ArticleRow, site: SiteRow): Post { title: article.title, slug: article.slug, excerpt: article.excerpt || article.meta_description || null, - html: article.content_html, + html: `${jsonLd}\n${article.content_html}`, markdown: article.content_markdown, status: "published", - published_at: new Date().toISOString(), - updated_at: new Date().toISOString(), - author: null, + published_at: publishedAt, + updated_at: updatedAt, + author: site.author_name + ? { name: site.author_name, ...(site.author_url ? { url: site.author_url } : {}) } + : null, tags: article.tags ?? [], categories: [], featured_image: article.image_url ? { url: article.image_url } : null, @@ -85,7 +167,7 @@ export async function deliverArticle( .eq("id", articleId) .eq("status", "ready") .select( - "id, site_id, target_site_id, is_guest_post, title, slug, meta_description, excerpt, content_markdown, content_html, image_url, tags, outbound_links, internal_links, status, webhook_delivery_id, webhook_attempts, created_at", + "id, site_id, target_site_id, is_guest_post, title, slug, meta_description, excerpt, content_markdown, content_html, image_url, tags, outbound_links, internal_links, status, webhook_delivery_id, webhook_attempts, created_at, published_at, updated_at", ) .maybeSingle<ArticleRow & { target_site_id: string | null; is_guest_post: boolean | null }>(); if (!claimed) { @@ -104,7 +186,9 @@ export async function deliverArticle( const deliveryTargetId = claimed.target_site_id ?? claimed.site_id; const { data: site } = await supabase .from("lx_site") - .select("id, domain, blog_root_url, webhook_url, webhook_secret") + .select( + "id, domain, blog_root_url, webhook_url, webhook_secret, author_name, author_url", + ) .eq("id", deliveryTargetId) .maybeSingle<SiteRow>(); if (!site?.webhook_url || !site?.webhook_secret) { @@ -153,7 +237,9 @@ export async function deliverArticle( .from("lx_article") .update({ status: "published", - published_at: new Date().toISOString(), + // Same value we put in the payload and the JSON-LD, so the row and + // the receiver never disagree about when this post was published. + published_at: post.published_at, webhook_delivery_id: event.id, webhook_response_code: result.status, webhook_attempts: result.attempts, diff --git a/supabase/migrations/20260813120000_autoblog_quality_and_eeat.sql b/supabase/migrations/20260813120000_autoblog_quality_and_eeat.sql new file mode 100644 index 00000000..76a5f092 --- /dev/null +++ b/supabase/migrations/20260813120000_autoblog_quality_and_eeat.sql @@ -0,0 +1,41 @@ +-- Autoblog quality gate + E-E-A-T delivery fields. +-- +-- Two independent additions that ship together because both touch the +-- generate → deliver path: +-- +-- 1. lx_article.slop_score / slop_issues — the verdict from the pre-publish +-- gate (lib/lx/qualityGate.ts). Recorded on every accepted draft so a +-- site's content quality is a trend, not just a pass/fail at write time. +-- +-- 2. lx_site.author_name / author_url — the byline shipped with each post. +-- crawlproof's own audit penalises sites for missing author attribution +-- and Person markup (lib/audit/checks/content.ts "content.author"), while +-- the autoblog was delivering posts with author: null. These columns let +-- the webhook payload carry a real byline so receivers can render one. + +alter table public.lx_article + add column if not exists slop_score smallint, + add column if not exists slop_issues jsonb not null default '[]'::jsonb, + -- dateModified for Article/BlogPosting markup on the receiver. Distinct + -- from created_at: a regenerated or edited post keeps its original + -- published_at while updated_at moves. + add column if not exists updated_at timestamptz not null default now(); + +comment on column public.lx_article.slop_score is + 'Slop score 0-100 from the pre-publish quality gate. Lower is better; see lib/lx/qualityGate.ts.'; +comment on column public.lx_article.slop_issues is + 'SlopIssue[] recorded at generation time. Evidence for the score.'; + +-- Trend query support: "show me this site''s recent quality". +create index if not exists lx_article_site_slop_idx + on public.lx_article(site_id, created_at desc) + where slop_score is not null; + +alter table public.lx_site + add column if not exists author_name text, + add column if not exists author_url text; + +comment on column public.lx_site.author_name is + 'Byline shipped with each delivered post. Null means no author is asserted.'; +comment on column public.lx_site.author_url is + 'Author profile URL for the byline, used as schema.org Person.url / sameAs.'; diff --git a/tests/autoblog-quality-gate.test.ts b/tests/autoblog-quality-gate.test.ts new file mode 100644 index 00000000..625471ea --- /dev/null +++ b/tests/autoblog-quality-gate.test.ts @@ -0,0 +1,199 @@ +import { describe, it, expect } from "vitest"; +import { + DEFAULT_MAX_SLOP_SCORE, + runQualityGate, + type PriorBody, +} from "@/lib/lx/qualityGate"; +import { + buildRepairPrompt, + collectLinkViolations, +} from "@/lib/lx/articleGen"; +import { buildArticleJsonLd } from "@/lib/lx/webhookDeliver"; + +/** + * A body in the shape the autoblog actually emits: long, with a table, code, + * and concrete numbers — the signals that keep it out of "no first-party + * evidence". Padding sentences vary so the shingler does not see repetition + * within a single document. + */ +function goodBody(topic: string, seed = 0): string { + const paras: string[] = []; + for (let i = 0; i < 60; i++) { + const n = seed * 1000 + i; + paras.push( + `<p>Section ${n} of the ${topic} rollout measured ${n % 97} failed requests against a ${n % 43} second budget, which changed how the ${topic} team sequenced their ${n % 17} remaining migrations before the ${2000 + (n % 26)} cutover deadline.</p>`, + ); + } + return [ + `<h2>Why ${topic} breaks in production</h2>`, + ...paras, + "<table><tr><th>Approach</th><th>Latency</th></tr><tr><td>Batched</td><td>6ms</td></tr></table>", + "<pre><code>npx ctl apply --staged</code></pre>", + '<blockquote>Practical rule: measure before you shard.</blockquote>', + ].join("\n"); +} + +const BASE = { + title: "How teams actually run staged migrations", + metaDescription: + "A practical look at what breaks when teams sequence database migrations badly, and the workflow that fixes it.", +}; + +describe("runQualityGate", () => { + it("passes a substantive, unique article", () => { + const result = runQualityGate({ ...BASE, html: goodBody("migration") }); + expect(result.violations).toEqual([]); + expect(result.ok).toBe(true); + expect(result.score).toBeLessThanOrEqual(DEFAULT_MAX_SLOP_SCORE); + }); + + it("reports the word count it measured", () => { + const result = runQualityGate({ ...BASE, html: goodBody("migration") }); + expect(result.metrics.wordCount).toBeGreaterThan(500); + }); + + it("rejects filler phrasing and explains the fix", () => { + const filler = [ + "<p>In today's fast-paced world, it is important to note that we must delve into", + "the ever-evolving digital landscape. At the end of the day, this is a game changer", + "that will revolutionize the way you unlock the potential of your business and", + "take your business to the next level with a robust solution. In conclusion, look no", + "further — let's dive in and navigate the complexities of a seamless integration.</p>", + ].join(" "); + const result = runQualityGate({ ...BASE, html: goodBody("migration") + filler }); + + expect(result.ok).toBe(false); + expect(result.violations.join(" ")).toMatch(/filler/i); + expect(result.issues.map((i) => i.key)).toContain("content.filler"); + }); + + it("rejects a draft that repeats an existing post on the same blog", () => { + const shared = goodBody("migration"); + const priors: PriorBody[] = [{ slug: "staged-migrations-guide", body: shared }]; + const result = runQualityGate({ ...BASE, html: shared, priorBodies: priors }); + + expect(result.ok).toBe(false); + expect(result.issues.map((i) => i.key)).toContain("content.near_duplicate"); + // The violation must name the offending post so the repair pass can steer + // away from it, not just say "too similar". + expect(result.violations.join(" ")).toContain("staged-migrations-guide"); + }); + + it("accepts a draft that shares a topic but not its text", () => { + const priors: PriorBody[] = [ + { slug: "older-post", body: goodBody("caching", 7) }, + ]; + const result = runQualityGate({ + ...BASE, + html: goodBody("migration", 1), + priorBodies: priors, + }); + expect(result.violations).toEqual([]); + }); + + it("flags a thin draft the receiver would reject on word count", () => { + const result = runQualityGate({ + ...BASE, + html: "<p>Short post about migrations that says almost nothing at all.</p>", + }); + expect(result.ok).toBe(false); + expect(result.violations.join(" ")).toMatch(/word count|Thin/i); + }); + + it("ignores host-template concerns that are not the article's fault", () => { + // No viewport meta, a deprecated <center>, and an inline style — all + // properties of whatever page wraps our body, not of the body itself. + const result = runQualityGate({ + ...BASE, + html: `<center style="color:red">${goodBody("migration")}</center>`, + }); + expect(result.issues.map((i) => i.key)).not.toContain("code.deprecated_tags"); + expect(result.issues.map((i) => i.key)).not.toContain("design.no_viewport"); + }); + + it("does not count the table of contents against link density", () => { + // A 40-entry TOC of in-page anchors must not read as link spam. + const toc = Array.from( + { length: 40 }, + (_, i) => `<a href="#section-${i}">Section ${i}</a>`, + ).join(" "); + const result = runQualityGate({ ...BASE, html: toc + goodBody("migration") }); + expect(result.metrics.linkCount).toBe(0); + expect(result.violations).toEqual([]); + }); +}); + +describe("collectLinkViolations", () => { + const article = { + markdown_body: "Body text linking to [a page](https://x.test/a) and nothing else.", + used_internal_link_urls: ["https://x.test/a", "https://x.test/missing"], + used_exchange_link_urls: ["https://partner.test/p"], + // eslint-disable-next-line @typescript-eslint/no-explicit-any + } as any; + + it("names URLs claimed but not placed", () => { + const v = collectLinkViolations(article, new Set(["https://partner.test/p"])); + expect(v.join(" ")).toContain("https://x.test/missing"); + expect(v.join(" ")).not.toContain("https://x.test/a "); + }); + + it("rejects exchange URLs that were never offered", () => { + const v = collectLinkViolations(article, new Set()); + expect(v.join(" ")).toContain("not in the partner-blog candidate list"); + }); + + it("returns nothing when every claimed link is present", () => { + const clean = { + markdown_body: "Text with [a](https://x.test/a).", + used_internal_link_urls: ["https://x.test/a"], + used_exchange_link_urls: [], + // eslint-disable-next-line @typescript-eslint/no-explicit-any + } as any; + expect(collectLinkViolations(clean, new Set())).toEqual([]); + }); +}); + +describe("buildRepairPrompt", () => { + it("keeps the original brief and appends a numbered fix list", () => { + const out = buildRepairPrompt("ORIGINAL BRIEF", ["fix one", "fix two"]); + expect(out).toContain("ORIGINAL BRIEF"); + expect(out).toContain("1. fix one"); + expect(out).toContain("2. fix two"); + }); +}); + +describe("buildArticleJsonLd", () => { + const input = { + url: "https://blog.test/posts/staged-migrations", + title: "Staged migrations", + description: "What breaks and why.", + imageUrl: "https://cdn.test/hero.png", + publishedAt: "2026-08-01T10:00:00.000Z", + updatedAt: "2026-08-09T10:00:00.000Z", + authorName: "Dana Ruiz", + authorUrl: "https://blog.test/authors/dana", + publisherName: "blog.test", + tags: ["migrations", "databases"], + }; + + it("emits a Person byline with both dates", () => { + const html = buildArticleJsonLd(input); + const json = JSON.parse(html.replace(/^<script[^>]*>|<\/script>$/g, "")); + expect(json["@type"]).toBe("BlogPosting"); + expect(json.author).toMatchObject({ "@type": "Person", name: "Dana Ruiz" }); + expect(json.datePublished).toBe("2026-08-01T10:00:00.000Z"); + expect(json.dateModified).toBe("2026-08-09T10:00:00.000Z"); + }); + + it("falls back to an Organization byline when no author is configured", () => { + const html = buildArticleJsonLd({ ...input, authorName: null, authorUrl: null }); + const json = JSON.parse(html.replace(/^<script[^>]*>|<\/script>$/g, "")); + expect(json.author).toEqual({ "@type": "Organization", name: "blog.test" }); + }); + + it("escapes < so a title cannot close the script tag", () => { + const html = buildArticleJsonLd({ ...input, title: "</script><img onerror=x>" }); + expect(html).not.toContain("</script><img"); + expect(html.match(/<\/script>/g)).toHaveLength(1); + }); +});