diff --git a/lib/ads/creative.ts b/lib/ads/creative.ts index df03366f..df3d47cb 100644 --- a/lib/ads/creative.ts +++ b/lib/ads/creative.ts @@ -66,16 +66,23 @@ const CopySchema = z.object({ // The two prose lengths. Everything above is display copy sized for a box; // these are for placements that sit *inside* somebody's writing, where the ad // is read rather than glanced at. + // Generous hard caps, for the reason spelled out above `body`: the SDK strips + // maxLength from the schema it sends and validates client-side instead, so a + // cap set to the length we actually want makes the *whole generation throw* + // when the model runs a few characters over. It did — one campaign failed + // with `too_big` on a 400-char ceiling. The real limits are applied by + // cleanSummary after the fact, where going over is a trim rather than an + // error. Length guidance stays in the description, which is advisory. summaryShort: z .string() - .max(400) + .max(1200) .describe( "One or two plain sentences saying what this is and who it is for, in third person. " + "Reads as an editorial note, not a slogan — no exclamation marks, no second person, no CTA.", ), summaryLong: z .string() - .max(1600) + .max(6000) .describe( "Two or three short paragraphs, separated by a blank line, for a sponsored section of a " + "blog post. Third person, factual, specific to this product. Describe what it does, who " + @@ -85,6 +92,39 @@ const CopySchema = z.object({ }); type AdCopy = z.infer; +/** + * The summary half of CopySchema on its own, derived from it rather than + * retyped so the two can never disagree about lengths or descriptions. + */ +export const SummarySchema = CopySchema.pick({ summaryShort: true, summaryLong: true }); + +/** + * How the editorial summaries must be written. + * + * Split out because two callers need exactly these words: the full creative + * generation above, and `generateAdSummary` below, which the backfill script + * uses to fill in campaigns that predate the feature. If the two prompts + * drifted, a backfilled campaign would read differently from a freshly + * generated one and nobody would know why. + * + * The register instruction is doing real work. Without it the model returns + * three restatements of the headline, which is useless: these run inside other + * people's writing, next to their prose, and ad voice is exactly what makes + * them read as an intrusion and get skipped. + */ +export const SUMMARY_RULES: readonly string[] = [ + "Write two editorial summaries of the advertiser, in a different register from", + "display ad copy: third person, calm, factual, the way a journalist would", + "describe the product in one line of a round-up. No slogans, no second person, no", + "calls to action, no exclamation marks, no hype adjectives ('revolutionary',", + "'game-changing', 'seamless'). summaryShort is one or two sentences. summaryLong", + "is two or three short paragraphs separated by blank lines.", + "Never invent anything, and note this binds harder here than for a headline:", + "these are longer, so there is more room to fabricate. Every fact must come from", + "the page content. If the page does not say who it is for, or what it costs, or", + "how it works, then neither do you — write less rather than filling the space.", +]; + const SYSTEM_PROMPT = [ "You are a senior performance-marketing copywriter and brand designer.", "Given a company's website content and detected brand colours, write a single", @@ -94,20 +134,7 @@ const SYSTEM_PROMPT = [ "invent features, prices, or claims. Keep it concrete and specific to this product.", "Colours must be readable: high contrast between background and foreground.", "Prefer the site's real brand/accent colour when the palette makes it obvious.", - // The summaries are a different register from the display copy, and saying so - // explicitly is what stops the model returning three restatements of the - // headline. They run inside other people's writing, next to their prose, so - // ad voice is exactly what makes them read as an intrusion and get skipped. - "ALSO write two editorial summaries of the advertiser, in a different register", - "from the ad copy above: third person, calm, factual, the way a journalist would", - "describe the product in one line of a round-up. No slogans, no second person, no", - "calls to action, no exclamation marks, no hype adjectives ('revolutionary',", - "'game-changing', 'seamless'). summaryShort is one or two sentences. summaryLong", - "is two or three short paragraphs separated by blank lines.", - "The same no-invention rule binds them, and binds them harder: these are longer,", - "so there is more room to fabricate. Every fact must come from the page content.", - "If the page does not say who it is for, or what it costs, or how it works, then", - "neither do you — write less rather than filling the space.", + ...SUMMARY_RULES, ].join(" "); function buildUserPrompt(brand: SiteBrand): string { @@ -200,7 +227,127 @@ export function cleanSummary(v: unknown, maxLen: number): string { .map((line) => line.trim()) .join("\n") .trim(); - return text.slice(0, maxLen); + + if (text.length <= maxLen) return text; + + // Cut on a boundary. A hard slice lands mid-word about as often as not, and + // this prose is published as an advertiser's own description — "a deployment + // tool for small te" is worse than a sentence less. + const cut = text.slice(0, maxLen); + // Prefer a sentence break, and prefer it fairly eagerly: a clean sentence + // that loses a clause beats a fragment that ends mid-thought. The fraction + // only guards the pathological case where the sole break sits right at the + // start, which would throw away almost the whole budget. + const sentence = Math.max(cut.lastIndexOf(". "), cut.lastIndexOf(".\n")); + if (sentence > maxLen * 0.4) return cut.slice(0, sentence + 1).trim(); + const space = cut.lastIndexOf(" "); + return (space > 0 ? cut.slice(0, space) : cut).trim(); +} + +/** + * Just the two summaries, for a campaign that already has creatives. + * + * Deliberately not `generateAdCreatives`. That regenerates the whole concept — + * colours, four creatives, and a hero image through gpt-image when a Supabase + * client is passed — which for a backfill would be both far more expensive and + * actively destructive: it would replace copy an advertiser may have edited by + * hand. This reads the page, writes the prose, and touches nothing else. + * + * The prompt is the same SUMMARY_RULES the full generator uses, so a backfilled + * campaign reads like a freshly generated one. + */ +export async function generateAdSummary( + rawUrl: string, + opts: { anthropic?: Anthropic | null; openai?: OpenAI | null } = {}, +): Promise<{ summary: AdSummary; provider: string; title: string }> { + const brand = await extractSiteBrand(rawUrl); + + // Nothing readable came back, so there is nothing to summarise. Asked anyway, + // a model that has been told not to invent does the honest thing and + // describes *the fetch* — "a Tor .onion address; the page contains no + // readable text" — which is accurate, useless, and would be published inside + // somebody's blog post as though it were ad copy. An empty summary is the + // correct answer and every caller already treats it as "no prose". + if (readableTextLength(brand.text) < MIN_SUMMARY_SOURCE_CHARS) { + return { + summary: { short: "", long: "", domain: summaryDomain(brand.url || rawUrl) }, + provider: "skipped:no-content", + title: brand.title ?? "", + }; + } + + const anthropic = + opts.anthropic ?? (env.anthropicApiKey ? new Anthropic({ apiKey: env.anthropicApiKey }) : null); + const openai = + opts.openai ?? (env.openaiApiKey ? new OpenAI({ apiKey: env.openaiApiKey }) : null); + + const { provider, output } = await generateStructuredOutput<{ + summaryShort: string; + summaryLong: string; + }>({ + name: "ad_summary", + schema: SummarySchema, + system: ["You are a technology journalist writing a neutral product note.", ...SUMMARY_RULES].join( + " ", + ), + user: buildUserPrompt(brand), + maxTokens: 3000, + anthropicModel: CLAUDE_MODEL, + openaiModel: OPENAI_MODEL, + anthropic, + openai, + anthropicEffort: "low", + }); + + const short = cleanSummary(output.summaryShort, 400); + const long = cleanSummary(output.summaryLong, 1600); + + // Second line of defence. The input guard catches an empty page; this catches + // the page that had *some* text but not enough to describe, where the model + // writes about the document instead of the product. Either way the prose is + // discarded rather than published. + const usable = !describesTheFetch(short) && !describesTheFetch(long); + + return { + summary: { + short: usable ? short : "", + long: usable ? long : "", + domain: summaryDomain(brand.url || rawUrl), + }, + provider: usable ? provider : `rejected:meta:${provider}`, + title: brand.title ?? "", + }; +} + +/** Minimum readable characters on a page before it is worth summarising. */ +const MIN_SUMMARY_SOURCE_CHARS = 200; + +/** Words-worth of text, ignoring the whitespace a stripped page is mostly made of. */ +function readableTextLength(text: string | null | undefined): number { + return String(text ?? "").replace(/\s+/g, " ").trim().length; +} + +/** + * Does this summary describe the page we fetched rather than the product? + * + * A model told never to invent will, given an empty or unreadable page, write + * something true about the *document* — that it has no readable text, that it + * could not be accessed, that it appears to be a placeholder. True, and exactly + * what must never be rendered as an advertiser's description of themselves. + * + * Matching on phrases is crude, but the failure it guards is loud and narrow: + * real product copy does not talk about fetching, page content, or what could + * not be determined. + */ +export function __test_describesTheFetch(text: string): boolean { + return describesTheFetch(text); +} + +function describesTheFetch(text: string): boolean { + if (!text) return false; + return /\b(no readable (text|content)|the (fetched |retrieved )?page (contains|provides|has)\b|could not (be )?(access|retriev|fetch|determin|load)|unable to (access|determine|retrieve|load)|no information (about|is)|appears to be (empty|a placeholder|blank)|not enough information|no content (was )?(found|available)|placeholder page|under construction)/i.test( + text, + ); } export async function generateAdCreatives( diff --git a/scripts/backfill-ad-summaries.ts b/scripts/backfill-ad-summaries.ts new file mode 100644 index 00000000..80a70321 --- /dev/null +++ b/scripts/backfill-ad-summaries.ts @@ -0,0 +1,189 @@ +// Backfill the editorial summaries for campaigns that predate them. +// +// npx tsx scripts/backfill-ad-summaries.ts --env ~/crawlproof-env-backup.txt --dry-run +// npx tsx scripts/backfill-ad-summaries.ts --env ~/crawlproof-env-backup.txt --limit 5 +// npx tsx scripts/backfill-ad-summaries.ts --env ~/crawlproof-env-backup.txt +// +// TypeScript rather than the .mjs the other scripts use, and run through tsx, +// so it can import `generateAdSummary` from lib/ads/creative — the same prompt +// the live generator uses. Reimplementing the prompt here in plain JS would +// mean backfilled prose reading differently from freshly generated prose, with +// nothing to say why. +// +// Safe to re-run. It only selects campaigns that still need prose, so an +// interrupted pass continues where it stopped, and a campaign whose summary was +// written by hand afterwards is left alone. + +import { readFileSync } from "node:fs"; +import { createClient } from "@supabase/supabase-js"; + +// --- environment ------------------------------------------------------------ +// Loaded before importing anything that reads `env`, because lib/env captures +// process.env at module scope: a top-level import of creative.ts would see an +// empty key and construct no client at all. +const args = process.argv.slice(2); +const flag = (name: string): string | null => { + const i = args.indexOf(`--${name}`); + return i >= 0 ? (args[i + 1] ?? "") : null; +}; +const has = (name: string) => args.includes(`--${name}`); + +const envPath = flag("env") ?? `${process.env.HOME}/crawlproof-env-backup-2026-07-28.txt`; +const dryRun = has("dry-run"); +const limit = Number(flag("limit") ?? "0") || 0; +/** How many sites to read and summarise at once. */ +const CONCURRENCY = Number(flag("concurrency") ?? "3") || 3; + +for (const [k, v] of Object.entries(readEnvFile(envPath))) { + if (!process.env[k]) process.env[k] = v; +} + +function readEnvFile(path: string): Record { + try { + return Object.fromEntries( + readFileSync(path, "utf8") + .split("\n") + .filter((l) => l.trim() && !l.trim().startsWith("#") && l.includes("=")) + .map((l) => { + const i = l.indexOf("="); + return [l.slice(0, i).trim(), l.slice(i + 1).trim().replace(/^["']|["']$/g, "")]; + }), + ); + } catch (err) { + console.error(`Could not read env file ${path}: ${(err as Error).message}`); + process.exit(1); + } +} + +// Imported after the env is populated, for the reason above. +const { generateAdSummary } = await import("../lib/ads/creative"); + +// --- database --------------------------------------------------------------- + +const url = process.env.NEXT_PUBLIC_SUPABASE_URL; +const key = process.env.SUPABASE_SERVICE_ROLE_KEY; +if (!url || !key) { + console.error("Missing NEXT_PUBLIC_SUPABASE_URL or SUPABASE_SERVICE_ROLE_KEY."); + process.exit(1); +} +const sb = createClient(url, key, { auth: { persistSession: false } }); + +type Row = { + id: string; + name: string; + destination_url: string; + destination_domain: string | null; + summary_short: string | null; + summary_domain: string | null; +}; + +// Everything without usable prose: never generated, or generated for a domain +// the campaign no longer points at (serving treats that as absent, so it is +// exactly as unhelpful as never having had one). +const { data, error } = await sb + .from("ad_campaigns") + .select("id, name, destination_url, destination_domain, summary_short, summary_domain") + .order("created_at", { ascending: true }); + +if (error) { + console.error("Could not list campaigns:", error.message); + process.exit(1); +} + +const stale = (r: Row) => + !r.summary_short || + !r.summary_domain || + (r.destination_domain ?? "").toLowerCase() !== r.summary_domain.toLowerCase(); + +let todo = (data as Row[]).filter(stale); +if (limit > 0) todo = todo.slice(0, limit); + +console.log( + `${data.length} campaigns, ${(data as Row[]).filter(stale).length} need prose` + + (limit > 0 ? `, running the first ${todo.length}` : "") + + (dryRun ? " (DRY RUN — nothing will be written)" : ""), +); +if (todo.length === 0) process.exit(0); + +// --- the pass --------------------------------------------------------------- + +let done = 0; +let written = 0; +let failed = 0; +const failures: Array<{ name: string; url: string; why: string }> = []; + +async function one(row: Row): Promise { + const n = ++done; + const label = `[${n}/${todo.length}] ${row.name.slice(0, 40)}`; + try { + const { summary, provider } = await generateAdSummary(row.destination_url); + + // generateAdSummary returns empty prose for a page it could not read, and + // for prose that described the fetch rather than the product. Both are a + // deliberate "no summary" rather than a failure, and the provider string + // says which — but either way there is nothing to store. + if (!summary.short && !summary.long) { + throw new Error( + provider.startsWith("skipped:") || provider.startsWith("rejected:") + ? provider + : "model returned nothing usable", + ); + } + // The same guard the server action applies: prose is only stored when it + // describes where the campaign actually points. A redirect to another host + // is the usual way this trips. + const points = (row.destination_domain ?? "").toLowerCase(); + if (points && summary.domain && summary.domain !== points) { + throw new Error(`resolved to ${summary.domain}, campaign points at ${points}`); + } + + if (dryRun) { + console.log(`${label} — would write (${provider})`); + console.log(` short: ${summary.short.slice(0, 110)}`); + console.log(` long : ${summary.long.split("\n\n").length} paragraphs`); + return; + } + + const { error: uErr } = await sb + .from("ad_campaigns") + .update({ + summary_short: summary.short || null, + summary_long: summary.long || null, + summary_domain: summary.domain || points || null, + summary_generated_at: new Date().toISOString(), + }) + .eq("id", row.id); + if (uErr) throw new Error(uErr.message); + + written += 1; + console.log(`${label} — ok (${provider}, ${summary.long.split("\n\n").length}¶)`); + } catch (err) { + failed += 1; + const why = err instanceof Error ? err.message : String(err); + failures.push({ name: row.name, url: row.destination_url, why }); + // A dead or hostile site is ordinary in a list this old; it must not stop + // the pass, and the campaign simply keeps falling back to its short body. + console.log(`${label} — SKIP: ${why.slice(0, 120)}`); + } +} + +// A small fixed pool. These calls each fetch a third-party site and then hit an +// LLM, so the limit is politeness to the sites as much as rate-limit safety. +const queue = [...todo]; +await Promise.all( + Array.from({ length: Math.min(CONCURRENCY, queue.length) }, async () => { + for (;;) { + const row = queue.shift(); + if (!row) return; + await one(row); + } + }), +); + +console.log( + `\nDone. ${written} written, ${failed} skipped, ${todo.length} attempted${dryRun ? " (dry run)" : ""}.`, +); +if (failures.length > 0) { + console.log("\nSkipped:"); + for (const f of failures) console.log(` ${f.name.slice(0, 40)} — ${f.url} — ${f.why.slice(0, 90)}`); +} diff --git a/tests/ads-summary-domain.test.ts b/tests/ads-summary-domain.test.ts index 47c1b25a..183a9f02 100644 --- a/tests/ads-summary-domain.test.ts +++ b/tests/ads-summary-domain.test.ts @@ -55,3 +55,59 @@ describe("cleanSummary", () => { expect(summaryParagraphs(cleanSummary("a\r\n\r\nb", 400))).toEqual(["a", "b"]); }); }); + +describe("summaries that describe the page instead of the product", () => { + // A model told never to invent will, handed an empty or unreadable page, + // write something true about the *document*. Accurate, useless, and the last + // thing that should be published inside somebody's blog post as though the + // advertiser wrote it. Caught on the very first dry run against real data: + // a .onion campaign produced "a Tor .onion address; the fetched page contains + // no readable text". + const meta = [ + "A Tor .onion address; the fetched page contains no readable text and provides no information.", + "The page could not be accessed, so no description is possible.", + "This appears to be a placeholder page with no content found.", + "Unable to determine what this product does from the page.", + "There is not enough information on the page to describe the service.", + ]; + + const real = [ + "Widgets is a deployment tool for small teams that do not run a platform group.", + "CoinPay is an open-source, non-custodial crypto payment gateway with escrow and web wallets.", + "The rollback path is the same one used to deploy, so it is exercised on every release.", + // Must not trip on ordinary copy that merely mentions pages or content. + "A page builder for marketing sites, with content blocks you can reorder.", + ]; + + it("recognises prose about the fetch", async () => { + const { __test_describesTheFetch: fn } = await import("@/lib/ads/creative"); + for (const t of meta) expect(fn(t), t).toBe(true); + }); + + it("leaves real product copy alone", async () => { + const { __test_describesTheFetch: fn } = await import("@/lib/ads/creative"); + for (const t of real) expect(fn(t), t).toBe(false); + }); +}); + +describe("truncation", () => { + // The schema caps are deliberately generous because the SDK validates + // maxLength client-side and throws the whole generation on an overrun; the + // real limit is applied here, where going over is a trim rather than an error. + it("cuts on a sentence boundary when there is one", () => { + const text = "First sentence here. Second sentence runs past the limit and keeps going."; + const out = cleanSummary(text, 40); + expect(out).toBe("First sentence here."); + }); + + it("falls back to a word boundary rather than cutting mid-word", () => { + const out = cleanSummary("a deployment tool for small teams everywhere", 20); + expect(out.endsWith("te")).toBe(false); + expect(out.length).toBeLessThanOrEqual(20); + expect(/\s$/.test(out)).toBe(false); + }); + + it("leaves text under the cap untouched", () => { + expect(cleanSummary("short enough", 400)).toBe("short enough"); + }); +});