diff --git a/app/(marketing)/slop/opengraph-image.tsx b/app/(marketing)/slop/opengraph-image.tsx new file mode 100644 index 00000000..a0f728be --- /dev/null +++ b/app/(marketing)/slop/opengraph-image.tsx @@ -0,0 +1,75 @@ +import { ImageResponse } from "next/og"; + +// Static card for the /slop landing page. No data fetch, so Next generates it +// once at build time. + +export const alt = "Slop Score — free carelessness scan for your site"; +export const size = { width: 1200, height: 630 }; +export const contentType = "image/png"; + +const BG = "#0b0d10"; +const FG = "#e7e9ee"; +const MUTED = "#9aa3b2"; +const BORDER = "#1f2630"; +const ACCENT = "#6ee7b7"; +const WARN = "#fbbf24"; +const FAIL = "#f87171"; + +export default function Image() { + return new ImageResponse( + ( +
+
+
+ CRAWLPROOF +
+
+ FREE · NO SIGNUP +
+
+ +
+
+ Find the careless +
+
+ mistakes on your site. +
+
+ Sweeps up to 50 pages for observable defects — content, code, design. +
+ + {/* The banded dial, matching the page: 0 pristine → 100 maximum slop. */} +
+
+
+
+
+
+
0 — pristine
+
100 — maximum slop
+
+
+ +
+
+ Not an AI-detector — observable defects only +
+
crawlproof.com/slop
+
+
+ ), + size, + ); +} diff --git a/app/(marketing)/slop/page.tsx b/app/(marketing)/slop/page.tsx new file mode 100644 index 00000000..18634eef --- /dev/null +++ b/app/(marketing)/slop/page.tsx @@ -0,0 +1,208 @@ +import Link from "next/link"; +import { HeroAuditForm } from "@/components/hero-audit-form"; + +export const metadata = { + title: "Slop Score — free carelessness scan for your site", + description: + "Free scan of up to 50 pages for observable defects: placeholder copy left in production, near-duplicate pages, leaked template variables, dead links, stale dates, missing alt text, and design drift. No signup, no LLM, no guesswork.", + alternates: { canonical: "/slop" }, + openGraph: { + title: "Slop Score — free carelessness scan", + description: + "Sweep up to 50 pages for the careless mistakes that ship to production. Free, no signup.", + url: "/slop", + type: "website", + }, + // This block has to be declared even though it only restates the openGraph + // one. Without it the page inherits the ROOT LAYOUT's twitter.images, which + // outranks the generated ./opengraph-image.tsx — so X alone would fall back + // to the generic /banner.png while every other platform showed the card. + // Neither block declares `images`: the file convention supplies them. + twitter: { + card: "summary_large_image", + title: "Slop Score — free carelessness scan", + description: + "Sweep up to 50 pages for the careless mistakes that ship to production. Free, no signup.", + }, +}; + +// The three dimensions, and their hints, mirror components/report/slop-meter.tsx +// so the landing page and the report describe the same thing in the same words. +const DIMENSIONS = [ + { + label: "Content", + hint: "Filler and placeholder copy, thin pages, near-duplicate pages, stale dates, claims with no evidence behind them.", + }, + { + label: "Code", + hint: "Leaked template variables, preview-host URLs pointing at staging, dead links, broken or missing resources.", + }, + { + label: "Design", + hint: "Missing viewport, missing alt text, layout-shift risks, palette and type sprawl across pages.", + }, +]; + +const EXAMPLES = [ + { + finding: "“Coming soon” shipped to production", + detail: + "A standalone element still reading as a placeholder, months after launch — on a page that is linked from your nav.", + }, + { + finding: "Leaked template variable", + detail: + "A literal {{product_name}} or [insert company] rendered to visitors because a merge field never resolved.", + }, + { + finding: "Preview host in a live link", + detail: + "A production page linking to your-app.vercel.app — the staging copy, indexed and reachable.", + }, + { + finding: "Near-duplicate pages", + detail: + "Six location pages that differ only by the city name, splitting their own ranking signal.", + }, +]; + +export default function SlopPage() { + return ( +
+
+

+ Free · No signup · Up to 50 pages +

+

+ Find the careless mistakes on your site. +

+

+ The Slop Score sweeps up to 50 pages and reports what is{" "} + + observably broken or sloppy + {" "} + — placeholder copy that shipped, template variables that never + resolved, staging URLs in live links, duplicate pages, stale dates, + missing alt text. One score, and a per-page list of what to fix. +

+
+ +
+

+ Runs no AI model, so there is nothing to queue behind and nothing to + pay for. The report opens on-page in seconds and is yours to share. +

+
+ +
+

How the score reads

+

+ The dial runs the opposite way to a grade: 0 is pristine, + 100 is maximum slop. Lower is better. +

+
+
+
+
+

+ 0–25 · Clean +

+

+ Nothing obviously careless. Fix the stragglers and move on. +

+
+
+

+ 26–50 · Careless +

+

+ Real defects a visitor can hit. Usually a handful of pages doing + most of the damage. +

+
+
+

+ 51–100 · Sloppy +

+

+ Systemic — the same mistake repeated across templates rather + than one bad page. +

+
+
+
+
+ +
+

What gets swept

+
+ {DIMENSIONS.map((d) => ( +
+

{d.label}

+

{d.hint}

+
+ ))} +
+
+ +
+

The kind of thing it catches

+

+ Every finding points at a specific page and a specific line of + evidence — never a vibe. +

+
+ {EXAMPLES.map((e) => ( +
+

{e.finding}

+

{e.detail}

+
+ ))} +
+
+ + {/* The positioning guardrail, stated as a feature. This is deliberate and + load-bearing: an "is this AI-written?" score is unfalsifiable, misfires + on non-native English writers, and would accuse paying customers. + tests/slop.test.ts guards the engine against drifting into it. */} +
+
+

What this is not

+

+ This is not an AI-detector. We + never estimate whether a human or a model wrote your page, and we + never report a probability that it did. Those classifiers cannot be + checked against the truth, and they misfire hardest on people + writing in a second language. +

+

+ Every Slop Score finding is an{" "} + observable defect — something you + can open the page and see for yourself. If you disagree with one, + you can prove us wrong, which is the whole point. +

+
+
+ +
+

Want the AI-crawler view too?

+

+ The Slop Score covers carelessness. The free AEO audit covers what + ChatGPT, Claude, Perplexity, and Google AI Overviews can actually find + on your site — schema, robots rules, AI-bot access, and positioning. +

+ + Run a free AEO audit + +
+
+ ); +} diff --git a/app/actions/watchScan.ts b/app/actions/watchScan.ts new file mode 100644 index 00000000..0ad64550 --- /dev/null +++ b/app/actions/watchScan.ts @@ -0,0 +1,226 @@ +"use server"; + +import { headers } from "next/headers"; +import { serviceClient } from "@/lib/supabase/service"; +import { hashIp } from "@/lib/rateLimit"; +import { newShareToken } from "@/lib/shareToken"; +import { recordLead } from "@/lib/marketing"; +import { sendWatchConfirmEmail } from "@/lib/email"; +import { buildShareCard } from "@/lib/audit/share-card"; +import { + MAX_WATCHES_PER_EMAIL, + isWatchCadence, + isWatchEngine, + normalizeWatchEmail, + type WatchCadence, + type WatchEngine, +} from "@/lib/watches"; +import { env } from "@/lib/env"; + +// A watch is a standing promise to email an address. Both caps below exist so +// that promise can't be manufactured in bulk against addresses that never +// asked — the confirmation link is the real defence, but these keep the +// confirmation mail itself from becoming the abuse vector. +const WATCHES_PER_IP_PER_DAY = 5; + +type Ok = { ok: true }; +type Err = { ok: false; error: string }; + +/** + * Create (or refresh) a watch from a public report page. + * + * Always returns the same neutral success, and always sends the same + * confirmation mail, whether or not this address already watches this URL. + * Reporting "you already watch this" would let anyone probe which addresses + * are watching which sites, and confirming twice is harmless. + */ +export async function createWatch(input: { + token: string; + email: string; + cadence?: string; +}): Promise { + const email = normalizeWatchEmail(input.email); + if (!email) return { ok: false, error: "Enter a valid email address." }; + + const cadence: WatchCadence = isWatchCadence(input.cadence) ? input.cadence : "weekly"; + + const svc = serviceClient(); + + const { data: audit } = await svc + .from("audits") + .select("id, target_url, status, score, engine, summary") + .eq("share_token", input.token) + .maybeSingle(); + if (!audit) return { ok: false, error: "That report link is no longer valid." }; + + // Only the free, self-hosted engines can be put on a recurring schedule, so + // a watch can never become recurring LLM spend on an address we never + // charged. A report from any other engine falls back to the rule engine. + const engine: WatchEngine = isWatchEngine(audit.engine) ? audit.engine : "rule"; + const targetUrl = audit.target_url as string; + + // Respect the global marketing suppression list. Someone who unsubscribed + // from CrawlProof entirely should not be able to be re-subscribed by a + // form, even one they filled in themselves. + // ilike here matches the rest of lib/marketing.ts, whose rows predate any + // normalization guarantee. Over-matching only ever errs toward suppressing + // a send, which is the safe direction for a suppression check. + const { data: contact } = await svc + .from("marketing_contacts") + .select("unsubscribed_at") + .ilike("email", email) + .maybeSingle(); + if (contact?.unsubscribed_at) { + return { + ok: false, + error: "That address has unsubscribed from CrawlProof email. Reply to any past email to re-enable it.", + }; + } + + const hdrs = await headers(); + const ip = + hdrs.get("x-forwarded-for")?.split(",")[0]?.trim() || hdrs.get("x-real-ip") || "unknown"; + const ipHash = hashIp(ip); + + const since = new Date(Date.now() - 24 * 60 * 60 * 1000).toISOString(); + const { count: fromIp } = await svc + .from("scan_watches") + .select("id", { count: "exact", head: true }) + .eq("created_ip_hash", ipHash) + .gte("created_at", since); + if ((fromIp ?? 0) >= WATCHES_PER_IP_PER_DAY) { + return { ok: false, error: "Too many watch requests from this network today. Try again tomorrow." }; + } + + // eq, not ilike: `_` and `%` are ILIKE wildcards and both are legal in an + // address, so `john_doe@x.com` would match `johnXdoe@x.com` and wrongly + // count against someone else's cap. This column always stores the + // normalized form, so an exact match is both correct and sufficient. + const { data: existingRows } = await svc + .from("scan_watches") + .select("id, target_url, engine") + .eq("email", email) + .is("unsubscribed_at", null); + + const existing = (existingRows ?? []).find( + (r) => r.target_url === targetUrl && r.engine === engine, + ); + + if (!existing && (existingRows ?? []).length >= MAX_WATCHES_PER_EMAIL) { + return { + ok: false, + error: `That address already watches ${MAX_WATCHES_PER_EMAIL} sites, which is the limit.`, + }; + } + + const confirmToken = newShareToken(); + + if (existing) { + // Refresh rather than stack a duplicate: new confirm token, restored from + // any prior unsubscribe, cadence updated to whatever was just chosen. + const { error } = await svc + .from("scan_watches") + .update({ + cadence, + confirm_token: confirmToken, + unsubscribed_at: null, + created_ip_hash: ipHash, + }) + .eq("id", existing.id); + if (error) return { ok: false, error: "Could not set up that watch." }; + } else { + const { error } = await svc.from("scan_watches").insert({ + email, + target_url: targetUrl, + engine, + cadence, + confirm_token: confirmToken, + unsubscribe_token: newShareToken(), + origin_audit_id: audit.id, + // Seed the baseline from the report they are looking at, so the first + // re-scan can be reported as a real change rather than a first sighting. + last_score: buildShareCard(audit as Parameters[0]).score, + created_ip_hash: ipHash, + }); + if (error) return { ok: false, error: "Could not set up that watch." }; + } + + // A captured email, not a marketing opt-in — recordLead never upgrades + // consent, and watch mail is transactional (they asked for this specific + // thing about this specific URL). + await recordLead({ email, source: "watch" }); + + const card = buildShareCard(audit as Parameters[0]); + const base = env.siteUrl.replace(/\/$/, ""); + const res = await sendWatchConfirmEmail({ + to: email, + host: card.host, + label: card.label, + cadence, + confirmUrl: `${base}/watch/confirm/${confirmToken}`, + }); + if (!res.sent) { + return { ok: false, error: "Could not send the confirmation email. Try again shortly." }; + } + + return { ok: true }; +} + +/** Confirm a watch. Idempotent — clicking twice is a no-op, not an error. */ +export async function confirmWatchByToken( + token: string, +): Promise<{ ok: boolean; host?: string; cadence?: string }> { + if (!token || token.length < 8) return { ok: false }; + const svc = serviceClient(); + const { data: row } = await svc + .from("scan_watches") + .select("id, target_url, cadence, verified_at") + .eq("confirm_token", token) + .maybeSingle(); + if (!row) return { ok: false }; + + const host = (() => { + try { + return new URL(row.target_url as string).hostname.replace(/^www\./, ""); + } catch { + return row.target_url as string; + } + })(); + + if (!row.verified_at) { + await svc + .from("scan_watches") + .update({ + verified_at: new Date().toISOString(), + unsubscribed_at: null, + // Start the clock at confirmation, not at request time. + next_run_at: new Date().toISOString(), + }) + .eq("id", row.id); + } + + return { ok: true, host, cadence: row.cadence as string }; +} + +/** Stop a watch. Also idempotent, and reachable from every email we send. */ +export async function stopWatchByToken( + token: string, +): Promise<{ ok: boolean; host?: string }> { + if (!token || token.length < 8) return { ok: false }; + const svc = serviceClient(); + const { data: row } = await svc + .from("scan_watches") + .update({ unsubscribed_at: new Date().toISOString() }) + .eq("unsubscribe_token", token) + .select("target_url") + .maybeSingle(); + if (!row) return { ok: false }; + const host = (() => { + try { + return new URL(row.target_url as string).hostname.replace(/^www\./, ""); + } catch { + return row.target_url as string; + } + })(); + return { ok: true, host }; +} diff --git a/app/api/cron/scan-watches/route.ts b/app/api/cron/scan-watches/route.ts new file mode 100644 index 00000000..e3760408 --- /dev/null +++ b/app/api/cron/scan-watches/route.ts @@ -0,0 +1,205 @@ +import { NextResponse } from "next/server"; +import { serviceClient } from "@/lib/supabase/service"; +import { newShareToken } from "@/lib/shareToken"; +import { buildShareCard } from "@/lib/audit/share-card"; +import { sendWatchChangeEmail } from "@/lib/email"; +import { getProspectsOrgId } from "@/lib/orgs"; +import { nextRunAt, watchSubject, watchVerdict, type WatchCadence } from "@/lib/watches"; +import { env } from "@/lib/env"; + +export const runtime = "nodejs"; + +const ENQUEUE_BATCH = 50; +const DELIVER_BATCH = 100; + +type WatchRow = { + id: string; + email: string; + target_url: string; + engine: string; + cadence: string; + unsubscribe_token: string; + pending_audit_id: string | null; + last_score: number | null; +}; + +export async function GET(req: Request) { + return POST(req); +} + +/** + * One tick does two things, in this order: + * + * 1. DELIVER — a re-scan enqueued on an earlier tick has finished, so + * compare it to the stored baseline and email only if it really moved. + * 2. ENQUEUE — start re-scans whose cadence has come due. + * + * Deliver runs first so a scan that finished since the last tick is reported + * before that watch is considered for its next run. + */ +export async function POST(req: Request) { + const incoming = + req.headers.get("x-cron-secret") ?? + req.headers.get("authorization")?.replace(/^Bearer\s+/i, ""); + if (incoming !== env.cronSecret) { + return NextResponse.json({ ok: false, error: "unauthorized" }, { status: 401 }); + } + + const svc = serviceClient(); + const base = env.siteUrl.replace(/\/$/, ""); + const now = new Date(); + + let delivered = 0; + let unchanged = 0; + let scan_failed = 0; + let enqueued = 0; + + // ---- 1. Deliver finished re-scans ------------------------------------- + const { data: pending } = await svc + .from("scan_watches") + .select("id, email, target_url, engine, cadence, unsubscribe_token, pending_audit_id, last_score") + .not("pending_audit_id", "is", null) + .is("unsubscribed_at", null) + .limit(DELIVER_BATCH); + + for (const w of (pending ?? []) as WatchRow[]) { + const { data: audit } = await svc + .from("audits") + .select("target_url, status, score, engine, summary, share_token") + .eq("id", w.pending_audit_id!) + .maybeSingle(); + + // Row vanished — clear the pointer so the watch isn't wedged forever. + if (!audit) { + await svc.from("scan_watches").update({ pending_audit_id: null }).eq("id", w.id); + continue; + } + if (audit.status === "queued" || audit.status === "running") continue; + + if (audit.status !== "complete") { + // A failed scan is our problem, not news for the subscriber. Clear it + // and let the next cadence tick try again. + scan_failed++; + await svc + .from("scan_watches") + .update({ pending_audit_id: null, last_scanned_at: now.toISOString() }) + .eq("id", w.id); + continue; + } + + const card = buildShareCard(audit as Parameters[0]); + if (card.score === null) { + await svc.from("scan_watches").update({ pending_audit_id: null }).eq("id", w.id); + continue; + } + + const verdict = watchVerdict({ + engineKind: card.kind, + previousScore: w.last_score, + nextScore: card.score, + }); + + const update: Record = { + pending_audit_id: null, + last_scanned_at: now.toISOString(), + // Re-baseline on every completed scan, notified or not. Otherwise a + // series of sub-threshold drifts would never add up to an email, and + // then one day report a jump that never happened in one step. + last_score: card.score, + }; + + if (verdict.notify) { + const res = await sendWatchChangeEmail({ + to: w.email, + subject: watchSubject({ + host: card.host, + label: card.label, + score: card.score, + verdict, + }), + host: card.host, + label: card.label, + score: card.score, + previousScore: w.last_score, + improved: verdict.improved, + first: verdict.kind === "first", + scaleHint: card.scaleHint, + reportUrl: `${base}/r/${audit.share_token}`, + // The API URL, not the page: RFC 8058 requires List-Unsubscribe and + // List-Unsubscribe-Post to name the same URL, and only the route + // handler can answer the client's POST. A human clicking it gets + // redirected to the page. + stopUrl: `${base}/api/watch/stop/${w.unsubscribe_token}`, + cadence: w.cadence, + }); + if (res.sent) { + update.last_notified_at = now.toISOString(); + delivered++; + } + } else { + unchanged++; + } + + await svc.from("scan_watches").update(update).eq("id", w.id); + } + + // ---- 2. Enqueue due re-scans ------------------------------------------ + const { data: due, error } = await svc + .from("scan_watches") + .select("id, email, target_url, engine, cadence, unsubscribe_token, pending_audit_id, last_score") + .not("verified_at", "is", null) + .is("unsubscribed_at", null) + .is("pending_audit_id", null) + .lt("next_run_at", now.toISOString()) + .limit(ENQUEUE_BATCH); + if (error) return NextResponse.json({ ok: false, error: error.message }, { status: 500 }); + + // Watch scans are anonymous by design (no account behind them), so tag them + // to the Prospects org — the same bucket the hero form's anonymous scans go + // to, which is what makes them workable from /recent. + const prospectsOrgId = await getProspectsOrgId().catch(() => null); + + for (const w of (due ?? []) as WatchRow[]) { + const next = nextRunAt(w.cadence as WatchCadence, now).toISOString(); + + const { data: row, error: insErr } = await svc + .from("audits") + .insert({ + target_url: w.target_url, + owner_id: null, + organization_id: prospectsOrgId, + status: "queued", + share_token: newShareToken(), + triggered_by: "watch", + engine: w.engine, + }) + .select("id") + .maybeSingle(); + + if (insErr || !row) { + // Push the schedule out anyway so one bad row can't be retried every + // tick forever. + await svc.from("scan_watches").update({ next_run_at: next }).eq("id", w.id); + continue; + } + + await svc + .from("scan_watches") + .update({ pending_audit_id: row.id, next_run_at: next }) + .eq("id", w.id); + + if (env.workerUrl) { + fetch(`${env.workerUrl}/enqueue`, { + method: "POST", + headers: { + "content-type": "application/json", + "x-worker-secret": env.workerSecret, + }, + body: JSON.stringify({ auditId: row.id }), + }).catch(() => {}); + } + enqueued++; + } + + return NextResponse.json({ ok: true, delivered, unchanged, scan_failed, enqueued }); +} diff --git a/app/api/watch/stop/[token]/route.ts b/app/api/watch/stop/[token]/route.ts new file mode 100644 index 00000000..2fe3715b --- /dev/null +++ b/app/api/watch/stop/[token]/route.ts @@ -0,0 +1,31 @@ +import { NextResponse } from "next/server"; +import { stopWatchByToken } from "@/app/actions/watchScan"; +import { env } from "@/lib/env"; + +export const runtime = "nodejs"; + +// RFC 8058 one-click unsubscribe. Both List-Unsubscribe and +// List-Unsubscribe-Post must name the SAME URL, and mail clients issue a POST +// to it — which a page route can't answer. So this is the URL in the headers: +// a POST performs the stop and returns 200, a GET (a human clicking the link +// in the footer) redirects to the page that explains what happened. + +export async function POST( + _req: Request, + { params }: { params: Promise<{ token: string }> }, +) { + const { token } = await params; + await stopWatchByToken(token); + // Always 200, even for an unknown token: the sender must not be able to + // learn which tokens are live by watching status codes. + return NextResponse.json({ ok: true }); +} + +export async function GET( + _req: Request, + { params }: { params: Promise<{ token: string }> }, +) { + const { token } = await params; + const base = env.siteUrl.replace(/\/$/, ""); + return NextResponse.redirect(`${base}/watch/stop/${encodeURIComponent(token)}`, 302); +} diff --git a/app/r/[token]/opengraph-image.tsx b/app/r/[token]/opengraph-image.tsx new file mode 100644 index 00000000..a47af0a2 --- /dev/null +++ b/app/r/[token]/opengraph-image.tsx @@ -0,0 +1,173 @@ +import { ImageResponse } from "next/og"; +import { serviceClient } from "@/lib/supabase/service"; +import { buildShareCard, type ShareCard } from "@/lib/audit/share-card"; + +// Per-report social card. Before this existed every /r/ link shared the +// one static /banner.png, so a report for acme.com and a report for +// example.org produced byte-identical previews in Slack, X, and LinkedIn. +// The card's value is entirely in naming the scanned site and its score. + +export const runtime = "nodejs"; +// Scans complete asynchronously, so a card fetched seconds after the share +// link may still read "Scan running…". Short revalidate lets it settle without +// re-querying on every crawler hit. +export const revalidate = 60; + +export const alt = "CrawlProof report card"; +export const size = { width: 1200, height: 630 }; +export const contentType = "image/png"; + +const BG = "#0b0d10"; +const FG = "#e7e9ee"; +const MUTED = "#9aa3b2"; +const BORDER = "#1f2630"; +const ACCENT = "#6ee7b7"; + +const TONE: Record = { + pass: "#34d399", + warn: "#fbbf24", + fail: "#f87171", + neutral: "#64748b", +}; + +/** Long hostnames must not wrap or overflow — step the size down instead. */ +function hostSize(host: string): number { + if (host.length > 30) return 52; + if (host.length > 22) return 66; + if (host.length > 15) return 82; + return 96; +} + +export default async function Image({ params }: { params: Promise<{ token: string }> }) { + const { token } = await params; + + let card: ShareCard | null = null; + try { + const svc = serviceClient(); + const { data } = await svc + .from("audits") + .select("target_url, status, score, engine, summary") + .eq("share_token", token) + .maybeSingle(); + if (data) card = buildShareCard(data as Parameters[0]); + } catch { + // A card is decoration on a link preview — never let a DB blip 500 the + // image and leave the crawler with no thumbnail at all. + card = null; + } + + const tone = TONE[card?.tone ?? "neutral"]; + + return new ImageResponse( + ( +
+
+
+ CRAWLPROOF +
+
+ {card ? card.label.toUpperCase() : "SITE AUDIT"} +
+
+ + {card ? ( +
+
+ {card.host} +
+ +
+ {card.score !== null ? ( +
+
+ {card.score} +
+
+ /100 +
+
+ ) : ( +
+ {card.headline} +
+ )} + {card.score !== null && ( +
+ {card.headline} +
+ )} +
+ + {card.score !== null && ( +
+
+
+
+
+ {card.scaleHint} +
+
+ )} +
+ ) : ( +
+
+ See your site the way AI crawlers do. +
+
+ )} + +
+
+ {card ? card.footer : "SEO · AEO · GEO audit — free, no signup"} +
+
crawlproof.com
+
+
+ ), + size, + ); +} diff --git a/app/r/[token]/page.tsx b/app/r/[token]/page.tsx index c26e38be..f1beaa00 100644 --- a/app/r/[token]/page.tsx +++ b/app/r/[token]/page.tsx @@ -15,10 +15,12 @@ import { LivePoller } from "@/components/report/live-poller"; import { CopyLink } from "@/components/copy-link"; import { ShareBanner } from "@/components/share-banner"; import { EmailReportForm } from "@/components/report/email-report-form"; +import { WatchForm } from "@/components/report/watch-form"; import { createClient } from "@/lib/supabase/server"; import { serviceClient } from "@/lib/supabase/service"; import type { Finding } from "@/lib/audit/types"; import { loadConsolidatedOrSoloMarkdown } from "@/lib/audit/summary-markdown"; +import { buildShareCard } from "@/lib/audit/share-card"; import { env } from "@/lib/env"; export const dynamic = "force-dynamic"; @@ -33,6 +35,7 @@ type SeoAudit = { created_at: string; owner_id: string | null; engine: string | null; + summary: Record | null; }; function hostOf(url: string): string { @@ -47,7 +50,7 @@ async function loadSeoAudit(token: string): Promise { const svc = serviceClient(); const { data } = await svc .from("audits") - .select("target_url, status, score, completed_at, created_at, owner_id, engine") + .select("target_url, status, score, completed_at, created_at, owner_id, engine, summary") .eq("share_token", token) .maybeSingle(); return (data as SeoAudit | null) ?? null; @@ -77,50 +80,52 @@ export async function generateMetadata({ ? { index: true, follow: true, googleBot: { index: true, follow: true } } : { index: false, follow: false, googleBot: { index: false, follow: false } }; - const scoreLabel = - audit.status === "complete" && audit.score !== null - ? ` — Score ${audit.score}/100` - : audit.status === "failed" - ? " — Failed" - : audit.status === "queued" || audit.status === "running" - ? " — Running" - : ""; + // Derive the headline from the same model the OG card renders, so the text + // preview and the image can never disagree. They would otherwise: for a slop + // scan the headline number lives in `summary.slopScore` (0 = pristine) while + // `audits.score` holds the conventional AEO-style score, so a naive title + // showed "78/100" beside a card reading "34/100". + const card = buildShareCard(audit); + const complete = card.state === "complete" && card.score !== null; + const stateSuffix = card.state === "failed" ? " — Failed" : " — Running"; - const description = - audit.status === "complete" && audit.score !== null - ? `AEO audit for ${host} scored ${audit.score}/100. See exactly what AI crawlers — GPTBot, ClaudeBot, PerplexityBot, Google-Extended — can find on the site, plus a prioritised to-do list of fixes.` - : `AEO audit for ${host} from CrawlProof. See what AI crawlers can find on the site.`; + const title = complete + ? card.kind === "slop" + ? `Slop Score ${card.score}/100 for ${host}` + : `AEO audit for ${host} — Score ${card.score}/100` + : card.kind === "slop" + ? `Slop Score for ${host}${stateSuffix}` + : `AEO audit for ${host}${stateSuffix}`; + + const description = complete + ? card.kind === "slop" + ? `${host} scored ${card.score}/100 on the CrawlProof Slop Score, where 0 is pristine — ${card.headline}. A free sweep for observable defects: placeholder copy, near-duplicate pages, leaked template variables, stale dates, and design drift.` + : `AEO audit for ${host} scored ${card.score}/100. See exactly what AI crawlers — GPTBot, ClaudeBot, PerplexityBot, Google-Extended — can find on the site, plus a prioritised to-do list of fixes.` + : `AEO audit for ${host} from CrawlProof. See what AI crawlers can find on the site.`; return { - title: `AEO audit for ${host}${scoreLabel}`, + title, description, alternates: { canonical: url }, robots, openGraph: { type: "article", url, - title: `AEO audit for ${host}${scoreLabel}`, + title, description, siteName: "CrawlProof", publishedTime: audit.created_at, modifiedTime: audit.completed_at ?? audit.created_at, - // Page-level openGraph replaces the layout block wholesale — no - // merging — so the banner has to be re-declared here or X/social - // previews end up with no thumbnail. - images: [ - { - url: "/banner.png", - width: 1200, - height: 630, - alt: `CrawlProof AEO audit for ${host}`, - }, - ], + // No `images` here on purpose. Declaring one would override the + // generated per-report card in ./opengraph-image.tsx and put every + // report back on the identical static banner. The file convention + // supplies og:image, its dimensions, and the alt text. }, twitter: { card: "summary_large_image", - title: `AEO audit for ${host}${scoreLabel}`, + title, description, - images: ["/banner.png"], + // Same reason — the card file feeds twitter:image too. }, other: { "article:section": "AEO Audit", @@ -218,6 +223,11 @@ export default async function PublicReportPage({ const reportAuditIds = reportAudits.map((row) => row.id); const seo = await loadSeoAudit(token); + // Same model the OG card uses, so the watch form names the same score the + // visitor is looking at (slop and AEO differ). + const watchCard = buildShareCard( + seo ?? { target_url: audit.target_url, status: audit.status, score: audit.score, engine: null }, + ); const canonicalUrl = `${env.siteUrl.replace(/\/$/, "")}/r/${token}`; const findingsByAuditId = new Map(); @@ -313,11 +323,20 @@ export default async function PublicReportPage({ )} {audit.status !== "failed" && ( -
+
+ {/* The recurring capture sits beside the one-shot PDF ask: the PDF + ends the conversation, the watch continues it. */} + {seo && ( + + )}
)} diff --git a/app/sitemap.ts b/app/sitemap.ts index 24320783..441b10bb 100644 --- a/app/sitemap.ts +++ b/app/sitemap.ts @@ -14,6 +14,7 @@ export default async function sitemap(): Promise { { url: `${base}/`, changeFrequency: "weekly", priority: 1.0 }, { url: `${base}/pricing`, changeFrequency: "monthly", priority: 0.9 }, { url: `${base}/hire`, changeFrequency: "monthly", priority: 0.9 }, + { url: `${base}/slop`, changeFrequency: "monthly", priority: 0.9 }, { url: `${base}/get-guide`, changeFrequency: "monthly", priority: 0.9 }, { url: `${base}/about`, changeFrequency: "monthly", priority: 0.7 }, { url: `${base}/press`, changeFrequency: "monthly", priority: 0.6 }, diff --git a/app/watch/confirm/[token]/page.tsx b/app/watch/confirm/[token]/page.tsx new file mode 100644 index 00000000..ea5f803e --- /dev/null +++ b/app/watch/confirm/[token]/page.tsx @@ -0,0 +1,40 @@ +import Link from "next/link"; +import { confirmWatchByToken } from "@/app/actions/watchScan"; + +export const metadata = { + title: "Confirm watch", + robots: { index: false, follow: false }, +}; +export const dynamic = "force-dynamic"; + +export default async function ConfirmWatchPage({ + params, +}: { + params: Promise<{ token: string }>; +}) { + const { token } = await params; + const result = await confirmWatchByToken(token); + + return ( +
+

+ {result.ok ? "You're watching it" : "Confirmation link not recognized"} +

+ {result.ok ? ( +

+ We'll re-scan {result.host} {result.cadence} and + email you when its score actually moves. Every message has a one-click + stop link. +

+ ) : ( +

+ We couldn't find a watch for that link. It may have already been + replaced by a newer confirmation email, or the link may be malformed. +

+ )} + + ← Back to CrawlProof + +
+ ); +} diff --git a/app/watch/stop/[token]/page.tsx b/app/watch/stop/[token]/page.tsx new file mode 100644 index 00000000..f71e9af9 --- /dev/null +++ b/app/watch/stop/[token]/page.tsx @@ -0,0 +1,43 @@ +import Link from "next/link"; +import { stopWatchByToken } from "@/app/actions/watchScan"; + +export const metadata = { + title: "Stop watching", + robots: { index: false, follow: false }, +}; +export const dynamic = "force-dynamic"; + +// The human-facing landing page. Mail clients issuing an RFC 8058 one-click +// unsubscribe send a POST, which a page route cannot answer — that lands on +// /api/watch/stop/[token], which redirects GETs here. +export default async function StopWatchPage({ + params, +}: { + params: Promise<{ token: string }>; +}) { + const { token } = await params; + const result = await stopWatchByToken(token); + + return ( +
+

+ {result.ok ? "Stopped" : "Link not recognized"} +

+ {result.ok ? ( +

+ We've stopped watching {result.host}. You + won't get any more score-change emails for it. Other CrawlProof + email (reports you request, receipts) is unaffected. +

+ ) : ( +

+ We couldn't find a watch for that link. It may already have been + stopped. +

+ )} + + ← Back to CrawlProof + +
+ ); +} diff --git a/components/report/watch-form.tsx b/components/report/watch-form.tsx new file mode 100644 index 00000000..e0624a3d --- /dev/null +++ b/components/report/watch-form.tsx @@ -0,0 +1,121 @@ +"use client"; + +import { useState, useTransition } from "react"; +import { createWatch } from "@/app/actions/watchScan"; + +// The recurring lead capture (M2 of docs/lead-engine-prd.md). Sits alongside +// the PDF form at the bottom of /r/. +// +// The report itself stays fully public — gating it would kill the sharing loop +// the OG card exists to create. What costs an email is the ONGOING watch, and +// the ask is self-qualifying: whoever wants a site re-scanned every week is +// the person responsible for that site. +export function WatchForm({ + token, + host, + label, +}: { + token: string; + host: string; + label: string; +}) { + const [pending, startTransition] = useTransition(); + const [email, setEmail] = useState(""); + const [cadence, setCadence] = useState<"weekly" | "monthly">("weekly"); + const [error, setError] = useState(null); + const [done, setDone] = useState(false); + + function submit(e: React.FormEvent) { + e.preventDefault(); + setError(null); + if (!email.trim() || !email.includes("@")) { + setError("Enter a valid email address."); + return; + } + startTransition(async () => { + const res = await createWatch({ token, email: email.trim(), cadence }); + if (!res.ok) { + setError(res.error ?? "Could not set up that watch."); + return; + } + setDone(true); + }); + } + + if (done) { + return ( +
+

Check your inbox

+

+ We sent a confirmation link to {email}. We won't + scan {host} again — or send anything else — until you click it. +

+
+ ); + } + + return ( +
+
+

Watch this URL

+

+ We'll re-scan {host} and email you when its {label} actually + moves — not on a schedule, only when something changes. +

+
+ + + +
+ + Re-scan + +
+ {(["weekly", "monthly"] as const).map((c) => ( + + ))} +
+
+ + +

+ Free. We email you to confirm first, and every message has a one-click + stop link. +

+ {error &&

{error}

} +
+ ); +} diff --git a/components/site-footer.tsx b/components/site-footer.tsx index 7879d805..3b2a2dad 100644 --- a/components/site-footer.tsx +++ b/components/site-footer.tsx @@ -14,6 +14,7 @@ export function SiteFooter() {
Product
  • Pricing
  • +
  • Slop Score
  • Get guide
  • Recent scans
  • Blog
  • diff --git a/docs/lead-engine-prd.md b/docs/lead-engine-prd.md new file mode 100644 index 00000000..23204278 --- /dev/null +++ b/docs/lead-engine-prd.md @@ -0,0 +1,167 @@ +# Crawlproof Lead Engine — PRD + +> Goal: turn the **free scan we already run** into a lead-generation loop. Today an anonymous visitor types a URL into the hero form, gets a full report, and leaves — we keep a row in `audits` and, if they wanted a PDF, an email. Nothing about that flow is designed to *propagate* or to *come back*. +> +> This PRD adds four milestones, ordered by leverage-per-day-of-work, that reuse infrastructure that already exists: the free no-LLM **Slop Score** engine, share tokens (`lib/shareToken.ts` → `/r/[token]`), the two public embeddable scripts (`/stats.js`, `/ad.js`), the referral store (`lib/referrals.ts`), the Prospects org + `/recent` outreach surface, and the alerts/cron spine. +> +> **Non-goal, deliberately:** identifying *people*. The Audience Hub (visitor identify / contact graph / reverse lookup) was removed 2026-06-18 as an info-handling risk, and nothing here reopens it. Every lead in this PRD is someone who **typed their own URL into a box** or **asked to be emailed**. That is a volunteered intent signal, not surveillance. + +--- + +## 0. The problem, stated precisely + +The free scan is a genuinely good lead magnet that currently has **no distribution and no return path**: + +1. **No propagation.** `/r/` does set `twitter:card` and an `og:image` — but it points every report at the same static `/banner.png`. A report for acme.com and a report for example.org therefore produce **byte-identical** previews in Slack, X, and LinkedIn. The single most shareable artifact we produce carries no information about what was scanned. *(Corrects an earlier draft of this line, which claimed there was no OG image at all.)* +2. **No return path.** The report is a one-shot. `EmailReportForm` exists only to mail a PDF, so the only captured leads are people who wanted a PDF. There is no recurring reason to email a scanned site again. +3. **No third-party surface.** `/stats.js` and `/ad.js` prove we can ship a public embeddable script, but neither is a *funnel* — one is analytics for existing customers, one serves ads. Nobody can put "free AI-readiness check" on their own site and send us the traffic. +4. **No public scan API.** `startAuditFromForm` is a server action only. A prod scan cannot be triggered by `curl`, which blocks any embed, any partner integration, and any bulk flow. + +Fixing 1–4 is the whole PRD. + +--- + +## 1. Status / phasing + +**Phase 0 — PRD: this document.** + +**M1 — Shareable scorecard (OG image): SHIPPED.** Per-report generated card at `app/r/[token]/opengraph-image.tsx`, driven by the pure model in `lib/audit/share-card.ts` (18 unit tests in `tests/share-card.test.ts`), plus the `/slop` landing page and its own static card. + +**M2 — Watch this URL (recurring lead capture): SHIPPED (migration not yet applied).** `scan_watches` + double opt-in (`app/actions/watchScan.ts`), a two-phase cron at `app/api/cron/scan-watches` (deliver finished re-scans, then enqueue due ones), score-change email with RFC 8058 one-click stop, and the capture form on `/r/[token]`. Decision logic is pure and tested in `lib/watches.ts` / `tests/watches.test.ts` (21 tests). + +> **Deploy order matters:** apply `20260726120000_scan_watches.sql` *after* the code is live. The migration schedules a pg_cron job that POSTs to `/api/cron/scan-watches` every 15 minutes; applied first, that route 404s until the deploy lands. + +**M3 — `/scan.js` embeddable widget + public scan API + referral attribution: PLANNED.** ~1 week. + +**M4 — Prospect Scan (lead-gen *for customers*, credit-burning): PLANNED, larger.** Revenue-side; ships after M1–M3 prove the funnel. + +--- + +## 2. M1 — Shareable scorecard (the OG image) + +**The mechanic:** every shared report becomes an ad that carries the scanned site's own name. + +Add `app/r/[token]/opengraph-image.tsx` using Next's `ImageResponse` (already available in Next 16, no new dependency): + +- Big score dial — reuse the visual language of `components/report/slop-meter.tsx`. +- The scanned hostname, large. **This is the point**: "acme.com scored 34/100" is a far stronger click than "CrawlProof report". +- One-line headline derived from the findings, e.g. `12 pages with careless defects` or `Blocks GPTBot, ClaudeBot, PerplexityBot`. +- CrawlProof mark, small, bottom-right. + +`twitter:card = summary_large_image` was already set on `/r/[token]`, so it only needed the image swapped. Two wiring gotchas found while building it, both worth knowing before adding cards to other routes: + +- A page that declares `openGraph.images` **overrides** the generated `opengraph-image.tsx`. The static `/banner.png` had to be *removed* from `generateMetadata` for the card to take effect. +- A page that declares an `openGraph` block but **no** `twitter` block inherits the root layout's `twitter.images` — which also outranks the file convention. `/slop` initially shipped with the new card everywhere and the old banner on X alone. Every page with a generated card needs its own `twitter` block, with neither block declaring `images`. + +Also `/slop` gets its own static card (§2.1). + +**Why it's first:** it costs a day, it has no schema change, no new endpoint, no privacy surface, and it retroactively upgrades *every report we have ever generated* — all existing share links start rendering as cards the moment it deploys. + +**Cache:** the image must be generated from the stored audit row, not by re-running the scan. Cache by `token` + `completed_at`. + +### 2.1 `/slop` landing page + +The Slop Score is free, deterministic, and runs **no LLM** — so unlike Autoblog it cannot be stalled by a shared provider quota, and unlimited scans cost us nothing but bandwidth. It is therefore the correct engine to put at the top of the funnel, and it currently has no page of its own. + +Ship `/slop` as a dedicated landing page: the promise ("find the careless mistakes on your site — free, 50 pages, no signup"), the scan box, and 3–4 real anonymized example findings. Keep the existing hero form on `/` unchanged. + +**Positioning guardrail, non-negotiable and already enforced by `tests/slop.test.ts`:** the page reports *observable defects*, never "this was written by AI". An AI-probability claim is unfalsifiable, misfires on non-native English writers, and would accuse paying customers. Marketing copy must not drift into it. + +--- + +## 3. M2 — "Watch this URL" (the recurring lead capture) + +**The mechanic:** stop gating the *report*; gate the *ongoing relationship*. + +The full report stays 100% visible to anonymous visitors — that is what makes it shareable, and hiding it would kill M1. Instead, three things ask for an email, on the report page itself: + +| Ask | Value exchange | Lead quality | +|---|---|---| +| Email me the PDF | *(exists today)* | Medium | +| **Watch this URL** — re-scan weekly, email me when the score changes | **The lead engine** | **High** | +| Export the per-page fix list (CSV/markdown) | Practical | High | + +**"Watch this URL" is the important one**, for three reasons: + +1. It is **self-qualifying**. Somebody who wants a weekly re-scan of a site is, by definition, someone responsible for that site. That is our buyer. No enrichment, no scoring model needed — the request *is* the qualification. +2. It creates a **recurring, wanted, non-spam reason to email them**. "Your score went from 34 to 51" is a message people open. Every send is a re-entry point to the product, and the unsubscribe is genuine. +3. It **reuses the cron + alerts spine wholesale** — `lib/alerts/`, the uptime re-alert pattern (`20260714124903_uptime_down_realert.sql`), and the existing global unsubscribe route `/unsubscribe/[token]`. + +**Implementation notes:** +- New table `scan_watches` (email, target_url, engine, cadence, verified_at, unsubscribed_at, last_score, last_scanned_at). Double opt-in: the first email confirms; nothing recurring sends until `verified_at` is set. +- Tag the resulting audits to the existing **Prospects** org via `audits.organization_id`, so `/recent`'s outreach form can work them without inventing a fake project per lead — which is exactly what `20260606133000_prospects_outreach_configs.sql` was built for. +- Every send carries `List-Unsubscribe` headers (`lib/outreach.ts` already does this) and honours globally-unsubscribed `marketing_contacts`. +- Cap: one watch per email per URL, and a hard cap on watches per email, or this becomes a free monitoring tier by accident. + +**Compliance line:** a watch is opt-in, double-confirmed, and about the subscriber's own site. Cold-emailing every domain that ever appeared in `/recent` is a *different activity* with real CAN-SPAM/GDPR exposure — keep the two separated in the code and in the org's habits. + +--- + +## 4. M3 — `/scan.js` embeddable widget + public scan API + +**The mechanic:** let other people host our funnel. + +This is the Ahrefs-free-tools / "backlink checker embedded on 400 blogs" play, and we are unusually well set up for it because the embeddable-script pattern is already proven twice in this repo (`/stats.js`, 210 lines; `/ad.js`, 105 lines) and the CORS ingest pattern already exists in `/api/track`. + +### 4.1 `POST /api/scan` — the missing primitive + +There is **no public endpoint to start a scan today** (server action only). Add one: + +``` +POST /api/scan { url, engine?: "rule" | "slop", ref?: } + → 202 { token, pollUrl } +``` + +- Same guardrails as the hero form, not looser: `lib/rateLimit.ts` anonymous per-IP limits, the existing URL-safety checks (no private ranges, no SSRF), and `ANON_ENGINES` restricted to `["rule", "slop"]` — **no LLM engines**, so an abusive embed can never burn provider budget. +- Per-origin rate limit in addition to per-IP, keyed on the embedding site, so one hostile embed cannot exhaust the global pool. +- This endpoint also unblocks partner integrations and the MCP surface generally. + +### 4.2 The widget + +`GET /scan.js` serves a tiny dependency-free script, same shape as `stats.js`: + +```html + +``` + +Renders an inline "How AI-ready is your site?" input. On submit it calls `/api/scan`, shows the score inline, and links to the full report at `/r/?ref=`. Styling inherits from the host page with a minimal reset; failures must never break the host page (same rule as `stats.js`). + +### 4.3 Attribution + +`lib/referrals.ts` (`@profullstack/stack/referrals`) + `/api/referrals` + `/r/[token]` already exist. Wire the widget's `data-ref` through the scan into the referral store, so an agency or newsletter that embeds the widget earns credits on signups it drives. That gives embedders an actual reason to install it and keep it installed — the same two-sided logic that makes the ad network work. + +--- + +## 5. M4 — Prospect Scan (leads *for the customer*, and revenue for us) + +Everything above generates leads **for CrawlProof**. M4 is the other reading of the request: sell lead generation *as a feature*. + +**The mechanic:** the customer brings the list; we supply the pitch. + +1. Customer uploads or pastes a list of prospect domains (their own list — an event attendee list, a directory export, an industry roster, their existing CRM). +2. CrawlProof bulk-runs the free/cheap engines across the list, worst-first ranked. +3. For each prospect, generate a one-page branded "here's what's broken on your site" PDF — reusing the existing PDF worker and the remediation quote already on the report cover (`4df320c`). +4. Customer exports the ranked list + PDFs and runs **their own** outreach. + +**Why this shape and not the obvious one:** the tempting version is visitor de-anonymization — reverse-IP the tracker's traffic into company names. That is the Audience Hub again in a new hat, and it was removed as an info-handling risk. This version touches **no personal data at all**: the input list is the customer's, the scanned sites are public, and CrawlProof is never in the sending path. It's the difference between selling a surveillance product and selling a *sales-collateral generator*. + +**Revenue:** bulk scanning is direct credit burn — 200 prospects is 200 scans. It monetizes on the exact axis the platform is already built to bill. + +**Check overlap** with `docs/agency-prd.md` (multi-site management) — different thing, but agencies are the buyer for both, so the surfaces should be adjacent in the UI. + +--- + +## 6. What we are explicitly not building + +- **Visitor de-anonymization / reverse-IP company identification.** Removed once already (Audience Hub, 2026-06-18) as an info risk. Not reopening it. +- **Contact enrichment / email appending.** Same reason. +- **An "AI-written probability" score** as a lead hook. Unfalsifiable, misfires on non-native speakers, insults customers. `tests/slop.test.ts` guards this. +- **Mass cold-emailing every domain in `/recent`.** The infrastructure could do it; the legal exposure and the brand damage are not worth it, and it would poison the deliverability that M2 depends on. + +--- + +## 7. Sequencing rationale + +M1 before everything because it is a day of work that upgrades every report ever generated, and because M3's widget is worthless if the reports it produces don't render when shared. M2 before M3 because there is no point pouring embedded traffic into a funnel with no return path. M4 last because it is the only one that needs the funnel to already work. + +**Success signal to watch before committing to M3:** share-link click-through after M1, and watch-signup rate on anonymous reports after M2. If M2's opt-in rate is under a few percent, the value exchange is wrong and M3 will amplify a leak. diff --git a/docs/reshare-network-prd.md b/docs/reshare-network-prd.md new file mode 100644 index 00000000..b6ce7f91 --- /dev/null +++ b/docs/reshare-network-prd.md @@ -0,0 +1,221 @@ +# Crawlproof Reshare Network — PRD + +> Goal: a consent-based, quality-gated **reshare network** inside crawlproof.com, spanning **all connected social platforms**. A customer opts one or more of their already-connected accounts into the network; when another member publishes an on-topic post, we **reshare it from the customer's account** — and, reciprocally, network members reshare the customer's posts. Same credit ledger as audits / autoblog / social posting. +> +> This is the `lib/lx/` **link-exchange reciprocity model applied to social reshares instead of backlinks**, riding on the `lib/sp/` connected-account infrastructure that already exists (14 platforms, OAuth + browser modes). The wedge: we already own both ends — the content generator *and* the distribution accounts — and we already have a niche/quality **gate** (`@profullstack/autoblog/quality` `gatePost`) to keep the network from becoming a spam swamp. +> +> **All social platforms are in scope.** The catch: "reshare" is not one mechanism. Some platforms expose a **free, first-class API repost action**; some gate it behind a **paid or spam-sensitive API**; some have **no repost API at all** (browser automation only). We roll out **by repost mechanism**, cleanest first — nothing is dropped, but the risky surfaces ship behind the same gated, opt-in disclosure model as Social Posting Phase 3. See §7 for the full matrix. + +--- + +## 1. Repost mechanism per platform — the organizing fact + +Every platform in `sp_account` (`bluesky, mastodon, reddit, linkedin, threads, pinterest, tumblr, x, facebook_page, instagram_business, youtube, tiktok, instagram, snapchat`) is covered. What differs is *how* a reshare happens and how much risk it carries: + +| Platform | Reshare mechanism | API? | Cost | Risk tier | +|---|---|---|---|---| +| **Bluesky** | `app.bsky.feed.repost` record | ✅ official | free | **1** | +| **Mastodon** | `POST /statuses/:id/reblog` | ✅ official | free | **1** | +| **Tumblr** | `POST /blog/:id/post/reblog` (native reblog) | ✅ official | free | **1** | +| **Telegram** | `forwardMessage` / `copyMessage` (Bot API) | ✅ official | free | **1** | +| **X** | `POST /2/users/:id/retweets` | ✅ official | **paid tier** | **2** | +| **Threads** | repost endpoint (Graph, newer/stricter) | ✅ official | free | **2** | +| **Reddit** | crosspost (`/api/submit` kind=crosspost) | ✅ official | free | **2** (per-subreddit rules) | +| **Pinterest** | save/repin to a board | ✅ official | free | **2** | +| **Discord** | crosspost announcement message | ✅ official | free | **2** (announcement channels only) | +| **LinkedIn** | in-UI reshare — no clean API reshare | ❌ browser | — | **3** | +| **Facebook Page** | share (Graph share deprecated) | ❌ browser | — | **3** | +| **Instagram / IG Business** | no native repost — browser | ❌ browser | — | **3** | +| **TikTok** | in-app repost only, no API — browser | ❌ browser | — | **3** | +| **YouTube** | no repost concept (community reshare not API-exposed) | ❌ n/a | — | **excluded** | + +Tiering tracks risk almost exactly: the API-native reshare actions are normal platform behavior; the browser-automated ones put the **customer's own account** at suspension risk (§7). We ship Tier 1 → 2 → 3. + +--- + +## 2. Status / phasing + +**Phase 0 — PRD: this document.** + +**Phase 1 — Tier 1 (free, first-class API reshare): PLANNED, ships first.** +- Bluesky, Mastodon, Tumblr, Telegram. Add a `repost()` action to each adapter (today `lib/sp/platforms/*` publish *original* posts only). Lowest risk, no paid tiers, no browser. This is where we prove the mechanic. + +**Phase 2 — Tier 2 (official API, paid or spam-sensitive): GATED.** +- X (paid API tier + disclosure), Threads, Reddit (crosspost → requires a target subreddit + honors subreddit rules), Pinterest (repin to board), Discord (announcement-channel crosspost). Each ships behind a per-platform risk disclosure; opens only after Phase 1 shows a near-zero suspension rate (§10). + +**Phase 3 — Tier 3 (browser-mode reshare): GATED, OPT-IN, mirrors Social Posting Phase 3.** +- LinkedIn, Facebook Page, Instagram, TikTok. Reuses the existing browser runner (`lib/sp/platforms/browser.ts`, `browserSemaphore.ts`) with `cookie`/`puppeteer` `auth_mode`, isolated runner cluster, residential proxies, encrypted-credential vault, and a hard disclosure modal. Highest account-ban risk — opt-in per account, capped hardest, and the first surface we pause if suspensions appear. + +**YouTube — excluded** from reshare (no API repost primitive; a "community post" is original content, not a reshare). Stays a Social-Posting-only platform. + +--- + +## 3. Why this fits crawlproof's existing rails + +### 3.1 `lib/sp/` — connected accounts + publishing (SHIPPED) +- `sp_account` — user-scoped pool, all 14 platforms already enumerated, with `auth_mode in ('oauth','cookie','puppeteer')`, `enc_access_token`, `status`, `last_post_at`, `consecutive_failures`. +- `sp_site_account` — M:N site↔account binding + `auto` flag. +- `sp_post` / `sp_publish_attempt` — queued/sent log + attempt trail. +- Adapters `lib/sp/platforms/{bluesky,mastodon,reddit,linkedin,threads,facebook,telegram,discord,x}.ts` + `browser.ts`; OAuth refresh (`sessionRefresh.ts`); vault (`vault.ts`, `SOCIAL_VAULT_KEY`); browser concurrency (`browserSemaphore.ts`). +- **What's missing:** every adapter does *original* posts only (`createTweet`, Bluesky `createRecord`, Mastodon `POST /statuses`, LinkedIn `ShareContent`, browser flows). No reshare action exists yet — that's the core new adapter work (§5). + +### 3.2 `lib/lx/` — the reciprocity network we're cloning +- outrank.so-style multi-tenant exchange: any opted-in customer both gives and receives. +- Per-participant **niche allowlist + heuristic + LLM quality score** via `gatePost` (loose case-insensitive niche overlap, fail-open on LLM error). +- Per-site opt-in toggle; credit ledger; a matcher pairing givers to receivers. + +The reshare network is `lx` with the edge type changed from *backlink* to *reshare*, and the target from *a blog article* to *a social post*. + +--- + +## 4. Positioning & the "army" framing + +- Sold as an **add-on**, like Autoblog and Social Posting. +- Consumer name: "Reshare Network" / amplification network. Internally it is **consent-based reciprocal distribution, not an anonymous bot ring**. Every reshare comes from a **real member's real account that opted in**, resharing **on-topic** content. That distinction is the entire defensibility story (§7). +- Tightly tied to Autoblog + Social Posting: when a customer's autoblog article auto-posts to their connected accounts (existing `auto` pipeline), that post becomes eligible for network amplification. We own the content *and* the amplification. + +--- + +## 5. Data model (Supabase) + +New tables, all `rn_`-prefixed. Mirrors `lx_*`, rides on existing `sp_account`. `platform` columns use the **same full enum as `sp_account`** so every platform is representable from day one, even before its adapter ships. + +```sql +create table rn_membership ( + id uuid primary key default gen_random_uuid(), + user_id uuid not null references public.profiles(id) on delete cascade, + sp_account_id uuid not null references public.sp_account(id) on delete cascade, + + niches text[] not null default '{}', -- ['seo','devtools'] + languages text[] not null default '{en}', + + give_enabled boolean not null default true, -- will reshare others + receive_enabled boolean not null default true, -- wants to be reshared + max_reshares_per_day int not null default 5 + check (max_reshares_per_day between 0 and 20), + min_gate_score numeric not null default 0.6, + + status text not null default 'active' + check (status in ('active','paused','suspended','user_disabled')), + created_at timestamptz not null default now(), + updated_at timestamptz not null default now(), + unique (sp_account_id) +); + +create table rn_source_post ( + id uuid primary key default gen_random_uuid(), + membership_id uuid not null references rn_membership(id) on delete cascade, + -- full sp_account platform enum — every platform representable + platform text not null check (platform in ( + 'bluesky','mastodon','reddit','linkedin','threads','pinterest','tumblr', + 'x','facebook_page','instagram_business','tiktok','instagram')), + external_post_id text not null, -- at:// uri, status id, tweet id, pin id… + post_url text, + text_snippet text, + niches text[] not null default '{}', + gate_score numeric, + gate_ok boolean, + target_reshares int not null default 0, + reshares_done int not null default 0, + status text not null default 'pending' + check (status in ('pending','gating','active','done','rejected')), + -- Reddit crosspost needs a destination; Discord needs a followed + -- announcement channel. Per-actor routing lives in rn_reshare.target_ref. + created_at timestamptz not null default now(), + unique (platform, external_post_id) +); + +create table rn_reshare ( + id uuid primary key default gen_random_uuid(), + source_post_id uuid not null references rn_source_post(id) on delete cascade, + actor_membership_id uuid not null references rn_membership(id) on delete cascade, + platform text not null, + mechanism text not null check (mechanism in ( + 'repost','reblog','forward','retweet','crosspost','repin','browser')), + target_ref text, -- subreddit / board / channel, when required + external_reshare_id text, -- the repost/reblog record id + status text not null default 'queued' + check (status in ('queued','sent','failed','skipped','undone')), + gate_score numeric, + scheduled_for timestamptz, -- jittered (§6) + sent_at timestamptz, + error text, + created_at timestamptz not null default now(), + unique (source_post_id, actor_membership_id) +); +``` + +RLS: `rn_membership` owner-scoped via `sp_account.user_id`. `rn_source_post` / `rn_reshare` are service-role-written by the worker; members read only rows tied to their memberships. + +--- + +## 6. The matcher + execution + +**Matcher** (worker, on new `rn_source_post`): +1. **Gate the source** with `gatePost` (`@profullstack/autoblog/quality`) — reject spam/off-niche/low-quality; cache `gate_score`/`gate_ok`. Default `min_gate_score = 0.6` (stricter than backlinks — a bad reshare pollutes the actor's *public timeline*, not just a footer). +2. **Candidate actors:** `rn_membership` where `give_enabled`, **same platform**, `status='active'`, **niche + language overlap**, not the author, under `max_reshares_per_day`. +3. **Reciprocity weighting:** prefer actors the author reshares back — give/receive balance per pair, so the network stays mutual (mirrors `lx`'s ledger). +4. **Fan-out cap:** `target_reshares = min(cap, eligible actors, per-source ceiling)`. Never amplify one post to the whole network at once — that burst is *the* detectable manipulation signal (§7). +5. Insert `rn_reshare` rows with the platform's `mechanism`, `target_ref` where required (Reddit subreddit / Discord channel / Pinterest board), and **jittered `scheduled_for`**. + +**Execution** (worker drains due `rn_reshare`): dispatch to the new adapter reshare action per §1's mechanism column. Spread over time (jitter + `max_reshares_per_day` + actor active-window skew from `sp_account.last_post_at`), never bursts. On success: write `external_reshare_id`, `sent_at`, bump `reshares_done`, debit 1 credit. On `suspended_by_platform`/auth failure: set `sp_account.status`, pause the `rn_membership`, stop scheduling it, email the owner. **Account safety > completing a fan-out.** + +New adapter work (the only genuinely new code), one reshare action each: +- Tier 1: `bluesky.repost` (createRecord `app.bsky.feed.repost`), `mastodon.reblog` (`/statuses/:id/reblog`), `tumblr.reblog` (`/post/reblog`), `telegram.forward` (`forwardMessage`/`copyMessage`). +- Tier 2: `x.retweet` (`/2/users/:id/retweets`, paid), `threads.repost`, `reddit.crosspost` (`kind=crosspost` → `target_ref` subreddit), `pinterest.repin` (save → board), `discord.crosspost` (announcement message). +- Tier 3: browser reshare flows in `browser.ts` for LinkedIn / Facebook / Instagram / TikTok (`cookie`/`puppeteer` mode, gated). + +--- + +## 7. Platform risk — the load-bearing section + +A reciprocal auto-reshare network is mechanically an **engagement pod / retweet ring / "reciprocal amplification"** pattern. **Every major platform's spam & manipulation policy prohibits it, and integrity systems detect it.** The party hurt is **the customer** (their real account rate-limited, shadow-banned, or suspended) — a growth product that suspends its own users churns catastrophically. Crawlproof already pulled the **Audience** feature for "info risk"; the same judgment gates this. + +**The tiering in §1 is the risk-management spine, not a convenience ordering:** +- **Tier 1 (Bluesky, Mastodon, Tumblr, Telegram)** — reshare/reblog/forward is *normal, free, first-class API behavior* on open/federated networks with little-to-no centralized ring-detection. Lowest risk; ship and learn here. +- **Tier 2 (X, Threads, Reddit, Pinterest, Discord)** — official API exists, but each adds risk: X runs the most aggressive reciprocal-RT detection *and* charges for API; Reddit crossposts hit per-subreddit spam filters + rules; Threads review is strict. Gated behind per-platform disclosure; opens only after Tier 1 proves safe. +- **Tier 3 (LinkedIn, Facebook, Instagram, TikTok)** — **no API reshare → browser automation**, which most ToS forbid and which most endangers the customer's account. Opt-in with a hard disclosure modal, isolated runner + residential proxies, capped hardest, first to be paused on any suspension signal. + +**Mitigations baked into the design (mirroring the `lx` gate philosophy), all platforms:** +1. **Real, opted-in accounts only** — never synthetic. The single biggest thing separating this from a botnet. +2. **Relevance gating** — `gatePost` niche/quality filter; only on-topic reshares. Off-topic amplification is both spammy and the clearest inauthenticity signal. +3. **Explicit per-account opt-in + disclosure modal**, with the platform-suspension risk spelled out; instant pause. +4. **Human-like cadence** — per-account daily caps, jitter, active-window skew, per-source fan-out ceiling. No bursts. +5. **Fail-safe on pushback** — first `suspended_by_platform`/throttle → pause that membership and notify, rather than pushing through. +6. **Phased by risk** — Tier 1 first; Tier 2/3 open only on a proven-safe suspension rate; browser-mode is opt-in and capped hardest. + +**Explicit non-goals:** buying/selling engagement, fake accounts, follow-back schemes, comment pods, off-topic mass amplification, evading platform rate limits, or acting on any account without the owner's consent. + +--- + +## 8. Credits + +- **1 credit per reshare sent** (per actor, per source post), via the existing `consume_credit` RPC against `credit_ledger` — same ledger as audits/autoblog/social-posting. Being reshared (`receive_enabled`) is **free**; the actors pay. Keeps spend on the person getting value and makes joining attractive. +- Free-tier throttle: low `max_reshares_per_day` for free/low-balance accounts; full cap unlocks with balance. (Optional later: earn-back — reshare N others → get M free.) + +--- + +## 9. UI surface + +- **`/reshare`** (or under `/social`) — enroll accounts already in `sp_account`, grouped by platform with each platform's mechanism + risk tier shown. Per-account: niches, languages, give/receive, daily cap, min gate score, pause. +- **Network activity** — "your posts amplified: N reshares from M accounts" + "you amplified N." Reciprocity balance meter. +- **Disclosure** — a real modal at opt-in per account (harder wording for Tier 2/3), explaining exactly what auto-resharing does and the suspension risk; consent recorded with timestamp. Tier 3 additionally reuses the Social Posting Phase 3 browser-automation disclosure. + +--- + +## 10. Build order + +1. Tier 1 reshare actions in `bluesky/mastodon/tumblr/telegram` adapters (+ unit tests per `lib/sp/platforms/*` pattern). +2. `rn_*` migration (§5) with RLS. +3. Matcher + gate reuse + jittered execution + `consume_credit` debit (§6, §8). +4. `/reshare` enroll + activity UI + disclosure (§9). +5. Ops: suspension handling, **network-health dashboard — per-platform suspension rate is the KPI that gates every later tier.** +6. Tier 2 adapters (X paid / Threads / Reddit crosspost / Pinterest / Discord) behind per-platform disclosure — **open only after §5 metrics prove Tier 1 safe.** +7. Tier 3 browser-mode reshare (LinkedIn / Facebook / Instagram / TikTok) on the existing browser runner — **opt-in, capped hardest, gated on the same safety metric.** + +--- + +## 11. Success / kill criteria + +- **Tier gate:** a tier opens only when the **per-platform account-suspension rate attributable to reshares stays near zero** across a meaningful active pool *and* members report reach lift. Climbing suspensions → do **not** advance a tier; tighten caps or kill that platform (the Audience precedent). +- **Kill criteria:** any evidence a platform's network is functioning as a detectable ring (coordinated-behavior strikes, mass throttling) → pause that platform network-wide, not per-account. diff --git a/lib/audit/share-card.ts b/lib/audit/share-card.ts new file mode 100644 index 00000000..405a6472 --- /dev/null +++ b/lib/audit/share-card.ts @@ -0,0 +1,159 @@ +// The data model behind the generated OpenGraph card for a shared report. +// +// Kept as a pure function, separate from the image rendering, for two reasons: +// it is the part with real branching (two engines whose scores run in OPPOSITE +// directions, plus three run states), and it can be unit-tested without +// rasterising a PNG. +// +// The card's whole job is to carry the SCANNED SITE'S NAME, so a link pasted +// into Slack or X reads as "acme.com scored 34/100" rather than as a generic +// CrawlProof banner. Everything here serves that headline. + +export type CardTone = "pass" | "warn" | "fail" | "neutral"; + +export type ShareCard = { + /** Hostname of the scanned site — the headline of the card. */ + host: string; + /** Which score is being shown; the two run in opposite directions. */ + kind: "aeo" | "slop"; + /** Human label for the number, e.g. "AEO Score". */ + label: string; + state: "complete" | "pending" | "failed"; + /** null unless state === "complete". */ + score: number | null; + tone: CardTone; + /** One line of supporting detail under the score. */ + headline: string; + /** Direction hint, so a low number is never misread as a bad one. */ + scaleHint: string; + /** 0–100 width of the meter, matching the number above it. */ + fill: number; + /** Footer strapline — describes the scan that actually ran. */ + footer: string; +}; + +export type ShareCardAudit = { + target_url: string; + status: string; + score: number | null; + engine: string | null; + summary?: Record | null; +}; + +export function hostOf(url: string): string { + try { + return new URL(url).hostname.replace(/^www\./, ""); + } catch { + // Fall back to the raw string rather than showing nothing — a malformed + // stored URL should still produce a card. + return url.replace(/^https?:\/\//, "").replace(/^www\./, "").split("/")[0] || url; + } +} + +function num(value: unknown): number | null { + return typeof value === "number" && Number.isFinite(value) ? value : null; +} + +function plural(n: number, word: string): string { + return `${n} ${word}${n === 1 ? "" : "s"}`; +} + +/** + * Slop runs 0 = pristine → 100 = maximum slop, so a LOW number is good. This + * inverts the usual dial and is the single easiest thing to get wrong on a + * card, where there is no surrounding copy to explain it. + */ +function slopTone(score: number): CardTone { + if (score <= 25) return "pass"; + if (score <= 50) return "warn"; + return "fail"; +} + +/** AEO runs the conventional direction: 100 = best. */ +function aeoTone(score: number): CardTone { + if (score >= 80) return "pass"; + if (score >= 50) return "warn"; + return "fail"; +} + +export function buildShareCard(audit: ShareCardAudit): ShareCard { + const host = hostOf(audit.target_url); + const summary = (audit.summary ?? {}) as Record; + const slopScore = num(summary.slopScore); + // Trust the engine column, but fall back to the summary shape: sibling rows + // in a multi-engine scan_run can carry a null engine. + const isSlop = audit.engine === "slop" || (audit.engine == null && slopScore !== null); + + const kind = isSlop ? "slop" : "aeo"; + const label = isSlop ? "Slop Score" : "AEO Score"; + const scaleHint = isSlop ? "0 = pristine · 100 = maximum slop" : "out of 100 · higher is better"; + // Name the scan that actually ran — an "SEO · AEO · GEO audit" strapline + // under a Slop Score is describing a different product. + const footer = isSlop + ? "Free carelessness scan — content, code, design" + : "SEO · AEO · GEO audit — free, no signup"; + + if (audit.status === "failed") { + return { + host, + kind, + label, + state: "failed", + score: null, + tone: "neutral", + headline: "Scan failed", + scaleHint, + fill: 0, + footer, + }; + } + + const score = isSlop ? slopScore : num(audit.score); + if (audit.status !== "complete" || score === null) { + return { + host, + kind, + label, + state: "pending", + score: null, + tone: "neutral", + headline: "Scan running…", + scaleHint, + fill: 0, + footer, + }; + } + + const pages = num(summary.pagesCrawled); + const parts: string[] = []; + + if (isSlop) { + const grade = typeof summary.slopGrade === "string" ? summary.slopGrade : null; + const issues = num(summary.slopIssues); + if (grade) parts.push(grade); + if (issues !== null) parts.push(plural(issues, "issue")); + if (pages !== null) parts.push(`${plural(pages, "page")} swept`); + } else { + const fail = num(summary.fail); + const pass = num(summary.pass); + if (fail !== null && fail > 0) parts.push(`${plural(fail, "check")} failed`); + else if (pass !== null) parts.push(`all ${plural(pass, "check")} passed`); + if (pages !== null) parts.push(`${plural(pages, "page")} crawled`); + } + + return { + host, + kind, + label, + state: "complete", + score, + tone: isSlop ? slopTone(score) : aeoTone(score), + // Never leave the line empty — an older row may predate these summary keys. + headline: parts.length > 0 ? parts.join(" · ") : "See the full report", + scaleHint, + // Both dials fill in the direction of their own number, so the bar always + // agrees with the digits printed above it. + fill: Math.max(2, Math.min(100, score)), + footer, + }; +} diff --git a/lib/email.ts b/lib/email.ts index b1e81be3..f6e9fc60 100644 --- a/lib/email.ts +++ b/lib/email.ts @@ -656,3 +656,176 @@ export async function sendOrgInviteEmail(input: { if (!res.sent) return { sent: false, error: res.error }; return { sent: true }; } + +// ============================================================ +// "Watch this URL" — M2 of docs/lead-engine-prd.md. +// ============================================================ + +export function watchConfirmEmailHtml(input: { + host: string; + label: string; + cadence: string; + confirmUrl: string; +}): string { + const innerHtml = ` + +

    + Confirm you want ${escapeHtml(input.host)} watched +

    +

    + Someone asked us to re-scan + ${escapeHtml(input.host)} + ${escapeHtml(input.cadence)} and email this address when its + ${escapeHtml(input.label)} changes. Click below to start — we won't + send anything until you do. +

    + + + + + + Start watching ${escapeHtml(input.host)} → + + + + + +

    + Or copy this link into your browser:
    + ${input.confirmUrl} +

    + + `; + return emailShell({ + title: `Confirm watching ${input.host}`, + innerHtml, + footerNote: + "If you didn't request this, ignore this email and nothing further will be sent. " + + "We only start watching a site after this link is clicked.", + }); +} + +export async function sendWatchConfirmEmail(input: { + to: string; + host: string; + label: string; + cadence: string; + confirmUrl: string; +}): Promise<{ sent: boolean; error?: string }> { + const c = client(); + if (!c) return { sent: false, error: "RESEND_API_KEY not set" }; + const res = await c.send({ + from: env.resendFrom, + to: input.to, + subject: `Confirm: watch ${input.host} for changes`, + html: watchConfirmEmailHtml(input), + }); + if (!res.sent) return { sent: false, error: res.error }; + return { sent: true }; +} + +export function watchChangeEmailHtml(input: { + host: string; + label: string; + score: number; + previousScore: number | null; + /** Already accounts for the inverted slop dial. */ + improved: boolean; + first: boolean; + scaleHint: string; + reportUrl: string; + stopUrl: string; + cadence: string; +}): string { + const accent = input.first + ? { bg: "#6ee7b7", fg: "#042f1a" } + : input.improved + ? { bg: "#6ee7b7", fg: "#042f1a" } + : { bg: "#fca5a5", fg: "#3a0808" }; + + const movement = input.first + ? `This is the baseline we'll compare future scans against.` + : `${input.improved ? "Better" : "Worse"} than last time — it was + ${input.previousScore}.`; + + const innerHtml = ` + +

    + ${input.first ? `Now watching ${escapeHtml(input.host)}` : `${escapeHtml(input.host)} ${input.improved ? "improved" : "got worse"}`} +

    +

    + ${movement} +

    + + + + + + + + + +
    + ${input.score} / 100 + + ${escapeHtml(input.label)}
    + ${escapeHtml(input.scaleHint)} +
    + + + + + + See what changed → + + + + + +

    + Full report:
    + ${input.reportUrl} +

    + + `; + return emailShell({ + title: `${input.host} — ${input.label} ${input.score}/100`, + innerHtml, + footerNote: + `You asked us to re-scan ${escapeHtml(input.host)} ${escapeHtml(input.cadence)}. ` + + `Stop watching this site.`, + }); +} + +export async function sendWatchChangeEmail(input: { + to: string; + subject: string; + host: string; + label: string; + score: number; + previousScore: number | null; + improved: boolean; + first: boolean; + scaleHint: string; + reportUrl: string; + stopUrl: string; + cadence: string; +}): Promise<{ sent: boolean; error?: string }> { + const c = client(); + if (!c) return { sent: false, error: "RESEND_API_KEY not set" }; + const res = await c.send({ + from: env.resendFrom, + to: input.to, + subject: input.subject, + html: watchChangeEmailHtml(input), + // Recurring mail, so mail clients get a native one-click stop button. + headers: { + "List-Unsubscribe": `<${input.stopUrl}>`, + "List-Unsubscribe-Post": "List-Unsubscribe=One-Click", + }, + }); + if (!res.sent) return { sent: false, error: res.error }; + return { sent: true }; +} diff --git a/lib/watches.ts b/lib/watches.ts new file mode 100644 index 00000000..958f52cf --- /dev/null +++ b/lib/watches.ts @@ -0,0 +1,130 @@ +// "Watch this URL" — the recurring lead capture behind M2 of the lead engine. +// +// The report on /r/ stays fully public; what asks for an email is the +// ONGOING relationship. Somebody who wants a weekly re-scan of a site is, by +// definition, the person responsible for that site — the request is the +// qualification, so no scoring model is needed. +// +// The decision logic lives here as pure functions because it is the part with +// real branching (two engines whose scores move in opposite directions) and +// the part that must not send a wrong or noisy email. + +import { buildShareCard, type ShareCardAudit } from "@/lib/audit/share-card"; + +export const WATCH_CADENCES = ["weekly", "monthly"] as const; +export type WatchCadence = (typeof WATCH_CADENCES)[number]; + +/** Engines a watch may re-run. Both are free and self-hosted, so a watch can + * never turn into recurring LLM spend on an address we never charged. */ +export const WATCH_ENGINES = ["rule", "slop"] as const; +export type WatchEngine = (typeof WATCH_ENGINES)[number]; + +/** + * Scores wobble by a point between runs (a page times out, a nav link + * changes). Mailing somebody about a 1-point flap trains them to ignore us, + * which costs more than the missed signal. Two points is the smallest change + * that is reliably real. + */ +export const WATCH_MIN_DELTA = 2; + +/** A watch is a standing promise to email; an unbounded list per address is a + * free monitoring tier and a spam vector. */ +export const MAX_WATCHES_PER_EMAIL = 10; + +export function normalizeWatchEmail(email: string): string | null { + const e = email.trim().toLowerCase(); + // Deliberately loose — real validation is the confirmation email, which is + // the only thing that proves the address wants this. + if (!e || e.length > 254 || !e.includes("@") || /\s/.test(e)) return null; + const [local, domain, ...rest] = e.split("@"); + if (rest.length > 0 || !local || !domain || !domain.includes(".")) return null; + return e; +} + +export function isWatchCadence(v: unknown): v is WatchCadence { + return typeof v === "string" && (WATCH_CADENCES as readonly string[]).includes(v); +} + +export function isWatchEngine(v: unknown): v is WatchEngine { + return typeof v === "string" && (WATCH_ENGINES as readonly string[]).includes(v); +} + +export function nextRunAt(cadence: WatchCadence, from: Date): Date { + const days = cadence === "weekly" ? 7 : 30; + return new Date(from.getTime() + days * 24 * 60 * 60 * 1000); +} + +/** + * The number a watch tracks — the same one the share card shows, which for a + * slop scan is `summary.slopScore` and NOT `audits.score`. Watching the wrong + * column would email people about a number their report never displays. + */ +export function watchScoreOf(audit: ShareCardAudit): number | null { + const card = buildShareCard(audit); + return card.state === "complete" ? card.score : null; +} + +export type WatchVerdict = { + notify: boolean; + /** "first" on the first completed scan; otherwise the direction of travel. */ + kind: "first" | "improved" | "worsened" | "unchanged"; + delta: number; + /** True when the site got better, accounting for the inverted slop dial. */ + improved: boolean; +}; + +/** + * Decide whether a completed re-scan is worth an email. + * + * The inversion matters more here than anywhere else in the codebase: a slop + * score falling from 60 to 40 is GOOD NEWS, and an email saying "your score + * dropped" would read as the opposite. + */ +export function watchVerdict(input: { + engineKind: "aeo" | "slop"; + previousScore: number | null; + nextScore: number; + minDelta?: number; +}): WatchVerdict { + const minDelta = input.minDelta ?? WATCH_MIN_DELTA; + + // First completed scan under the watch: always worth one email. It confirms + // what we are watching and sets the baseline. + if (input.previousScore === null) { + return { notify: true, kind: "first", delta: 0, improved: false }; + } + + const delta = input.nextScore - input.previousScore; + if (Math.abs(delta) < minDelta) { + return { notify: false, kind: "unchanged", delta, improved: false }; + } + + // AEO: up is better. Slop: down is better (0 = pristine). + const improved = input.engineKind === "slop" ? delta < 0 : delta > 0; + return { + notify: true, + kind: improved ? "improved" : "worsened", + delta, + improved, + }; +} + +/** Subject line for a change email — states the direction in plain words so + * the inverted slop dial is never left for the reader to work out. */ +export function watchSubject(input: { + host: string; + label: string; + score: number; + verdict: WatchVerdict; +}): string { + const { verdict } = input; + if (verdict.kind === "first") { + return `Now watching ${input.host} — ${input.label} ${input.score}/100`; + } + // The word carries the judgement; the sign reports the raw movement of the + // number. Deriving the sign from `improved` instead would print "−4" for a + // 4-point AEO gain, since the two run in opposite directions. + const word = verdict.improved ? "improved" : "got worse"; + const sign = verdict.delta > 0 ? "+" : "−"; + return `${input.host} ${word} — ${input.label} ${input.score}/100 (${sign}${Math.abs(verdict.delta)} pts)`; +} diff --git a/supabase/migrations/20260726120000_scan_watches.sql b/supabase/migrations/20260726120000_scan_watches.sql new file mode 100644 index 00000000..2c1e0914 --- /dev/null +++ b/supabase/migrations/20260726120000_scan_watches.sql @@ -0,0 +1,108 @@ +-- "Watch this URL" — recurring re-scans of a URL for an email address that +-- asked for them (M2 of docs/lead-engine-prd.md). +-- +-- Double opt-in by construction: a row is inert until verified_at is set by +-- the confirmation link. Nothing recurring is ever sent to an address that +-- did not click. + +create table if not exists public.scan_watches ( + id uuid primary key default gen_random_uuid(), + email text not null, + target_url text not null, + engine text not null default 'slop' check (engine in ('rule', 'slop')), + cadence text not null default 'weekly' check (cadence in ('weekly', 'monthly')), + + -- Separate secrets: confirming a watch and killing it are different + -- actions, and the kill link ships in every email we send. + confirm_token text not null unique, + unsubscribe_token text not null unique, + + verified_at timestamptz, + unsubscribed_at timestamptz, + + -- The audit this watch was created from, so the first re-scan can be + -- compared against a number the subscriber has already seen. + origin_audit_id uuid references public.audits(id) on delete set null, + -- An in-flight re-scan. Scans complete asynchronously in the worker, so the + -- cron enqueues here on one tick and delivers the email on a later one. + pending_audit_id uuid references public.audits(id) on delete set null, + + last_score int, + last_scanned_at timestamptz, + last_notified_at timestamptz, + next_run_at timestamptz not null default now(), + + created_ip_hash text, + created_at timestamptz not null default now(), + updated_at timestamptz not null default now() +); + +-- One watch per address per target per engine. Re-submitting the same pair +-- should re-send the confirmation, not stack duplicate rows. +create unique index if not exists scan_watches_email_target_engine_idx + on public.scan_watches (lower(trim(email)), target_url, engine); + +-- The cron's due-work query. +create index if not exists scan_watches_due_idx + on public.scan_watches (next_run_at) + where verified_at is not null and unsubscribed_at is null; + +-- The cron's delivery query. +create index if not exists scan_watches_pending_idx + on public.scan_watches (pending_audit_id) + where pending_audit_id is not null; + +create index if not exists scan_watches_email_idx + on public.scan_watches (lower(trim(email))); + +-- Watch re-scans need their own provenance. Without widening this, the cron's +-- inserts fail the check constraint (verified against prod: the constraint is +-- named audits_triggered_by_check and allows only manual/scheduled), and +-- reusing 'scheduled' would blend watch runs into project-schedule analytics. +alter table public.audits drop constraint if exists audits_triggered_by_check; +alter table public.audits add constraint audits_triggered_by_check + check (triggered_by in ('manual', 'scheduled', 'watch')); + +alter table public.scan_watches enable row level security; + +-- No policies, deliberately. Watches are created and read by anonymous +-- visitors through server actions using the service-role client, which +-- bypasses RLS; exposing them to anon/authenticated would let anyone +-- enumerate which addresses are watching which sites. + +comment on table public.scan_watches is + 'Opt-in recurring re-scans of a URL for an email address. Inert until ' + 'verified_at is set by the confirmation link. Service-role access only.'; + +drop trigger if exists scan_watches_set_updated_at on public.scan_watches; +create trigger scan_watches_set_updated_at + before update on public.scan_watches + for each row execute function public.lx_set_updated_at(); + +-- ============================================================ +-- Cron: tick every 15 minutes. The tick is cheap — it only picks up watches +-- whose own next_run_at has passed (weekly/monthly), so a frequent tick just +-- keeps delivery latency low rather than over-scanning. The same tick also +-- delivers results for scans enqueued on an earlier one. +-- ============================================================ +do $$ +begin + if exists (select 1 from cron.job where jobname = 'crawlproof-scan-watches') then + perform cron.unschedule('crawlproof-scan-watches'); + end if; +end $$; + +select cron.schedule( + 'crawlproof-scan-watches', + '*/15 * * * *', + $$ + select net.http_post( + url := (select value from public.cron_config where key = 'site_url') || '/api/cron/scan-watches', + headers := jsonb_build_object( + 'content-type', 'application/json', + 'x-cron-secret', (select value from public.cron_config where key = 'cron_secret') + ), + body := '{}'::jsonb + ); + $$ +); diff --git a/tests/share-card.test.ts b/tests/share-card.test.ts new file mode 100644 index 00000000..196773d0 --- /dev/null +++ b/tests/share-card.test.ts @@ -0,0 +1,154 @@ +import { describe, it, expect } from "vitest"; +import { buildShareCard, hostOf, type ShareCardAudit } from "@/lib/audit/share-card"; + +function audit(over: Partial = {}): ShareCardAudit { + return { + target_url: "https://www.acme.com/", + status: "complete", + score: 72, + engine: "rule", + summary: { pagesCrawled: 12, pass: 20, warn: 3, fail: 5 }, + ...over, + }; +} + +describe("hostOf", () => { + it("strips protocol and www", () => { + expect(hostOf("https://www.acme.com/pricing?x=1")).toBe("acme.com"); + }); + + it("falls back to something renderable for a malformed URL", () => { + // A card with no hostname at all is worse than a best-effort one. + expect(hostOf("acme.com/pricing")).toBe("acme.com"); + expect(hostOf("not a url")).toBe("not a url"); + }); +}); + +describe("buildShareCard — AEO engine", () => { + it("reads the score from audits.score", () => { + const card = buildShareCard(audit({ score: 72 })); + expect(card.kind).toBe("aeo"); + expect(card.label).toBe("AEO Score"); + expect(card.score).toBe(72); + expect(card.host).toBe("acme.com"); + }); + + it("runs the conventional direction — high is good", () => { + expect(buildShareCard(audit({ score: 92 })).tone).toBe("pass"); + expect(buildShareCard(audit({ score: 65 })).tone).toBe("warn"); + expect(buildShareCard(audit({ score: 20 })).tone).toBe("fail"); + }); + + it("leads with failures when there are any", () => { + expect(buildShareCard(audit()).headline).toBe("5 checks failed · 12 pages crawled"); + }); + + it("leads with passes on a clean scan", () => { + const card = buildShareCard(audit({ summary: { pagesCrawled: 1, pass: 28, fail: 0 } })); + expect(card.headline).toBe("all 28 checks passed · 1 page crawled"); + }); +}); + +describe("buildShareCard — slop engine", () => { + const slop = (over: Partial = {}) => + audit({ + engine: "slop", + // audits.score holds the AEO-style derived score; the headline number + // for a slop scan is summary.slopScore, and they differ. + score: 78, + summary: { + pagesCrawled: 50, + slopScore: 34, + slopGrade: "Careless", + slopIssues: 37, + }, + ...over, + }); + + it("reads summary.slopScore, not audits.score", () => { + const card = buildShareCard(slop()); + expect(card.kind).toBe("slop"); + expect(card.label).toBe("Slop Score"); + // The regression this guards: showing 78 next to a card reading 34. + expect(card.score).toBe(34); + }); + + it("inverts the dial — low is good", () => { + expect(buildShareCard(slop({ summary: { slopScore: 10 } })).tone).toBe("pass"); + expect(buildShareCard(slop({ summary: { slopScore: 40 } })).tone).toBe("warn"); + expect(buildShareCard(slop({ summary: { slopScore: 90 } })).tone).toBe("fail"); + }); + + it("scores the two engines in opposite directions for the same number", () => { + // 90 is excellent AEO and terrible slop. Getting this backwards would + // paint a badly-slopped site green on every social preview. + expect(buildShareCard(audit({ score: 90 })).tone).toBe("pass"); + expect(buildShareCard(slop({ summary: { slopScore: 90 } })).tone).toBe("fail"); + }); + + it("states the direction so a low number is not misread", () => { + expect(buildShareCard(slop()).scaleHint).toBe("0 = pristine · 100 = maximum slop"); + expect(buildShareCard(audit()).scaleHint).toBe("out of 100 · higher is better"); + }); + + it("builds the headline from grade, issues, and pages", () => { + expect(buildShareCard(slop()).headline).toBe("Careless · 37 issues · 50 pages swept"); + }); + + it("names the scan that actually ran in the footer", () => { + // An "SEO · AEO · GEO audit" strapline under a Slop Score describes a + // different product than the one that produced the number. + expect(buildShareCard(slop()).footer).toBe("Free carelessness scan — content, code, design"); + expect(buildShareCard(audit()).footer).toBe("SEO · AEO · GEO audit — free, no signup"); + // Present on every state, not just complete. + expect(buildShareCard(slop({ status: "failed" })).footer).toContain("carelessness"); + }); + + it("detects a slop row whose engine column is null", () => { + // Sibling rows in a multi-engine scan_run can carry a null engine. + const card = buildShareCard(slop({ engine: null })); + expect(card.kind).toBe("slop"); + expect(card.score).toBe(34); + }); +}); + +describe("buildShareCard — run states", () => { + it("renders a pending card while the scan runs", () => { + for (const status of ["queued", "running"]) { + const card = buildShareCard(audit({ status, score: null })); + expect(card.state).toBe("pending"); + expect(card.score).toBeNull(); + expect(card.headline).toBe("Scan running…"); + expect(card.tone).toBe("neutral"); + } + }); + + it("renders a failed card", () => { + const card = buildShareCard(audit({ status: "failed", score: null })); + expect(card.state).toBe("failed"); + expect(card.headline).toBe("Scan failed"); + }); + + it("treats a complete row with no score as pending rather than showing null", () => { + expect(buildShareCard(audit({ score: null })).state).toBe("pending"); + }); +}); + +describe("buildShareCard — degraded rows", () => { + it("never leaves the headline empty when summary keys are missing", () => { + // Rows predating the slopScore/slopIssues summary keys still get a card. + const card = buildShareCard(audit({ summary: {} })); + expect(card.headline).toBe("See the full report"); + }); + + it("tolerates a null summary", () => { + const card = buildShareCard(audit({ summary: null })); + expect(card.score).toBe(72); + expect(card.headline).toBe("See the full report"); + }); + + it("keeps the meter visible at a score of zero and clamps at 100", () => { + expect(buildShareCard(audit({ score: 0 })).fill).toBe(2); + expect(buildShareCard(audit({ score: 100 })).fill).toBe(100); + }); +}); diff --git a/tests/watches.test.ts b/tests/watches.test.ts new file mode 100644 index 00000000..336351b4 --- /dev/null +++ b/tests/watches.test.ts @@ -0,0 +1,171 @@ +import { describe, it, expect } from "vitest"; +import { + MAX_WATCHES_PER_EMAIL, + WATCH_MIN_DELTA, + isWatchCadence, + isWatchEngine, + nextRunAt, + normalizeWatchEmail, + watchScoreOf, + watchSubject, + watchVerdict, +} from "@/lib/watches"; + +describe("normalizeWatchEmail", () => { + it("lowercases and trims", () => { + expect(normalizeWatchEmail(" Person@Example.COM ")).toBe("person@example.com"); + }); + + it("rejects addresses that cannot receive a confirmation", () => { + for (const bad of ["", "nope", "a@b", "two@@at.com", "has space@x.com", "@x.com", "a@"]) { + expect(normalizeWatchEmail(bad)).toBeNull(); + } + }); + + it("rejects absurdly long input", () => { + expect(normalizeWatchEmail(`${"a".repeat(250)}@example.com`)).toBeNull(); + }); +}); + +describe("cadence and engine guards", () => { + it("accepts only the two cadences", () => { + expect(isWatchCadence("weekly")).toBe(true); + expect(isWatchCadence("monthly")).toBe(true); + expect(isWatchCadence("hourly")).toBe(false); + expect(isWatchCadence(undefined)).toBe(false); + }); + + it("accepts only the free self-hosted engines", () => { + // A watch is recurring, so allowing a paid engine here would bill an + // address we never charged, forever. + expect(isWatchEngine("rule")).toBe(true); + expect(isWatchEngine("slop")).toBe(true); + for (const paid of ["claude", "openai", "gemini", "perplexity"]) { + expect(isWatchEngine(paid)).toBe(false); + } + }); +}); + +describe("nextRunAt", () => { + const from = new Date("2026-07-26T00:00:00.000Z"); + + it("schedules a week out", () => { + expect(nextRunAt("weekly", from).toISOString()).toBe("2026-08-02T00:00:00.000Z"); + }); + + it("schedules 30 days out for monthly", () => { + expect(nextRunAt("monthly", from).toISOString()).toBe("2026-08-25T00:00:00.000Z"); + }); +}); + +describe("watchVerdict", () => { + it("always notifies on the first completed scan", () => { + const v = watchVerdict({ engineKind: "aeo", previousScore: null, nextScore: 61 }); + expect(v).toMatchObject({ notify: true, kind: "first", delta: 0 }); + }); + + it("stays quiet on sub-threshold jitter", () => { + const v = watchVerdict({ engineKind: "aeo", previousScore: 60, nextScore: 61 }); + expect(v.notify).toBe(false); + expect(v.kind).toBe("unchanged"); + // Guard the constant itself: raising it silently would mute real changes. + expect(WATCH_MIN_DELTA).toBe(2); + }); + + it("notifies once the change is real", () => { + const v = watchVerdict({ engineKind: "aeo", previousScore: 60, nextScore: 64 }); + expect(v).toMatchObject({ notify: true, kind: "improved", delta: 4, improved: true }); + }); + + it("reads a falling AEO score as worse", () => { + const v = watchVerdict({ engineKind: "aeo", previousScore: 70, nextScore: 55 }); + expect(v).toMatchObject({ kind: "worsened", improved: false, delta: -15 }); + }); + + it("reads a FALLING slop score as better", () => { + // The inversion that matters most: 60 → 40 slop is good news, and an + // email saying "your score dropped" would read as the opposite. + const v = watchVerdict({ engineKind: "slop", previousScore: 60, nextScore: 40 }); + expect(v).toMatchObject({ kind: "improved", improved: true, delta: -20 }); + }); + + it("reads a RISING slop score as worse", () => { + const v = watchVerdict({ engineKind: "slop", previousScore: 20, nextScore: 45 }); + expect(v).toMatchObject({ kind: "worsened", improved: false, delta: 25 }); + }); + + it("judges the same movement oppositely for the two engines", () => { + const move = { previousScore: 30, nextScore: 50 }; + expect(watchVerdict({ engineKind: "aeo", ...move }).improved).toBe(true); + expect(watchVerdict({ engineKind: "slop", ...move }).improved).toBe(false); + }); +}); + +describe("watchSubject", () => { + it("announces the baseline on the first scan", () => { + const verdict = watchVerdict({ engineKind: "slop", previousScore: null, nextScore: 34 }); + expect(watchSubject({ host: "acme.com", label: "Slop Score", score: 34, verdict })).toBe( + "Now watching acme.com — Slop Score 34/100", + ); + }); + + it("signs the delta by the raw movement, not by the judgement", () => { + // A 4-point AEO gain is "improved" AND "+4". Deriving the sign from + // `improved` would print "−4" and contradict the number beside it. + const verdict = watchVerdict({ engineKind: "aeo", previousScore: 60, nextScore: 64 }); + const subject = watchSubject({ host: "acme.com", label: "AEO Score", score: 64, verdict }); + expect(subject).toBe("acme.com improved — AEO Score 64/100 (+4 pts)"); + }); + + it("signs an improving slop score negative, and still says improved", () => { + const verdict = watchVerdict({ engineKind: "slop", previousScore: 60, nextScore: 40 }); + const subject = watchSubject({ host: "acme.com", label: "Slop Score", score: 40, verdict }); + expect(subject).toBe("acme.com improved — Slop Score 40/100 (−20 pts)"); + }); +}); + +describe("watchScoreOf", () => { + it("tracks summary.slopScore for a slop watch, not audits.score", () => { + expect( + watchScoreOf({ + target_url: "https://acme.com", + status: "complete", + score: 78, + engine: "slop", + summary: { slopScore: 34 }, + }), + ).toBe(34); + }); + + it("tracks audits.score for an AEO watch", () => { + expect( + watchScoreOf({ + target_url: "https://acme.com", + status: "complete", + score: 78, + engine: "rule", + summary: {}, + }), + ).toBe(78); + }); + + it("returns null for an unfinished scan so nothing is emailed", () => { + expect( + watchScoreOf({ + target_url: "https://acme.com", + status: "running", + score: null, + engine: "rule", + summary: {}, + }), + ).toBeNull(); + }); +}); + +describe("caps", () => { + it("bounds watches per address", () => { + // Unbounded watches per address is a free monitoring tier by accident. + expect(MAX_WATCHES_PER_EMAIL).toBeGreaterThan(0); + expect(MAX_WATCHES_PER_EMAIL).toBeLessThanOrEqual(25); + }); +});