From c034b525962ce76d83bcc492e57bb8c488ce6453 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Tue, 18 Aug 2026 14:32:18 +0000 Subject: [PATCH] Carry sponsored items in the syndicated feeds The site sells ads on its pages but gave them away on its feeds, which is where most of the reading actually happens: a topic river is a subscription somebody keeps in a reader for months, and until now it was the one surface distributing our content with no way to earn from it. An ad in a feed is not the web unit in a different place. There is no DOM for ad.js to fill and no second request a reader will make, so the ad has to be *in* the document, fetched while it is built. Two decisions are worth stating, because both are easy to undo later. We take crawlproof's `as=fields` rather than its ready-made ``. Splicing their XML into ours would be less code and would put two different pieces of software in charge of escaping inside one document -- and the day their idea of escaping differs from ours is the day every subscriber's reader reports a parse error on the whole feed, not on the ad. Taking raw fields and rendering them through buildRss/buildAtom/buildJsonFeed keeps that decision in the one place it is already made for the other fifty items. Failure is silent and total. Every path out of the fetcher returns []: no retry, no error raised, a 2s timeout, and the empty answer cached too so an unsold slot does not add a round trip to every response. A feed is the product; an ad is revenue on top of it. Placement is one in ten, capped at three. The cap matters more than it looks -- feeds here run to 200 items, and one in ten with no ceiling would be twenty ads, which is not a river with advertising in it. Ads never trail the feed, and `adSlotsFor` exists so the caller fetches exactly as many as will be placed: a fetched-but-dropped ad is an impression metered against an advertiser for reach nobody got. Playlists carry none. An .m3u is an ordered list of things to play and a sponsored line has nothing a player can open -- the same reason `video/youtube` enclosures are already excluded from them. FEED_ADS=0 turns the whole thing off without a deploy, which matters because what it switches off is written into documents other people keep. Co-Authored-By: Claude Opus 5 (1M context) --- .../app/api/following/feed/[format]/route.js | 33 ++- apps/web/src/lib/ads.js | 11 + apps/web/src/lib/feedAds.js | 174 ++++++++++++++ apps/web/src/lib/topicFeed.js | 38 ++- apps/web/test/feed-ads.test.js | 219 ++++++++++++++++++ packages/feed/index.js | 4 + packages/feed/src/syndicate.js | 132 ++++++++++- packages/feed/test/syndicate-ads.test.js | 181 +++++++++++++++ 8 files changed, 770 insertions(+), 22 deletions(-) create mode 100644 apps/web/src/lib/feedAds.js create mode 100644 apps/web/test/feed-ads.test.js create mode 100644 packages/feed/test/syndicate-ads.test.js diff --git a/apps/web/src/app/api/following/feed/[format]/route.js b/apps/web/src/app/api/following/feed/[format]/route.js index a4ca17f..b06cc0a 100644 --- a/apps/web/src/app/api/following/feed/[format]/route.js +++ b/apps/web/src/app/api/following/feed/[format]/route.js @@ -1,7 +1,13 @@ import { accounts } from '@rssamplifier/db'; -import { SYNDICATION_FORMATS, buildSyndication } from '@rssamplifier/feed'; +import { + SYNDICATION_FORMATS, + adSlotsFor, + buildSyndication, + interleaveAds, +} from '@rssamplifier/feed'; import { db, siteUrl } from '../../../../../lib/db.js'; +import { fetchFeedAds } from '../../../../../lib/feedAds.js'; import { RIVER_LIMIT, following, followingFeedUrl } from '../../../../../lib/following.js'; export const dynamic = 'force-dynamic'; @@ -58,6 +64,23 @@ export async function GET(req, { params }) { const origin = siteUrl(); + const rows = items.map((row) => ({ + ...row, + // The publisher's guid, the same identity every other feed on the site + // uses, so a re-crawl that renumbers our rows does not make a reader show + // the same post twice. + id: String(row.guid ?? row.url ?? ''), + })); + + // Sponsored items at the same one-in-ten rate as the topic feeds. The ad is + // not personalised and carries nothing about this account: the request to the + // ad network names the slot and nothing else, and the surface tag is what + // distinguishes this river from a topic's in the advertiser's own analytics. + // That matters here in a way it does not elsewhere — this is the one feed on + // the site that belongs to a particular person. + const wanted = adSlotsFor(rows.length); + const ads = wanted > 0 ? await fetchFeedAds(wanted, { src: 'following' }) : []; + const body = buildSyndication( format, { @@ -69,13 +92,7 @@ export async function GET(req, { params }) { link: `${origin}/following`, selfUrl: followingFeedUrl(origin, token, format), }, - items.map((row) => ({ - ...row, - // The publisher's guid, the same identity every other feed on the site - // uses, so a re-crawl that renumbers our rows does not make a reader show - // the same post twice. - id: String(row.guid ?? row.url ?? ''), - })), + interleaveAds(rows, ads), ); return new Response(body, { diff --git a/apps/web/src/lib/ads.js b/apps/web/src/lib/ads.js index be2535b..4fdedc4 100644 --- a/apps/web/src/lib/ads.js +++ b/apps/web/src/lib/ads.js @@ -19,10 +19,21 @@ * text/plain to CLI clients. ad.js has no size for it and cannot render it in a * browser, so it is deliberately unused here. * + * The syndicated feeds are monetised too, but nothing in this file does it: + * a feed has no DOM for ad.js to fill, so the ad has to be *in* the document + * and is fetched while it is built. See ./feedAds.js and `interleaveAds` in the + * feed package. + * * Deliberately *not* monetised: /llms.txt, /opml, /api/* and the rest of the * machine-readable surface (the clean copy for agents is the product's whole * pitch), the framed reader (someone else's article — see the reader page), and * /offline (no network, so the request could not succeed anyway). + * + * The line between "a feed carries ads" and "/api/* does not" is who the + * document is for. A feed is a subscription a person reads in a reader, and it + * is the same river the ad-carrying web pages show. /api/* and /llms.txt are + * the machine-readable copy an agent consumes, where an ad is noise in a data + * structure rather than a placement anybody sees. */ export const AD_SLOT = '2768fe0d-c51c-4629-8d86-0efba3d9ec1f'; diff --git a/apps/web/src/lib/feedAds.js b/apps/web/src/lib/feedAds.js new file mode 100644 index 0000000..a84ec19 --- /dev/null +++ b/apps/web/src/lib/feedAds.js @@ -0,0 +1,174 @@ +/* + * Sponsored items for the syndicated feeds. + * + * The site already carries CrawlProof's web units (see ./ads.js), but a feed is + * not a page: nothing here runs `ad.js`, there is no DOM to fill, and the + * reader is a piece of software that will keep the document for weeks. So the + * ad has to be *in* the document, fetched while we build it. + * + * Two decisions are worth stating, because both are easy to get wrong later. + * + * **We take `as=fields`, not `as=rss`.** CrawlProof will happily hand back a + * ready-made ``, and splicing that string into our XML would be less + * code. It would also mean two different pieces of software decide how a title + * gets escaped inside one document, and the day their idea of escaping differs + * from ours is the day every subscriber's reader reports a parse error on the + * whole feed. Taking the raw fields and rendering them through `buildRss` / + * `buildAtom` / `buildJsonFeed` keeps that decision in exactly one place — the + * same place it is made for the other fifty items. + * + * **Failure is silent and total.** Every path out of here returns `[]`. A feed + * is the product; an ad is revenue on top of it. A slow ad server, an expired + * slot, a network blip — none of those may cost a reader their subscription, so + * there is no retry, no error surfaced upward, and a hard timeout well under + * the time a reader would wait. + */ + +import { AD_SLOT } from './ads.js'; + +/** Where the ad network lives. */ +const CRAWLPROOF = 'https://crawlproof.com'; + +/** + * How long to wait for an ad before giving up on it. + * + * Deliberately short. The feed query has already run by the time we get here, + * so this is time added directly to a response the reader is waiting on, and an + * unsold slot costs nothing while a slow one costs everybody. + */ +const TIMEOUT_MS = 2000; + +/** + * How long a fetched ad is reused. + * + * The feeds are served with `max-age=300`, and CrawlProof's default identity + * rotation is daily — so refetching per request would burn an impression for + * every cache miss while returning an item carrying the same guid, which no + * reader would show twice anyway. Matching the feed's own cache window keeps + * the impression count honest about how often the ad was actually published. + */ +const CACHE_MS = 300_000; + +/** @type {Map} */ +const cache = new Map(); + +/** + * Is feed advertising on? + * + * Read through a non-literal property access: Next inlines `process.env.FOO` at + * build time, which would bake the build-time value into the image and ignore + * whatever Railway injects at runtime. Same reason `siteUrl()` does it. + * + * Defaults to on. Set `FEED_ADS=0` to turn every sponsored item off without a + * deploy — the kill switch matters more than the toggle, because the thing it + * switches off is written into documents other people keep. + * + * @returns {boolean} + */ +export function feedAdsEnabled() { + const env = process.env; + return String(env['FEED_ADS'] ?? '1') !== '0'; +} + +/** + * Fetch sponsored items, already in the shape `buildSyndication` renders. + * + * @param {number} count how many to ask for (CrawlProof caps at 5) + * @param {{ src?: string }} [opts] surface tag, so one slot can tell its + * surfaces apart in the advertiser's analytics + * @returns {Promise} items, or `[]` for every failure there is + */ +export async function fetchFeedAds(count, { src = 'feed' } = {}) { + const want = Math.min(5, Math.max(0, Math.floor(count))); + if (want === 0 || !feedAdsEnabled() || !AD_SLOT) return []; + + const key = `${want}:${src}`; + const hit = cache.get(key); + if (hit && Date.now() - hit.at < CACHE_MS) return hit.items; + + const url = + `${CRAWLPROOF}/api/ads/feed?slot=${encodeURIComponent(AD_SLOT)}` + + `&as=fields&n=${want}&src=${encodeURIComponent(src)}`; + + let items = []; + + try { + const res = await fetch(url, { + signal: AbortSignal.timeout(TIMEOUT_MS), + headers: { accept: 'application/json' }, + // Our own cache above is the one that decides; Next's would key on the + // URL and outlive the process in ways that make impressions unaccountable. + cache: 'no-store', + }); + if (!res.ok) return remember(key, []); + + const body = await res.json(); + items = Array.isArray(body?.items) ? body.items.map(toItem).filter(Boolean) : []; + } catch { + // Timeout, DNS, TLS, malformed JSON — all the same answer. + return remember(key, []); + } + + return remember(key, items); +} + +/** + * @param {string} key + * @param {object[]} items + * @returns {object[]} + */ +function remember(key, items) { + // An empty result is cached too, and on purpose: an unsold slot is the normal + // state of a new placement, and re-asking on every feed request would add the + // timeout to every response for nothing. + cache.set(key, { at: Date.now(), items }); + return items; +} + +/** + * One `as=fields` payload as a syndication item. + * + * The mapping is where the two vocabularies meet, so it is explicit rather than + * a spread: `guid` is our `id`, the *click* URL is our `url` (that redirector is + * what meters the click and pays the publisher — linking the advertiser + * directly would serve the ad for free), and `html` is the body every renderer + * puts in `content_html`. + * + * `title` is taken as CrawlProof rendered it, disclosure prefix included. We do + * not re-derive it from `headline`: the prefix is the disclosure a reader sees + * in a title-only list, and re-assembling it here would be a second place for + * it to go missing. + * + * @param {any} ad + * @returns {object|null} null when the payload is not usable + */ +function toItem(ad) { + const id = String(ad?.guid ?? ''); + const url = String(ad?.url ?? ''); + const title = String(ad?.title ?? ''); + // Without an identity a reader has nothing to deduplicate on and would show + // the ad again on every poll; without a link there is nothing to click. An ad + // missing either is not a degraded ad, it is a broken item. + if (!id || !url || !title) return null; + + return { + id, + url, + title, + summary: typeof ad.body === 'string' && ad.body ? ad.body : null, + content_html: typeof ad.html === 'string' && ad.html ? ad.html : null, + published_at: isoOrNull(ad.publishedAt), + author: null, + image_url: null, + sponsored: true, + }; +} + +/** + * @param {unknown} value + * @returns {string|null} + */ +function isoOrNull(value) { + const at = new Date(String(value ?? '')); + return Number.isNaN(at.getTime()) ? null : at.toISOString(); +} diff --git a/apps/web/src/lib/topicFeed.js b/apps/web/src/lib/topicFeed.js index 4b48eb1..7cbe504 100644 --- a/apps/web/src/lib/topicFeed.js +++ b/apps/web/src/lib/topicFeed.js @@ -1,7 +1,13 @@ import { q } from '@rssamplifier/db'; -import { SYNDICATION_FORMATS, buildSyndication } from '@rssamplifier/feed'; +import { + SYNDICATION_FORMATS, + adSlotsFor, + buildSyndication, + interleaveAds, +} from '@rssamplifier/feed'; import { db, siteUrl } from './db.js'; +import { fetchFeedAds } from './feedAds.js'; import { playerPath, wantsPlayer } from './player.js'; import { slugFromUrl, topicGroup } from './topicGroups.js'; @@ -140,17 +146,25 @@ export async function topicFeed({ selfUrl: `${page}.${format}`, }; - const body = buildSyndication( - format, - channel, - rows.map((row) => ({ - ...row, - // The publisher's guid is the item's identity everywhere else in this - // codebase, and it is what keeps a reader from showing the same post - // twice after a re-crawl renumbers our own row ids. - id: String(row.guid ?? row.url ?? ''), - })), - ); + const items = rows.map((row) => ({ + ...row, + // The publisher's guid is the item's identity everywhere else in this + // codebase, and it is what keeps a reader from showing the same post + // twice after a re-crawl renumbers our own row ids. + id: String(row.guid ?? row.url ?? ''), + })); + + // Sponsored items, one in ten. Only the document formats: a playlist carries + // an ordered list of things to *play*, and a sponsored line has nothing for a + // player to open — VLC handed one shows the reader an error, which is the + // same reason `video/youtube` enclosures are excluded from them. + // + // The count is worked out before the fetch rather than after, so a feed too + // short to carry an ad never pays for the round trip. + const wanted = spec.media ? 0 : adSlotsFor(items.length); + const ads = wanted > 0 ? await fetchFeedAds(wanted, { src: 'topic' }) : []; + + const body = buildSyndication(format, channel, interleaveAds(items, ads)); return new Response(body, { headers: { diff --git a/apps/web/test/feed-ads.test.js b/apps/web/test/feed-ads.test.js new file mode 100644 index 0000000..39e5c98 --- /dev/null +++ b/apps/web/test/feed-ads.test.js @@ -0,0 +1,219 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; + +// The fetcher's whole job is to be harmless when the ad network is not there, +// so most of what is tested here is failure. Each case stubs global fetch and +// re-imports the module, because the module caches its results by design and a +// shared cache across cases would make them test each other. + +const FIELDS = { + ok: true, + count: 1, + items: [ + { + guid: 'tag:crawlproof.com,2026:ad/slot-1/d/2026-08-18', + title: '[Sponsored] Ship faster with Widgets', + headline: 'Ship faster with Widgets', + body: 'Deploy in one command, roll back in one more.', + cta: 'Try it free', + url: 'https://crawlproof.com/api/ads/click?i=imp-1&s=slot-1', + publishedAt: '2026-08-18T00:00:00.000Z', + html: '

Sponsored

', + label: 'Sponsored', + tier: 'paid', + }, + ], +}; + +/** + * Load a fresh copy of the module with `fetch` stubbed. + * + * The cache-buster on the specifier is what makes each case independent: the + * module holds an in-process TTL cache, so a second import of the same URL + * would answer from the first case's results. + * + * @param {(url: string) => Promise|Response} impl + * @param {Record} [env] + */ +async function load(impl, env = {}) { + const saved = { fetch: globalThis.fetch, ...pick(env) }; + globalThis.fetch = async (url) => impl(String(url)); + for (const [k, v] of Object.entries(env)) process.env[k] = v; + + const mod = await import(`../src/lib/feedAds.js?case=${Math.random()}`); + + return { + mod, + restore() { + globalThis.fetch = saved.fetch; + for (const k of Object.keys(env)) { + if (saved[k] === undefined) delete process.env[k]; + else process.env[k] = saved[k]; + } + }, + }; +} + +function pick(env) { + return Object.fromEntries(Object.keys(env).map((k) => [k, process.env[k]])); +} + +const jsonRes = (body, status = 200) => + new Response(JSON.stringify(body), { status, headers: { 'content-type': 'application/json' } }); + +test('maps an as=fields payload onto a syndication item', async () => { + const { mod, restore } = await load(() => jsonRes(FIELDS)); + try { + const [item] = await mod.fetchFeedAds(1); + + assert.equal(item.id, FIELDS.items[0].guid); + // The click URL, not the advertiser's own site: that redirector is what + // meters the click and pays the publisher. + assert.equal(item.url, FIELDS.items[0].url); + // The title arrives with its disclosure prefix already applied and is used + // as-is, rather than re-assembled here from `headline`. + assert.equal(item.title, '[Sponsored] Ship faster with Widgets'); + assert.equal(item.content_html, FIELDS.items[0].html); + assert.equal(item.published_at, '2026-08-18T00:00:00.000Z'); + assert.equal(item.sponsored, true); + } finally { + restore(); + } +}); + +test('asks for the slot, the field shape, and a surface tag', async () => { + let seen = ''; + const { mod, restore } = await load((url) => { + seen = url; + return jsonRes(FIELDS); + }); + try { + await mod.fetchFeedAds(2, { src: 'topic' }); + + assert.match(seen, /^https:\/\/crawlproof\.com\/api\/ads\/feed\?/); + assert.match(seen, /as=fields/); + assert.match(seen, /[?&]n=2/); + assert.match(seen, /src=topic/); + } finally { + restore(); + } +}); + +test('returns nothing when the ad server errors', async () => { + const { mod, restore } = await load(() => jsonRes({ ok: false }, 500)); + try { + assert.deepEqual(await mod.fetchFeedAds(1), []); + } finally { + restore(); + } +}); + +test('returns nothing when the ad server hangs or refuses', async () => { + const { mod, restore } = await load(() => { + throw new Error('ECONNREFUSED'); + }); + try { + // A feed is the product and an ad is revenue on top of it, so there is no + // path out of here that throws into the caller. + assert.deepEqual(await mod.fetchFeedAds(1), []); + } finally { + restore(); + } +}); + +test('returns nothing when the payload is not the shape it claims', async () => { + const { mod, restore } = await load(() => new Response('oops', { status: 200 })); + try { + assert.deepEqual(await mod.fetchFeedAds(1), []); + } finally { + restore(); + } +}); + +test('drops an item missing an identity or a link', async () => { + const { mod, restore } = await load(() => + jsonRes({ + ok: true, + items: [ + { ...FIELDS.items[0], guid: '' }, + { ...FIELDS.items[0], url: '' }, + { ...FIELDS.items[0], title: '' }, + FIELDS.items[0], + ], + }), + ); + try { + // Without a guid a reader shows the ad again on every poll; without a link + // there is nothing to click. Neither is a degraded ad. + assert.equal((await mod.fetchFeedAds(4)).length, 1); + } finally { + restore(); + } +}); + +test('asks for nothing when no ad would be placed', async () => { + let calls = 0; + const { mod, restore } = await load(() => { + calls += 1; + return jsonRes(FIELDS); + }); + try { + assert.deepEqual(await mod.fetchFeedAds(0), []); + assert.equal(calls, 0, 'a feed too short to carry an ad must not pay for the round trip'); + } finally { + restore(); + } +}); + +test('FEED_ADS=0 turns every sponsored item off without a deploy', async () => { + let calls = 0; + const { mod, restore } = await load( + () => { + calls += 1; + return jsonRes(FIELDS); + }, + { FEED_ADS: '0' }, + ); + try { + assert.equal(mod.feedAdsEnabled(), false); + assert.deepEqual(await mod.fetchFeedAds(3), []); + assert.equal(calls, 0); + } finally { + restore(); + } +}); + +test('reuses a fetched ad rather than burning an impression per request', async () => { + let calls = 0; + const { mod, restore } = await load(() => { + calls += 1; + return jsonRes(FIELDS); + }); + try { + await mod.fetchFeedAds(1, { src: 'topic' }); + await mod.fetchFeedAds(1, { src: 'topic' }); + await mod.fetchFeedAds(1, { src: 'topic' }); + + // The default identity rotation is daily, so three fetches would have + // metered three impressions for an item carrying one guid — reach the + // advertiser paid for and nobody received. + assert.equal(calls, 1); + } finally { + restore(); + } +}); + +test('caches the empty answer too, so an unsold slot is not re-asked per request', async () => { + let calls = 0; + const { mod, restore } = await load(() => { + calls += 1; + return jsonRes({ ok: true, count: 0, items: [] }); + }); + try { + await mod.fetchFeedAds(1); + await mod.fetchFeedAds(1); + assert.equal(calls, 1); + } finally { + restore(); + } +}); diff --git a/packages/feed/index.js b/packages/feed/index.js index 3360b26..45a5f28 100644 --- a/packages/feed/index.js +++ b/packages/feed/index.js @@ -1,6 +1,10 @@ export { slugify, isReserved, uniqueSlug } from './src/slug.js'; export { parseOpml, buildOpml, opmlHead, opmlOutline, opmlFoot } from './src/opml.js'; export { + AD_EVERY, + AD_MAX, + adSlotsFor, + interleaveAds, SYNDICATION_FORMATS, buildSyndication, buildRss, diff --git a/packages/feed/src/syndicate.js b/packages/feed/src/syndicate.js index baf9aa4..954581f 100644 --- a/packages/feed/src/syndicate.js +++ b/packages/feed/src/syndicate.js @@ -100,8 +100,112 @@ export function buildSyndication(format, channel, items) { * @property {string} [feed_title] the publication it came from * @property {string} [feed_slug] * @property {string} [feed_url] that publication's own feed + * @property {boolean} [sponsored] a paid item rather than something a feed published + * @property {string|null} [content_html] a ready-made HTML body, used in place + * of `summary`. Only sponsored items carry one — a crawled post's body is + * somebody else's HTML and is deliberately reduced to a plain-text summary. */ +/** + * How often a sponsored item is placed, and how many a document may carry. + * + * One in ten is the ratio: frequent enough to be worth selling, rare enough + * that a reader scrolling a river meets nine real posts first. The cap matters + * more than it looks — feeds here go up to 200 items, and at one in ten with no + * ceiling a long topic feed would carry twenty ads, which is not a river with + * advertising in it, it is an advertising feed. + */ +export const AD_EVERY = 10; +export const AD_MAX = 3; + +/** + * How many sponsored items a list of this length will actually take. + * + * Exported because the caller has to decide how many ads to *fetch* before it + * can interleave them, and every fetched ad costs an impression the moment the + * ad network records it. If the two disagreed, the difference would be metered + * against an advertiser and then never published — the caller would pay for + * reach nobody got. So the count lives here, next to the placement rule it has + * to match, rather than being re-derived at each call site. + * + * @param {number} total how many real items the document will carry + * @param {{ every?: number, max?: number }} [opts] + * @returns {number} + */ +export function adSlotsFor(total, { every = AD_EVERY, max = AD_MAX } = {}) { + const n = Number(total) || 0; + if (n < every) return 0; + // Boundaries strictly inside the list: the last one is dropped when it would + // land at the end, for the same reason interleaveAds refuses to trail. + const boundaries = Math.ceil(n / every) - 1; + return Math.max(0, Math.min(max, boundaries)); +} + +/** + * Place sponsored items among real ones. + * + * Pure and index-based rather than date-based on purpose: the caller has + * already sorted the river, and an ad carrying today's date would otherwise + * sort itself to the top of every feed regardless of where it was meant to sit. + * + * A list shorter than one full interval gets nothing. A six-post feed cannot + * carry an ad without the ad becoming the feed, and the same rule already + * governs the web units (see apps/web/src/lib/ads.js). + * + * @param {Item[]} items the real posts, in the order they should appear + * @param {Item[]} ads sponsored items, already in Item shape + * @param {{ every?: number, max?: number }} [opts] + * @returns {Item[]} one list, ads interleaved + */ +export function interleaveAds(items, ads, { every = AD_EVERY, max = AD_MAX } = {}) { + if (!Array.isArray(ads) || ads.length === 0) return items; + if (items.length < every) return items; + + const out = []; + let placed = 0; + + for (let i = 0; i < items.length; i += 1) { + out.push(items[i]); + // Never trailing: an ad as the last entry of a feed reads as the feed + // having ended in an advertisement, and costs the slot nothing to skip. + const boundary = (i + 1) % every === 0 && i + 1 < items.length; + if (boundary && placed < max && placed < ads.length) { + out.push(ads[placed]); + placed += 1; + } + } + + return out; +} + +/** + * Wrap a body in CDATA. + * + * The one sequence a CDATA section cannot contain is its own terminator, and + * there is no escape for it — the section has to be closed and reopened around + * the `>`. Only sponsored bodies take this path and ours never produce `]]>`, + * but the copy inside them is written by an advertiser, so this is a + * correctness guard rather than a theoretical one. + * + * @param {unknown} html + * @returns {string} + */ +function cdata(html) { + const safe = String(html ?? '') + // eslint-disable-next-line no-control-regex + .replace(/[\u0000-\u0008\u000B\u000C\u000E-\u001F]/g, ''); + return `/g, ']]]]>')}]]>`; +} + +/** + * The disclosure label carried on a sponsored item. + * + * A constant rather than a per-item field: a reader filtering on `` + * needs one term to filter on, and letting an advertiser choose the wording + * their own disclosure appears under defeats the point of having one. + */ +const SPONSORED = 'Sponsored'; + // --------------------------------------------------------------------- RSS /** @@ -133,7 +237,15 @@ export function buildRss(channel, items) { if (item.url) parts.push(` ${esc(item.url)}`); if (item.published_at) parts.push(` ${esc(rfc822(item.published_at))}`); - if (item.summary) parts.push(` ${esc(item.summary)}`); + // A sponsored item brings its own HTML body and has to be labelled where + // a reader will actually see it: is the machine-readable half, + // and the title it arrived with already carries the human half. + if (item.sponsored) parts.push(` ${esc(SPONSORED)}`); + if (item.content_html) { + parts.push(` ${cdata(item.content_html)}`); + } else if (item.summary) { + parts.push(` ${esc(item.summary)}`); + } // dc:creator rather than , which RSS defines as an email address. // Almost nothing publishes one, and readers show dc:creator anyway. if (item.author) parts.push(` ${esc(item.author)}`); @@ -210,7 +322,15 @@ export function buildAtom(channel, items) { if (item.published_at) parts.push(` ${esc(item.published_at)}`); if (item.url) parts.push(` `); - if (item.summary) parts.push(` ${esc(item.summary)}`); + if (item.sponsored) { + parts.push(` `); + parts.push(` ${esc(SPONSORED)}`); + } + if (item.content_html) { + parts.push(` ${cdata(item.content_html)}`); + } else if (item.summary) { + parts.push(` ${esc(item.summary)}`); + } if (item.author) parts.push(` ${esc(item.author)}`); if (item.feed_title) parts.push(` ${esc(item.feed_title)}`); // Atom has no image element either, and the same Media RSS namespace is @@ -270,6 +390,14 @@ export function buildJsonFeed(channel, items) { if (item.url) out.url = item.url; if (item.summary) out.summary = item.summary; + if (item.content_html) out.content_html = item.content_html; + // JSON Feed has no notion of an ad, so the disclosure goes in both the + // human field readers display and an extension field an agent can branch + // on. The `_` prefix is the spec's own marker for "this is ours". + if (item.sponsored) { + out.tags = [SPONSORED]; + out._crawlproof = { sponsored: true, label: SPONSORED }; + } if (item.image_url) out.image = item.image_url; if (item.published_at) out.date_published = item.published_at; if (item.author) out.authors = [{ name: item.author }]; diff --git a/packages/feed/test/syndicate-ads.test.js b/packages/feed/test/syndicate-ads.test.js new file mode 100644 index 0000000..28c044f --- /dev/null +++ b/packages/feed/test/syndicate-ads.test.js @@ -0,0 +1,181 @@ +import test from 'node:test'; +import assert from 'node:assert/strict'; + +import { XMLParser, XMLValidator } from 'fast-xml-parser'; + +import { + AD_EVERY, + AD_MAX, + adSlotsFor, + buildAtom, + buildJsonFeed, + buildM3u, + buildPls, + buildRss, + interleaveAds, +} from '../src/syndicate.js'; + +// A sponsored item is the one entry in these documents we did not read off a +// publisher — it is third-party copy we insert ourselves. So what is tested +// here is that it cannot break the document it is inserted into, that a reader +// can always tell it apart from a post, and that it lands where it was meant +// to rather than at the top or the end. + +const parser = new XMLParser({ ignoreAttributes: false, attributeNamePrefix: '@' }); + +const channel = { + title: 'physics — RSS Amplifier', + description: 'Recent posts from the feeds that cover physics.', + link: 'https://rssamplifier.com/topics/physics', + selfUrl: 'https://rssamplifier.com/topics/physics.rss', +}; + +const post = (n) => ({ + id: `https://example.com/posts/${n}`, + url: `https://example.com/posts/${n}`, + title: `Post ${n}`, + summary: 'A real post from a real blog.', + published_at: '2026-08-18T09:00:00.000Z', + feed_title: 'Example Blog', + feed_url: 'https://example.com/feed.xml', +}); + +const posts = (n) => Array.from({ length: n }, (_, i) => post(i + 1)); + +const ad = { + id: 'tag:crawlproof.com,2026:ad/slot-1/d/2026-08-18', + url: 'https://crawlproof.com/api/ads/click?i=imp-1&s=slot-1&c=camp-1', + title: '[Sponsored] Ship faster with Widgets', + summary: 'Deploy in one command, roll back in one more.', + content_html: '

Sponsored · Widgets

', + published_at: '2026-08-18T00:00:00.000Z', + sponsored: true, +}; + +test('interleaveAds places one ad per interval, after real posts', () => { + const out = interleaveAds(posts(30), [ad, { ...ad, id: 'ad-2' }, { ...ad, id: 'ad-3' }]); + + // Boundaries fall after the 10th, 20th and 30th post — but the 30th is the + // last one, and an ad may never trail the feed. So two are placed, and the + // third is not: 30 posts + 2 ads. + assert.equal(out.length, 32); + assert.equal(out[10].sponsored, true); + assert.equal(out[21].sponsored, true); + assert.equal(out.filter((i) => i.sponsored).length, 2); + assert.equal(out.at(-1).sponsored, undefined, 'a feed must not end on an ad'); +}); + +test('adSlotsFor predicts exactly what interleaveAds will place', () => { + // The contract that keeps an impression from being metered for an ad that is + // then dropped on the floor. + const many = Array.from({ length: 20 }, (_, i) => ({ ...ad, id: `ad-${i}` })); + for (const total of [0, 1, 9, 10, 11, 19, 20, 21, 29, 30, 31, 50, 200]) { + assert.equal( + interleaveAds(posts(total), many).filter((i) => i.sponsored).length, + adSlotsFor(total), + `total=${total}`, + ); + } +}); + +test('interleaveAds leaves a short list alone', () => { + // Nine posts cannot carry an ad without the ad becoming the feed. + const short = posts(AD_EVERY - 1); + assert.deepEqual(interleaveAds(short, [ad]), short); +}); + +test('interleaveAds honours the cap on a long feed', () => { + // 200 items at one in ten would be twenty ads without the ceiling. + const out = interleaveAds(posts(200), Array.from({ length: 20 }, (_, i) => ({ ...ad, id: `ad-${i}` }))); + assert.equal(out.filter((i) => i.sponsored).length, AD_MAX); +}); + +test('interleaveAds is a no-op when nothing was sold', () => { + const items = posts(50); + assert.deepEqual(interleaveAds(items, []), items); + assert.deepEqual(interleaveAds(items, null), items); +}); + +test('a sponsored item does not break the RSS document', () => { + const xml = buildRss(channel, interleaveAds(posts(20), [ad])); + assert.equal(XMLValidator.validate(xml), true); + + const parsed = parser.parse(xml); + const found = parsed.rss.channel.item.find((i) => i.category === 'Sponsored'); + + assert.ok(found, 'the ad is labelled with a category a reader can filter on'); + assert.equal(found.title, ad.title); + // The click URL, not the advertiser's own: that redirector is what meters the + // click and pays us. Linking the advertiser directly would serve it for free. + assert.equal(found.link, ad.url); + // fast-xml-parser hands attributes back as strings unless told otherwise. + assert.equal(found.guid['@isPermaLink'], 'false'); + assert.match(String(found.description), /Sponsored/); +}); + +test('a sponsored item does not break the Atom document', () => { + const xml = buildAtom(channel, interleaveAds(posts(20), [ad])); + assert.equal(XMLValidator.validate(xml), true); + + const parsed = parser.parse(xml); + const found = parsed.feed.entry.find((e) => e.rights === 'Sponsored'); + + assert.ok(found); + assert.equal(found.category['@term'], 'sponsored'); + assert.equal(found.id, ad.id); + assert.ok(found.content, 'the ad body rides in , not '); +}); + +test('a sponsored item is marked in JSON Feed for humans and for agents', () => { + const doc = JSON.parse(buildJsonFeed(channel, interleaveAds(posts(20), [ad]))); + const found = doc.items.find((i) => i._crawlproof); + + assert.ok(found); + assert.equal(found._crawlproof.sponsored, true); + assert.deepEqual(found.tags, ['Sponsored']); + assert.equal(found.content_html, ad.content_html); +}); + +test('advertiser copy that is itself markup cannot escape its element', () => { + const hostile = { + ...ad, + title: 'Buy Injected', + content_html: 'ends with ]]> a terminator', + }; + const xml = buildRss(channel, interleaveAds(posts(20), [hostile])); + + assert.equal(XMLValidator.validate(xml), true); + const parsed = parser.parse(xml); + // 20 posts + 1 ad. An injected <item> would make it 22. + assert.equal(parsed.rss.channel.item.length, 21); + const found = parsed.rss.channel.item.find((i) => i.category === 'Sponsored'); + assert.equal(found.description, 'ends with ]]> a terminator'); +}); + +test('a real post still renders exactly as it did', () => { + // content_html and sponsored are opt-in; a crawled post carries neither, and + // its body must stay the plain-text summary it has always been. + const xml = buildRss(channel, posts(2)); + const parsed = parser.parse(xml); + + assert.equal(parsed.rss.channel.item[0].description, 'A real post from a real blog.'); + assert.equal(parsed.rss.channel.item[0].category, undefined); + assert.equal(XMLValidator.validate(xml), true); +}); + +test('playlists carry no ads, because an ad is not playable', () => { + // The call sites do not splice into media formats at all, but if one ever + // did, a sponsored line has no enclosure and must be skipped rather than + // handed to a player as a file it cannot open. + const mixed = interleaveAds( + posts(20).map((p) => ({ ...p, audio_url: 'https://example.com/a.mp3' })), + [ad], + ); + + const m3u = buildM3u(channel, mixed); + const pls = buildPls(channel, mixed); + + assert.ok(!m3u.includes('Sponsored')); + assert.ok(!pls.includes('Sponsored')); + assert.match(pls, /NumberOfEntries=20/); +});