Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 25 additions & 8 deletions apps/web/src/app/api/following/feed/[format]/route.js
Original file line number Diff line number Diff line change
@@ -1,7 +1,13 @@
import { accounts } from '@rssamplifier/db';
import { SYNDICATION_FORMATS, buildSyndication } from '@rssamplifier/feed';
import {
SYNDICATION_FORMATS,
adSlotsFor,
buildSyndication,
interleaveAds,
} from '@rssamplifier/feed';

import { db, siteUrl } from '../../../../../lib/db.js';
import { fetchFeedAds } from '../../../../../lib/feedAds.js';
import { RIVER_LIMIT, following, followingFeedUrl } from '../../../../../lib/following.js';

export const dynamic = 'force-dynamic';
Expand Down Expand Up @@ -58,6 +64,23 @@ export async function GET(req, { params }) {

const origin = siteUrl();

const rows = items.map((row) => ({
...row,
// The publisher's guid, the same identity every other feed on the site
// uses, so a re-crawl that renumbers our rows does not make a reader show
// the same post twice.
id: String(row.guid ?? row.url ?? ''),
}));

// Sponsored items at the same one-in-ten rate as the topic feeds. The ad is
// not personalised and carries nothing about this account: the request to the
// ad network names the slot and nothing else, and the surface tag is what
// distinguishes this river from a topic's in the advertiser's own analytics.
// That matters here in a way it does not elsewhere — this is the one feed on
// the site that belongs to a particular person.
const wanted = adSlotsFor(rows.length);
const ads = wanted > 0 ? await fetchFeedAds(wanted, { src: 'following' }) : [];

const body = buildSyndication(
format,
{
Expand All @@ -69,13 +92,7 @@ export async function GET(req, { params }) {
link: `${origin}/following`,
selfUrl: followingFeedUrl(origin, token, format),
},
items.map((row) => ({
...row,
// The publisher's guid, the same identity every other feed on the site
// uses, so a re-crawl that renumbers our rows does not make a reader show
// the same post twice.
id: String(row.guid ?? row.url ?? ''),
})),
interleaveAds(rows, ads),
);

return new Response(body, {
Expand Down
11 changes: 11 additions & 0 deletions apps/web/src/lib/ads.js
Original file line number Diff line number Diff line change
Expand Up @@ -19,10 +19,21 @@
* text/plain to CLI clients. ad.js has no size for it and cannot render it in a
* browser, so it is deliberately unused here.
*
* The syndicated feeds are monetised too, but nothing in this file does it:
* a feed has no DOM for ad.js to fill, so the ad has to be *in* the document
* and is fetched while it is built. See ./feedAds.js and `interleaveAds` in the
* feed package.
*
* Deliberately *not* monetised: /llms.txt, /opml, /api/* and the rest of the
* machine-readable surface (the clean copy for agents is the product's whole
* pitch), the framed reader (someone else's article — see the reader page), and
* /offline (no network, so the request could not succeed anyway).
*
* The line between "a feed carries ads" and "/api/* does not" is who the
* document is for. A feed is a subscription a person reads in a reader, and it
* is the same river the ad-carrying web pages show. /api/* and /llms.txt are
* the machine-readable copy an agent consumes, where an ad is noise in a data
* structure rather than a placement anybody sees.
*/

export const AD_SLOT = '2768fe0d-c51c-4629-8d86-0efba3d9ec1f';
Expand Down
174 changes: 174 additions & 0 deletions apps/web/src/lib/feedAds.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,174 @@
/*
* Sponsored items for the syndicated feeds.
*
* The site already carries CrawlProof's web units (see ./ads.js), but a feed is
* not a page: nothing here runs `ad.js`, there is no DOM to fill, and the
* reader is a piece of software that will keep the document for weeks. So the
* ad has to be *in* the document, fetched while we build it.
*
* Two decisions are worth stating, because both are easy to get wrong later.
*
* **We take `as=fields`, not `as=rss`.** CrawlProof will happily hand back a
* ready-made `<item>`, and splicing that string into our XML would be less
* code. It would also mean two different pieces of software decide how a title
* gets escaped inside one document, and the day their idea of escaping differs
* from ours is the day every subscriber's reader reports a parse error on the
* whole feed. Taking the raw fields and rendering them through `buildRss` /
* `buildAtom` / `buildJsonFeed` keeps that decision in exactly one place — the
* same place it is made for the other fifty items.
*
* **Failure is silent and total.** Every path out of here returns `[]`. A feed
* is the product; an ad is revenue on top of it. A slow ad server, an expired
* slot, a network blip — none of those may cost a reader their subscription, so
* there is no retry, no error surfaced upward, and a hard timeout well under
* the time a reader would wait.
*/

import { AD_SLOT } from './ads.js';

/** Where the ad network lives. */
const CRAWLPROOF = 'https://crawlproof.com';

/**
* How long to wait for an ad before giving up on it.
*
* Deliberately short. The feed query has already run by the time we get here,
* so this is time added directly to a response the reader is waiting on, and an
* unsold slot costs nothing while a slow one costs everybody.
*/
const TIMEOUT_MS = 2000;

/**
* How long a fetched ad is reused.
*
* The feeds are served with `max-age=300`, and CrawlProof's default identity
* rotation is daily — so refetching per request would burn an impression for
* every cache miss while returning an item carrying the same guid, which no
* reader would show twice anyway. Matching the feed's own cache window keeps
* the impression count honest about how often the ad was actually published.
*/
const CACHE_MS = 300_000;

/** @type {Map<string, { at: number, items: object[] }>} */
const cache = new Map();

/**
* Is feed advertising on?
*
* Read through a non-literal property access: Next inlines `process.env.FOO` at
* build time, which would bake the build-time value into the image and ignore
* whatever Railway injects at runtime. Same reason `siteUrl()` does it.
*
* Defaults to on. Set `FEED_ADS=0` to turn every sponsored item off without a
* deploy — the kill switch matters more than the toggle, because the thing it
* switches off is written into documents other people keep.
*
* @returns {boolean}
*/
export function feedAdsEnabled() {
const env = process.env;
return String(env['FEED_ADS'] ?? '1') !== '0';
}

/**
* Fetch sponsored items, already in the shape `buildSyndication` renders.
*
* @param {number} count how many to ask for (CrawlProof caps at 5)
* @param {{ src?: string }} [opts] surface tag, so one slot can tell its
* surfaces apart in the advertiser's analytics
* @returns {Promise<object[]>} items, or `[]` for every failure there is
*/
export async function fetchFeedAds(count, { src = 'feed' } = {}) {
const want = Math.min(5, Math.max(0, Math.floor(count)));
if (want === 0 || !feedAdsEnabled() || !AD_SLOT) return [];

const key = `${want}:${src}`;
const hit = cache.get(key);
if (hit && Date.now() - hit.at < CACHE_MS) return hit.items;

const url =
`${CRAWLPROOF}/api/ads/feed?slot=${encodeURIComponent(AD_SLOT)}` +
`&as=fields&n=${want}&src=${encodeURIComponent(src)}`;

let items = [];

try {
const res = await fetch(url, {
signal: AbortSignal.timeout(TIMEOUT_MS),
headers: { accept: 'application/json' },
// Our own cache above is the one that decides; Next's would key on the
// URL and outlive the process in ways that make impressions unaccountable.
cache: 'no-store',
});
if (!res.ok) return remember(key, []);

const body = await res.json();
items = Array.isArray(body?.items) ? body.items.map(toItem).filter(Boolean) : [];
} catch {
// Timeout, DNS, TLS, malformed JSON — all the same answer.
return remember(key, []);
}

return remember(key, items);
}

/**
* @param {string} key
* @param {object[]} items
* @returns {object[]}
*/
function remember(key, items) {
// An empty result is cached too, and on purpose: an unsold slot is the normal
// state of a new placement, and re-asking on every feed request would add the
// timeout to every response for nothing.
cache.set(key, { at: Date.now(), items });
return items;
}

/**
* One `as=fields` payload as a syndication item.
*
* The mapping is where the two vocabularies meet, so it is explicit rather than
* a spread: `guid` is our `id`, the *click* URL is our `url` (that redirector is
* what meters the click and pays the publisher — linking the advertiser
* directly would serve the ad for free), and `html` is the body every renderer
* puts in `content_html`.
*
* `title` is taken as CrawlProof rendered it, disclosure prefix included. We do
* not re-derive it from `headline`: the prefix is the disclosure a reader sees
* in a title-only list, and re-assembling it here would be a second place for
* it to go missing.
*
* @param {any} ad
* @returns {object|null} null when the payload is not usable
*/
function toItem(ad) {
const id = String(ad?.guid ?? '');
const url = String(ad?.url ?? '');
const title = String(ad?.title ?? '');
// Without an identity a reader has nothing to deduplicate on and would show
// the ad again on every poll; without a link there is nothing to click. An ad
// missing either is not a degraded ad, it is a broken item.
if (!id || !url || !title) return null;

return {
id,
url,
title,
summary: typeof ad.body === 'string' && ad.body ? ad.body : null,
content_html: typeof ad.html === 'string' && ad.html ? ad.html : null,
published_at: isoOrNull(ad.publishedAt),
author: null,
image_url: null,
sponsored: true,
};
}

/**
* @param {unknown} value
* @returns {string|null}
*/
function isoOrNull(value) {
const at = new Date(String(value ?? ''));
return Number.isNaN(at.getTime()) ? null : at.toISOString();
}
38 changes: 26 additions & 12 deletions apps/web/src/lib/topicFeed.js
Original file line number Diff line number Diff line change
@@ -1,7 +1,13 @@
import { q } from '@rssamplifier/db';
import { SYNDICATION_FORMATS, buildSyndication } from '@rssamplifier/feed';
import {
SYNDICATION_FORMATS,
adSlotsFor,
buildSyndication,
interleaveAds,
} from '@rssamplifier/feed';

import { db, siteUrl } from './db.js';
import { fetchFeedAds } from './feedAds.js';
import { playerPath, wantsPlayer } from './player.js';
import { slugFromUrl, topicGroup } from './topicGroups.js';

Expand Down Expand Up @@ -140,17 +146,25 @@ export async function topicFeed({
selfUrl: `${page}.${format}`,
};

const body = buildSyndication(
format,
channel,
rows.map((row) => ({
...row,
// The publisher's guid is the item's identity everywhere else in this
// codebase, and it is what keeps a reader from showing the same post
// twice after a re-crawl renumbers our own row ids.
id: String(row.guid ?? row.url ?? ''),
})),
);
const items = rows.map((row) => ({
...row,
// The publisher's guid is the item's identity everywhere else in this
// codebase, and it is what keeps a reader from showing the same post
// twice after a re-crawl renumbers our own row ids.
id: String(row.guid ?? row.url ?? ''),
}));

// Sponsored items, one in ten. Only the document formats: a playlist carries
// an ordered list of things to *play*, and a sponsored line has nothing for a
// player to open — VLC handed one shows the reader an error, which is the
// same reason `video/youtube` enclosures are excluded from them.
//
// The count is worked out before the fetch rather than after, so a feed too
// short to carry an ad never pays for the round trip.
const wanted = spec.media ? 0 : adSlotsFor(items.length);
const ads = wanted > 0 ? await fetchFeedAds(wanted, { src: 'topic' }) : [];

const body = buildSyndication(format, channel, interleaveAds(items, ads));

return new Response(body, {
headers: {
Expand Down
Loading