diff --git a/app/(app)/dashboard/ads/page.tsx b/app/(app)/dashboard/ads/page.tsx index b8864d8..5d24086 100644 --- a/app/(app)/dashboard/ads/page.tsx +++ b/app/(app)/dashboard/ads/page.tsx @@ -70,7 +70,13 @@ export default async function AdsPage({ // A failed stats query zero-fills, so without this the page would report a // confident 0 for a range it simply could not read. See Loaded<> in // lib/ads/series.ts. + // + // Tracked per loader, not as one flag: the chart and the per-campaign + // sparklines come from different RPCs, and either can fail on its own. Only + // the panel whose query actually failed should say so. let statsFailed = false; + let seriesFailed = false; + let dailyFailed = false; if (user) { const [{ data }, { data: profile }, accountSeries, campaignTotals] = await Promise.all([ supabase @@ -97,6 +103,8 @@ export default async function AdsPage({ 30, ); seriesById = daily.data; + seriesFailed = accountSeries.failed; + dailyFailed = daily.failed; statsFailed = accountSeries.failed || campaignTotals.failed || daily.failed; } @@ -185,7 +193,7 @@ export default async function AdsPage({ )}
- +
)} @@ -222,7 +230,7 @@ export default async function AdsPage({
- + {display.label} diff --git a/components/ads/account-trend.tsx b/components/ads/account-trend.tsx index 74f81b5..d33a343 100644 --- a/components/ads/account-trend.tsx +++ b/components/ads/account-trend.tsx @@ -26,9 +26,31 @@ import type { AccountPoint } from "@/lib/ads/series"; * and the split is the point — free backfill is delivery that earns nobody * anything, and it should be visible as a share of the whole. */ -export function AccountTrend({ data, range }: { data: AccountPoint[]; range: RangeDef }) { +export function AccountTrend({ + data, + range, + failed = false, +}: { + data: AccountPoint[]; + range: RangeDef; + failed?: boolean; +}) { const total = data.reduce((n, p) => n + p.impressions + p.freeImpressions + p.clicks, 0); + // A failed series arrives zero-filled and so is indistinguishable from a quiet + // range by its values alone. Saying "No delivery" here would be a statement of + // fact about the network, made from a query that never returned -- the same + // mistake #226 fixed for the tiles, still being made by the chart underneath + // them, and directly contradicting the banner the page renders above. + if (failed) { + return ( +
+ Couldn't load this chart. It isn't empty — the query didn't + return. Reloading often fixes it. +
+ ); + } + if (total === 0) { return (
diff --git a/components/ads/mini-trend.tsx b/components/ads/mini-trend.tsx index a4597c9..daca62c 100644 --- a/components/ads/mini-trend.tsx +++ b/components/ads/mini-trend.tsx @@ -9,11 +9,28 @@ export function MiniTrend({ data, width = 132, height = 34, + failed = false, }: { data: CampaignDailyPoint[]; width?: number; height?: number; + failed?: boolean; }) { + // Zero-filled on failure, so "no traffic yet" would assert something about the + // campaign that the query never established. See AccountTrend for the same + // distinction on the chart above. + if (failed) { + return ( +
+ unavailable +
+ ); + } + const total = data.reduce((n, p) => n + p.impressions, 0); // Count clicks as traffic too (a campaign can have clicks logged without a // matching impression row), matching CampaignTrend's empty check. `total` diff --git a/lib/ads/series.ts b/lib/ads/series.ts index 30a460e..5a54c21 100644 --- a/lib/ads/series.ts +++ b/lib/ads/series.ts @@ -352,12 +352,23 @@ type DailySeriesRow = { * Aggregated server-side by the ad_campaign_daily_series RPC (security definer, * scoped by its own `owner_id = auth.uid()` filter rather than by RLS — see * 20260901153000_ad_reporting_rpcs_security_definer.sql for why). We must NOT - * fetch and bucket raw ad_impressions rows here: PostgREST caps a response at 1000 rows, so once - * total impressions in the window exceed 1000 a few high-volume campaigns eat - * the whole page and every other campaign gets zero rows back — rendering - * "no traffic yet" despite having recent impressions. The RPC returns at most - * (campaigns * days) rows, so it never hits the cap. - * See migration 20260717032002_ad_campaign_daily_series_rpc.sql. + * fetch and bucket raw ad_impressions rows here: PostgREST caps a response at + * 1000 rows, so once total impressions in the window exceed 1000 a few + * high-volume campaigns eat the whole page and every other campaign gets zero + * rows back — rendering "no traffic yet" despite having recent impressions. + * + * The RPC used to return a row per campaign-day, which was said here to be + * safely under that cap and stopped being true as the account grew: 139 + * campaigns over 30 days is 2,731 rows, and the call was coming back + * `206 Partial Content, content-range 0-999/2731`. With no ORDER BY in the + * function, which two thirds got dropped was down to the join order — campaigns + * silently lost days off the end of their sparkline and five lost every row, + * rendering "no traffic yet" beside a row reading "Impressions: 24". + * + * So it now returns one jsonb array instead. One document is one row, and the + * cap cannot apply however many campaigns the account grows to. `data` is that + * array already parsed. See 20260902140100_ad_reporting_rpcs_on_rollups.sql, + * which is also where it stopped scanning raw events. */ export async function getCampaignDailySeries( supabase: SupabaseClient, diff --git a/supabase/migrations/20260902140000_ad_stats_rollups.sql b/supabase/migrations/20260902140000_ad_stats_rollups.sql new file mode 100644 index 0000000..1a9ce27 --- /dev/null +++ b/supabase/migrations/20260902140000_ad_stats_rollups.sql @@ -0,0 +1,256 @@ +-- Pre-aggregate ad delivery so the dashboards stop scanning raw events. +-- +-- /dashboard/ads renders no graphs, intermittently. The chart falls back to +-- "No delivery in this range." and every sparkline to "no traffic yet" because +-- ad_account_series and ad_campaign_daily_series are being cancelled by the 8s +-- statement_timeout on `authenticated`. Measured over 24h on 2026-09-02: +-- ad_account_series avg 1,453ms / max 8,039ms, ad_campaign_daily_series avg +-- 1,719ms / max 8,027ms, and 14 of ~150 reporting calls returned HTTP 500. +-- +-- This is NOT the RLS nested-loop from 20260901153000 -- that fix held, the +-- plan is a clean hash join. It is simply that every dashboard load aggregates +-- the whole event history from scratch: +-- +-- Seq Scan on ad_impressions (actual rows=177858) +-- Filter: ((NOT duplicate) AND (ts >= now() - '30 days')) +-- Rows Removed by Filter: 198394 +-- Buffers: shared hit=9767 -- the entire 78MB heap, every load +-- +-- and the page issues three such RPCs per render. ad_impressions is 376k rows +-- growing ~90k/day, so the cost rises with the archive rather than with the +-- window being asked for, and no index fixes that: the 30-day window selects +-- 47% of the table, far past the point an index scan can win. A covering +-- partial index (added below, and it is still worth having -- see the RPCs) was +-- measured at 157ms against 194ms for the seq scan, a 20% gain on a query that +-- needs to be 10x faster. +-- +-- So aggregate once, on a schedule, and let a page load read buckets instead of +-- events. The grain is chosen by what each surface actually plots: +-- +-- ad_stats_owner_hourly (owner_id, hour) ~24 rows/day +-- ad_stats_campaign_daily (campaign_id, day) ~135 rows/day +-- ad_stats_slot_daily (slot_id, day) ~27 rows/day +-- +-- A 30-day account query now reads ~720 rows where it read 246,506. +-- +-- Per-campaign is deliberately DAILY and not hourly: (campaign_id, hour) is +-- 2,997 distinct cells in a single day against 139 campaigns, which over the +-- archive is ~170k rows -- barely smaller than the raw table it replaces, so it +-- would buy nothing. Account-wide has no campaign dimension at all, which is +-- what makes hourly affordable there, and hourly is required there because the +-- 1W range plots 4-hour buckets. +-- +-- FRESHNESS. Rollup rows are only ever read for periods that have closed; the +-- current hour and the current UTC day are always read from raw. So a stale or +-- even a completely un-run refresh can never show wrong numbers for the live +-- edge, and the raw slices it reads are at most one hour / one day wide, which +-- is where the covering index earns its place. + +-- The reporting predicate is always (not duplicate) over a ts window, and the +-- columns wanted are few enough to ride along in the index. This serves the +-- narrow live-edge reads the RPCs below fall back to, and the refresh itself. +create index if not exists ad_impressions_reporting_idx + on public.ad_impressions (ts) include (campaign_id, slot_id, tier) + where not duplicate; + +create index if not exists ad_clicks_reporting_idx + on public.ad_clicks (ts) include (campaign_id, slot_id, tier, valid, charged_cents, publisher_earn_cents); + +-- --------------------------------------------------------------------------- +-- Rollup tables +-- --------------------------------------------------------------------------- +-- Column names mirror the RPC output they feed, including the split that has +-- caught this codebase out three times: `paid_*` and `free_*` are halves, never +-- totals. Anything reading one without the other is reading a network where +-- every fill is free backfill as though it were dead. + +create table if not exists public.ad_stats_owner_hourly ( + owner_id uuid not null, + hour timestamptz not null, + paid_impressions bigint not null default 0, + free_impressions bigint not null default 0, + valid_clicks bigint not null default 0, + free_clicks bigint not null default 0, + spent_cents bigint not null default 0, + primary key (owner_id, hour) +); + +create table if not exists public.ad_stats_campaign_daily ( + campaign_id uuid not null, + day date not null, + paid_impressions bigint not null default 0, + free_impressions bigint not null default 0, + valid_clicks bigint not null default 0, + free_clicks bigint not null default 0, + spent_cents bigint not null default 0, + primary key (campaign_id, day) +); + +create table if not exists public.ad_stats_slot_daily ( + slot_id uuid not null, + day date not null, + paid_impressions bigint not null default 0, + free_impressions bigint not null default 0, + valid_clicks bigint not null default 0, + free_clicks bigint not null default 0, + invalid_clicks bigint not null default 0, + earned_cents bigint not null default 0, + primary key (slot_id, day) +); + +-- Read exclusively through the security-definer RPCs below, which do their own +-- ownership filtering. RLS on with no policy is the intended posture: it denies +-- every direct PostgREST read while the definer functions (owned by the +-- migration role) still see the rows. +alter table public.ad_stats_owner_hourly enable row level security; +alter table public.ad_stats_campaign_daily enable row level security; +alter table public.ad_stats_slot_daily enable row level security; + +revoke all on public.ad_stats_owner_hourly from anon, authenticated; +revoke all on public.ad_stats_campaign_daily from anon, authenticated; +revoke all on public.ad_stats_slot_daily from anon, authenticated; + +-- --------------------------------------------------------------------------- +-- Refresh +-- --------------------------------------------------------------------------- + +/** + * Recompute every rollup for events at or after p_from and upsert the result. + * + * Idempotent, so re-running over a period that is already rolled up is a no-op + * in effect. Deliberately a full recompute of the touched periods rather than a + * delta: ad_clicks rows are updated after insert (a click is charged, or later + * invalidated), so counting only new rows would drift. Recomputing a bounded + * tail cannot. + * + * Periods with no events are simply absent; the RPCs zero-fill their own axis, + * as they always have. + */ +create or replace function public.ad_stats_rollup_refresh( + p_from timestamptz default (now() - interval '2 days') +) returns void +language plpgsql +security definer +set search_path to 'public' +as $fn$ +declare + -- Snap the window down to whole periods before recomputing. p_from is a plain + -- "two days ago" from the scheduler, which lands mid-day: filtering raw events + -- on it directly would recompute 08-31 from 13:40 onwards and then upsert that + -- partial count over the complete row already stored for that day. Every + -- period this function touches has to be recomputed in full or not at all. + v_from_hour timestamptz := date_trunc('hour', p_from); + v_from_day timestamptz := date_trunc('day', p_from at time zone 'UTC') at time zone 'UTC'; +begin + -- Account-wide hourly, advertiser side: an impression belongs to the owner of + -- the campaign that was shown. + insert into public.ad_stats_owner_hourly as t + (owner_id, hour, paid_impressions, free_impressions, valid_clicks, free_clicks, spent_cents) + select c.owner_id, + date_trunc('hour', e.ts), + sum(e.paid)::bigint, sum(e.free)::bigint, sum(e.clk)::bigint, + sum(e.fclk)::bigint, sum(e.spent)::bigint + from ( + select i.campaign_id, i.ts, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as spent + from public.ad_impressions i + where not i.duplicate and i.ts >= v_from_hour + union all + select cl.campaign_id, cl.ts, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + where cl.ts >= v_from_hour + ) e + join public.ad_campaigns c on c.id = e.campaign_id + group by 1, 2 + on conflict (owner_id, hour) do update set + paid_impressions = excluded.paid_impressions, + free_impressions = excluded.free_impressions, + valid_clicks = excluded.valid_clicks, + free_clicks = excluded.free_clicks, + spent_cents = excluded.spent_cents; + + insert into public.ad_stats_campaign_daily as t + (campaign_id, day, paid_impressions, free_impressions, valid_clicks, free_clicks, spent_cents) + select e.campaign_id, + (e.ts at time zone 'UTC')::date, + sum(e.paid)::bigint, sum(e.free)::bigint, sum(e.clk)::bigint, + sum(e.fclk)::bigint, sum(e.spent)::bigint + from ( + select i.campaign_id, i.ts, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as spent + from public.ad_impressions i + where not i.duplicate and i.ts >= v_from_day + union all + select cl.campaign_id, cl.ts, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + where cl.ts >= v_from_day + ) e + where e.campaign_id is not null + group by 1, 2 + on conflict (campaign_id, day) do update set + paid_impressions = excluded.paid_impressions, + free_impressions = excluded.free_impressions, + valid_clicks = excluded.valid_clicks, + free_clicks = excluded.free_clicks, + spent_cents = excluded.spent_cents; + + -- Publisher side. invalid_clicks is its own bucket on purpose: a refused + -- click is not delivery, and folding it into free_clicks would put a + -- double-digit CTR on the earnings page. + insert into public.ad_stats_slot_daily as t + (slot_id, day, paid_impressions, free_impressions, valid_clicks, free_clicks, invalid_clicks, earned_cents) + select e.slot_id, + (e.ts at time zone 'UTC')::date, + sum(e.paid)::bigint, sum(e.free)::bigint, sum(e.clk)::bigint, + sum(e.fclk)::bigint, sum(e.iclk)::bigint, sum(e.earned)::bigint + from ( + select i.slot_id, i.ts, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as iclk, 0 as earned + from public.ad_impressions i + where not i.duplicate and i.ts >= v_from_day + union all + select cl.slot_id, cl.ts, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when not cl.valid and cl.tier <> 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.publisher_earn_cents, 0) else 0 end + from public.ad_clicks cl + where cl.ts >= v_from_day + ) e + where e.slot_id is not null + group by 1, 2 + on conflict (slot_id, day) do update set + paid_impressions = excluded.paid_impressions, + free_impressions = excluded.free_impressions, + valid_clicks = excluded.valid_clicks, + free_clicks = excluded.free_clicks, + invalid_clicks = excluded.invalid_clicks, + earned_cents = excluded.earned_cents; +end; +$fn$; + +revoke all on function public.ad_stats_rollup_refresh(timestamptz) from anon, authenticated; + +-- Backfill the whole archive once. p_from covers every event ever recorded. +select public.ad_stats_rollup_refresh('epoch'::timestamptz); + +-- Close out finished hours/days. Ten minutes is well inside the one-hour +-- staleness the read path tolerates, and each run only touches two days. +select cron.schedule( + 'ad-stats-rollup', + '*/10 * * * *', + $cron$select public.ad_stats_rollup_refresh(now() - interval '2 days')$cron$ +); + diff --git a/supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql b/supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql new file mode 100644 index 0000000..7b9b35a --- /dev/null +++ b/supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql @@ -0,0 +1,404 @@ +-- Point the five ad reporting RPCs at the rollups from 20260902140000. +-- +-- Every function keeps its name, arguments and output columns, so nothing that +-- reads them has to change except ad_campaign_daily_series (see below). What +-- changes is where the numbers come from: +-- +-- closed periods -> ad_stats_* rollup rows +-- the live edge -> raw ad_impressions / ad_clicks +-- +-- "Closed" means strictly before the current hour (account series) or the +-- current UTC day (everything else). The live edge is therefore at most one +-- hour or one day wide no matter how long the requested window is, which is +-- what makes the cost proportional to the number of buckets plotted rather +-- than to the size of the archive. +-- +-- The split is exact, not approximate. When p_since lands mid-period the +-- leading partial period is read from raw too, so a window starting at +-- 13:24 still reports 13:24 onwards and not 14:00 onwards. That is the reason +-- for first_full_hour / first_full_day below rather than a plain date_trunc. +-- +-- Semantics are unchanged and deliberately restated here because they have +-- been misread three times: `impressions` and `clicks` are the PAID and VALID +-- halves, `free_impressions` / `free_clicks` are the others, and a caller that +-- reads one without the other sees zero on a network where every fill is free +-- backfill. + +-- --------------------------------------------------------------------------- +-- Account-wide series +-- --------------------------------------------------------------------------- +create or replace function public.ad_account_series( + p_since timestamptz default null, + p_bucket_seconds integer default 86400 +) returns table( + bucket timestamptz, impressions bigint, free_impressions bigint, + clicks bigint, free_clicks bigint, spent_cents bigint +) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + secs int := greatest(coalesce(p_bucket_seconds, 86400), 60); + step interval := make_interval(secs => secs); + uid uuid := auth.uid(); + this_hour timestamptz := date_trunc('hour', now()); + first_full_hour timestamptz; +begin + if uid is null then + return; + end if; + + -- Buckets finer than the rollup grain (the 1H/4H/1D tabs, at 60/300/1800s) + -- have to come from raw events. They only ever span the last 24 hours, so + -- the ts index makes that a narrow read rather than the full-archive scan + -- this function used to do for every range alike. + if secs < 3600 then + return query + with ev as ( + select date_bin(step, i.ts, timestamptz 'epoch') as b, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as spent + from public.ad_impressions i + join public.ad_campaigns c on c.id = i.campaign_id and c.owner_id = uid + where not i.duplicate and (p_since is null or i.ts >= p_since) + union all + select date_bin(step, cl.ts, timestamptz 'epoch'), 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + join public.ad_campaigns c on c.id = cl.campaign_id and c.owner_id = uid + where (p_since is null or cl.ts >= p_since) + ) + select b, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(spent)::bigint + from ev group by b order by b; + return; + end if; + + first_full_hour := case + when p_since is null then null + when p_since = date_trunc('hour', p_since) then p_since + else date_trunc('hour', p_since) + interval '1 hour' + end; + + return query + with ev as ( + select date_bin(step, r.hour, timestamptz 'epoch') as b, + r.paid_impressions as paid, r.free_impressions as free, + r.valid_clicks as clk, r.free_clicks as fclk, r.spent_cents as spent + from public.ad_stats_owner_hourly r + where r.owner_id = uid + and r.hour < this_hour + and (first_full_hour is null or r.hour >= first_full_hour) + union all + select date_bin(step, i.ts, timestamptz 'epoch'), + case when i.tier = 'free' then 0 else 1 end, + case when i.tier = 'free' then 1 else 0 end, + 0, 0, 0 + from public.ad_impressions i + join public.ad_campaigns c on c.id = i.campaign_id and c.owner_id = uid + where not i.duplicate + and (p_since is null or i.ts >= p_since) + and (i.ts >= this_hour + or (first_full_hour is not null and i.ts < first_full_hour)) + union all + select date_bin(step, cl.ts, timestamptz 'epoch'), 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + join public.ad_campaigns c on c.id = cl.campaign_id and c.owner_id = uid + where (p_since is null or cl.ts >= p_since) + and (cl.ts >= this_hour + or (first_full_hour is not null and cl.ts < first_full_hour)) + ) + select b, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(spent)::bigint + from ev group by b order by b; +end; +$$; + +-- --------------------------------------------------------------------------- +-- Per-campaign totals for a window +-- --------------------------------------------------------------------------- +create or replace function public.ad_campaign_totals( + p_since timestamptz default null +) returns table( + campaign_id uuid, impressions bigint, free_impressions bigint, + clicks bigint, free_clicks bigint, spent_cents bigint +) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + first_full_day date; + first_full_start timestamptz; +begin + if uid is null then + return; + end if; + + first_full_day := case + when p_since is null then null + when p_since = ((p_since at time zone 'UTC')::date::timestamp at time zone 'UTC') + then (p_since at time zone 'UTC')::date + else (p_since at time zone 'UTC')::date + 1 + end; + first_full_start := (first_full_day::timestamp at time zone 'UTC'); + + return query + with owned as ( + select id from public.ad_campaigns where owner_id = uid + ), + ev as ( + select r.campaign_id, + r.paid_impressions as paid, r.free_impressions as free, + r.valid_clicks as clk, r.free_clicks as fclk, r.spent_cents as spent + from public.ad_stats_campaign_daily r + where r.campaign_id in (select id from owned) + and r.day < today + and (first_full_day is null or r.day >= first_full_day) + union all + select i.campaign_id, + case when i.tier = 'free' then 0 else 1 end, + case when i.tier = 'free' then 1 else 0 end, + 0, 0, 0 + from public.ad_impressions i + where i.campaign_id in (select id from owned) + and not i.duplicate + and (p_since is null or i.ts >= p_since) + and (i.ts >= today_start + or (first_full_start is not null and i.ts < first_full_start)) + union all + select cl.campaign_id, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + where cl.campaign_id in (select id from owned) + and (p_since is null or cl.ts >= p_since) + and (cl.ts >= today_start + or (first_full_start is not null and cl.ts < first_full_start)) + ) + select ev.campaign_id, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(spent)::bigint + from ev group by ev.campaign_id; +end; +$$; + +-- --------------------------------------------------------------------------- +-- Per-campaign daily series +-- --------------------------------------------------------------------------- +-- Now returns a single jsonb array instead of a set of rows, because the set +-- had outgrown PostgREST's 1000-row response cap: 139 campaigns over 30 days is +-- 2,731 rows, and the request was coming back `206 Partial Content, +-- content-range 0-999/2731`. The function has no ORDER BY, so *which* two +-- thirds were dropped was down to the join order -- campaigns silently lost +-- recent days off their sparkline, and five lost every row and rendered +-- "no traffic yet" beside a row reading "Impressions: 24". +-- +-- One jsonb document is one row, so the cap cannot apply however many campaigns +-- the account grows to. +-- +-- The set-returning version has to go before the jsonb one can be created: +-- `create or replace` cannot change a function's return type, and leaving the +-- old signature in place would also let it win overload resolution for an +-- integer argument. +drop function if exists public.ad_campaign_daily_series(integer); + +create function public.ad_campaign_daily_series( + days integer default 30 +) returns jsonb +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + n int := greatest(coalesce(days, 30), 1); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + from_day date; +begin + if uid is null then + return '[]'::jsonb; + end if; + from_day := today - (n - 1); + + return coalesce(( + with owned as ( + select id from public.ad_campaigns where owner_id = uid + ), + ev as ( + select r.campaign_id, r.day, + (r.paid_impressions + r.free_impressions) as impressions, + r.valid_clicks as clicks, r.spent_cents as spent + from public.ad_stats_campaign_daily r + where r.campaign_id in (select id from owned) + and r.day >= from_day and r.day < today + union all + select i.campaign_id, today, 1::bigint, 0::bigint, 0::bigint + from public.ad_impressions i + where i.campaign_id in (select id from owned) + and not i.duplicate and i.ts >= today_start + union all + select cl.campaign_id, today, 0::bigint, + case when cl.valid then 1 else 0 end::bigint, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end::bigint + from public.ad_clicks cl + where cl.campaign_id in (select id from owned) + and cl.ts >= today_start + ) + -- The group-by has to finish before jsonb_agg runs, or aggregating over + -- grouped rows yields one single-element array per group instead of one + -- array of every day. + select jsonb_agg(row_to_json(g)) + from ( + select campaign_id, + day, + sum(impressions)::bigint as impressions, + sum(clicks)::bigint as clicks, + sum(spent)::bigint as spent_cents + from ev + group by campaign_id, day + -- A rollup row exists for any activity in the cell, including a day whose + -- only events were free or invalid clicks. The set-returning version this + -- replaces joined impressions to VALID clicks, so it never emitted such a + -- cell, and emitting it now would add all-zero points the caller already + -- zero-fills for itself. Keep the output identical. + having sum(impressions) > 0 or sum(clicks) > 0 + ) g + ), '[]'::jsonb); +end; +$$; + +-- --------------------------------------------------------------------------- +-- Publisher side +-- --------------------------------------------------------------------------- +create or replace function public.ad_slot_totals( + p_since timestamptz default null +) returns table( + slot_id uuid, impressions bigint, free_impressions bigint, clicks bigint, + free_clicks bigint, invalid_clicks bigint, earned_cents bigint +) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + first_full_day date; + first_full_start timestamptz; +begin + if uid is null then + return; + end if; + + first_full_day := case + when p_since is null then null + when p_since = ((p_since at time zone 'UTC')::date::timestamp at time zone 'UTC') + then (p_since at time zone 'UTC')::date + else (p_since at time zone 'UTC')::date + 1 + end; + first_full_start := (first_full_day::timestamp at time zone 'UTC'); + + return query + with owned as ( + select id from public.ad_slots where owner_id = uid + ), + ev as ( + select r.slot_id, r.paid_impressions as paid, r.free_impressions as free, + r.valid_clicks as clk, r.free_clicks as fclk, + r.invalid_clicks as iclk, r.earned_cents as earned + from public.ad_stats_slot_daily r + where r.slot_id in (select id from owned) + and r.day < today + and (first_full_day is null or r.day >= first_full_day) + union all + select i.slot_id, + case when i.tier = 'free' then 0 else 1 end, + case when i.tier = 'free' then 1 else 0 end, + 0, 0, 0, 0 + from public.ad_impressions i + where i.slot_id in (select id from owned) + and not i.duplicate + and (p_since is null or i.ts >= p_since) + and (i.ts >= today_start + or (first_full_start is not null and i.ts < first_full_start)) + union all + select cl.slot_id, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when not cl.valid and cl.tier <> 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.publisher_earn_cents, 0) else 0 end + from public.ad_clicks cl + where cl.slot_id in (select id from owned) + and (p_since is null or cl.ts >= p_since) + and (cl.ts >= today_start + or (first_full_start is not null and cl.ts < first_full_start)) + ) + select ev.slot_id, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(iclk)::bigint, sum(earned)::bigint + from ev group by ev.slot_id; +end; +$$; + +create or replace function public.ad_slot_daily_series( + days integer default 30 +) returns table(slot_id uuid, day date, clicks bigint, earned_cents bigint) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + n int := greatest(coalesce(days, 30), 1); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + from_day date; +begin + if uid is null then + return; + end if; + from_day := today - (n - 1); + + return query + with owned as ( + select id from public.ad_slots where owner_id = uid + ), + ev as ( + select r.slot_id, r.day, r.valid_clicks as clk, r.earned_cents as earned + from public.ad_stats_slot_daily r + where r.slot_id in (select id from owned) + and r.day >= from_day and r.day < today + union all + select cl.slot_id, today, 1::bigint, + coalesce(cl.publisher_earn_cents, 0)::bigint + from public.ad_clicks cl + where cl.slot_id in (select id from owned) + and cl.valid and cl.ts >= today_start + ) + select ev.slot_id, ev.day, sum(clk)::bigint, sum(earned)::bigint + from ev group by ev.slot_id, ev.day + -- Same reason as ad_campaign_daily_series above: this function has only ever + -- reported days on which a valid click landed, and ad_stats_slot_daily also + -- carries days that saw impressions alone. Without this the 30-day call goes + -- from 38 rows to 1,481, nearly all of them zero. + having sum(clk) > 0; +end; +$$; + diff --git a/tests/ads-campaign-daily-series.test.ts b/tests/ads-campaign-daily-series.test.ts new file mode 100644 index 0000000..5a53ac8 --- /dev/null +++ b/tests/ads-campaign-daily-series.test.ts @@ -0,0 +1,108 @@ +import { describe, expect, it } from "vitest"; +import { getCampaignDailySeries } from "@/lib/ads/series"; + +// ad_campaign_daily_series used to return one row per campaign-day. At 139 +// campaigns over 30 days that is 2,731 rows, and PostgREST caps a response at +// 1,000: the call came back `206 Partial Content, content-range 0-999/2731`. +// The function has no ORDER BY, so which two thirds survived was down to the +// join order — campaigns lost days off the end of their sparkline and five lost +// every row, rendering "no traffic yet" next to a row reading "Impressions: 24". +// +// It returns a single jsonb array now, which is one row whatever the campaign +// count. These tests pin the parsing of that shape, because the wire format is +// the only thing that changed and every ad surface reads this loader. + +const CAMPAIGNS = ["c1", "c2"]; + +/** Minimal stand-in: getCampaignDailySeries only calls .rpc. */ +const clientReturning = (data: unknown): any => ({ + rpc: async () => ({ data, error: null }), +}); +const failingClient = (): any => ({ + rpc: async () => ({ data: null, error: { message: "boom" } }), +}); + +const today = () => new Date().toISOString().slice(0, 10); + +describe("getCampaignDailySeries over the jsonb payload", () => { + it("reads a jsonb array straight through, no envelope to unwrap", async () => { + const { data, failed } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: today(), impressions: 24, clicks: 2, spent_cents: 0 }, + ]), + CAMPAIGNS, + 7, + ); + + expect(failed).toBe(false); + const point = data.get("c1")!.find((p) => p.date === today())!; + expect(point.impressions).toBe(24); + expect(point.clicks).toBe(2); + }); + + it("keeps every campaign's axis zero-filled, including ones with no rows", async () => { + const { data } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: today(), impressions: 5, clicks: 0, spent_cents: 0 }, + ]), + CAMPAIGNS, + 7, + ); + + // c2 sent nothing back. It still gets a full axis of zeros rather than + // being absent, so the row renders a flat sparkline instead of throwing. + expect(data.get("c2")).toHaveLength(7); + expect(data.get("c2")!.every((p) => p.impressions === 0)).toBe(true); + expect(data.get("c1")).toHaveLength(7); + }); + + it("takes bigint counts that arrived as JSON strings", async () => { + // jsonb renders bigint as a number, but a driver or a proxy that widens it + // to a string must not silently zero the sparkline. + const { data } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: today(), impressions: "48", clicks: "3", spent_cents: "150" }, + ]), + CAMPAIGNS, + 7, + ); + + const point = data.get("c1")!.find((p) => p.date === today())!; + expect(point.impressions).toBe(48); + expect(point.clicks).toBe(3); + expect(point.spentCents).toBe(150); + }); + + it("ignores days outside the requested window instead of misfiling them", async () => { + const { data } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: "2020-01-01", impressions: 999, clicks: 9, spent_cents: 0 }, + ]), + CAMPAIGNS, + 7, + ); + + expect(data.get("c1")!.every((p) => p.impressions === 0)).toBe(true); + }); + + it("treats an empty array as a quiet window, not a failure", async () => { + const { data, failed } = await getCampaignDailySeries(clientReturning([]), CAMPAIGNS, 7); + expect(failed).toBe(false); + expect(data.get("c1")!.every((p) => p.impressions === 0)).toBe(true); + }); + + it("flags a failed call, so a zero-filled axis is never read as real", async () => { + const { data, failed } = await getCampaignDailySeries(failingClient(), CAMPAIGNS, 7); + // The zero-fill stays — one dead panel should not take the page down — but + // `failed` is what lets MiniTrend say "unavailable" instead of asserting + // "no traffic yet" about a campaign it never managed to read. + expect(failed).toBe(true); + expect(data.get("c1")).toHaveLength(7); + }); + + it("does not call out at all when there are no campaigns", async () => { + const { data, failed } = await getCampaignDailySeries(failingClient(), [], 7); + expect(failed).toBe(false); + expect(data.size).toBe(0); + }); +});