From e23b55dadd16c6a9e3e1bf3fdd323d17ad9bc781 Mon Sep 17 00:00:00 2001 From: Anthony Ettinger Date: Wed, 2 Sep 2026 13:38:14 +0000 Subject: [PATCH] Stop the ad dashboard losing its graphs to a timed-out query /dashboard/ads renders no graphs, intermittently: the chart is replaced by "No delivery in this range." and the campaign sparklines by "no traffic yet", over a network that delivered 177,844 impressions in the window. Delivery was healthy the whole time. Three separate faults. 1. The reporting RPCs sat at the statement_timeout wall. Measured over 24h: ad_account_series avg 1,453ms / max 8,039ms, and ad_campaign_daily_series avg 1,719ms / max 8,027ms, against the 8s timeout on `authenticated` -- 14 of ~150 calls returned HTTP 500, so roughly one load in ten came back empty and a reload "fixed" it. Not the RLS nested loop from #226; that fix held and the plan is a clean hash join. Every load simply re-aggregated the whole archive: Seq Scan on ad_impressions (actual rows=177858) Filter: ((NOT duplicate) AND (ts >= now() - '30 days')) Rows Removed by Filter: 198394 Buffers: shared hit=9767 -- the entire 78MB heap, per load and the page fires three such RPCs per render. ad_impressions is 376k rows growing ~90k/day, so cost tracked the archive rather than the window asked for. No index fixes that: the 30-day window selects 47% of the table. A covering partial index measured 157ms against 194ms for the seq scan -- a 20% gain on a query needing to be 10x faster. So pre-aggregate. Three rollups, each at the grain its surface plots (owner/hour, campaign/day, slot/day), refreshed by pg_cron every ten minutes. Closed periods come from the rollup, the current hour and current day always from raw, so a stale or un-run refresh can never show a wrong live edge and the raw slice is never wider than a day. A 30-day account query now reads ~720 rows where it read 246,506. Verified by swapping the functions inside a REPEATABLE READ transaction and diffing old against new across all 8 ranges and 3 day windows: byte-identical output, 3,312ms -> 380ms for the battery. 2. ad_campaign_daily_series had outgrown PostgREST's 1000-row cap. 139 campaigns x 30 days is 2,731 rows and the call was returning `206 Partial Content, content-range 0-999/2731`. With no ORDER BY, which two thirds got dropped was down to join order: campaigns lost days off their sparkline and five lost every row, rendering "no traffic yet" beside a row reading "Impressions: 24". It returns one jsonb array now -- one row, whatever the campaign count. Live, that took the page from 9 "no traffic yet" rows to the 4 that really have no delivery. 3. AccountTrend and MiniTrend still asserted emptiness on failure. Both receive a zero-filled series when the query dies, so "No delivery in this range." was a claim about the network made from a query that never returned -- and it sat directly under the banner #226 added saying the figures could not be loaded. They now take `failed` and say so. Tracked per loader, since the chart and the sparklines come from different RPCs and either can fail alone. Both migrations are applied to prod. The rollups reconcile exactly against raw events across all 57 closed days on both the advertiser and publisher sides, including after the first scheduled mid-day refresh. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01Y7rVKbcZdT7tABboosr1Re --- app/(app)/dashboard/ads/page.tsx | 12 +- components/ads/account-trend.tsx | 24 +- components/ads/mini-trend.tsx | 17 + lib/ads/series.ts | 23 +- .../20260902140000_ad_stats_rollups.sql | 256 +++++++++++ ...902140100_ad_reporting_rpcs_on_rollups.sql | 404 ++++++++++++++++++ tests/ads-campaign-daily-series.test.ts | 108 +++++ 7 files changed, 835 insertions(+), 9 deletions(-) create mode 100644 supabase/migrations/20260902140000_ad_stats_rollups.sql create mode 100644 supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql create mode 100644 tests/ads-campaign-daily-series.test.ts diff --git a/app/(app)/dashboard/ads/page.tsx b/app/(app)/dashboard/ads/page.tsx index b8864d8c..5d240866 100644 --- a/app/(app)/dashboard/ads/page.tsx +++ b/app/(app)/dashboard/ads/page.tsx @@ -70,7 +70,13 @@ export default async function AdsPage({ // A failed stats query zero-fills, so without this the page would report a // confident 0 for a range it simply could not read. See Loaded<> in // lib/ads/series.ts. + // + // Tracked per loader, not as one flag: the chart and the per-campaign + // sparklines come from different RPCs, and either can fail on its own. Only + // the panel whose query actually failed should say so. let statsFailed = false; + let seriesFailed = false; + let dailyFailed = false; if (user) { const [{ data }, { data: profile }, accountSeries, campaignTotals] = await Promise.all([ supabase @@ -97,6 +103,8 @@ export default async function AdsPage({ 30, ); seriesById = daily.data; + seriesFailed = accountSeries.failed; + dailyFailed = daily.failed; statsFailed = accountSeries.failed || campaignTotals.failed || daily.failed; } @@ -185,7 +193,7 @@ export default async function AdsPage({ )}
- +
)} @@ -222,7 +230,7 @@ export default async function AdsPage({
- + {display.label} diff --git a/components/ads/account-trend.tsx b/components/ads/account-trend.tsx index 74f81b5a..d33a3434 100644 --- a/components/ads/account-trend.tsx +++ b/components/ads/account-trend.tsx @@ -26,9 +26,31 @@ import type { AccountPoint } from "@/lib/ads/series"; * and the split is the point — free backfill is delivery that earns nobody * anything, and it should be visible as a share of the whole. */ -export function AccountTrend({ data, range }: { data: AccountPoint[]; range: RangeDef }) { +export function AccountTrend({ + data, + range, + failed = false, +}: { + data: AccountPoint[]; + range: RangeDef; + failed?: boolean; +}) { const total = data.reduce((n, p) => n + p.impressions + p.freeImpressions + p.clicks, 0); + // A failed series arrives zero-filled and so is indistinguishable from a quiet + // range by its values alone. Saying "No delivery" here would be a statement of + // fact about the network, made from a query that never returned -- the same + // mistake #226 fixed for the tiles, still being made by the chart underneath + // them, and directly contradicting the banner the page renders above. + if (failed) { + return ( +
+ Couldn't load this chart. It isn't empty — the query didn't + return. Reloading often fixes it. +
+ ); + } + if (total === 0) { return (
diff --git a/components/ads/mini-trend.tsx b/components/ads/mini-trend.tsx index a4597c93..daca62cd 100644 --- a/components/ads/mini-trend.tsx +++ b/components/ads/mini-trend.tsx @@ -9,11 +9,28 @@ export function MiniTrend({ data, width = 132, height = 34, + failed = false, }: { data: CampaignDailyPoint[]; width?: number; height?: number; + failed?: boolean; }) { + // Zero-filled on failure, so "no traffic yet" would assert something about the + // campaign that the query never established. See AccountTrend for the same + // distinction on the chart above. + if (failed) { + return ( +
+ unavailable +
+ ); + } + const total = data.reduce((n, p) => n + p.impressions, 0); // Count clicks as traffic too (a campaign can have clicks logged without a // matching impression row), matching CampaignTrend's empty check. `total` diff --git a/lib/ads/series.ts b/lib/ads/series.ts index 30a460ef..5a54c219 100644 --- a/lib/ads/series.ts +++ b/lib/ads/series.ts @@ -352,12 +352,23 @@ type DailySeriesRow = { * Aggregated server-side by the ad_campaign_daily_series RPC (security definer, * scoped by its own `owner_id = auth.uid()` filter rather than by RLS — see * 20260901153000_ad_reporting_rpcs_security_definer.sql for why). We must NOT - * fetch and bucket raw ad_impressions rows here: PostgREST caps a response at 1000 rows, so once - * total impressions in the window exceed 1000 a few high-volume campaigns eat - * the whole page and every other campaign gets zero rows back — rendering - * "no traffic yet" despite having recent impressions. The RPC returns at most - * (campaigns * days) rows, so it never hits the cap. - * See migration 20260717032002_ad_campaign_daily_series_rpc.sql. + * fetch and bucket raw ad_impressions rows here: PostgREST caps a response at + * 1000 rows, so once total impressions in the window exceed 1000 a few + * high-volume campaigns eat the whole page and every other campaign gets zero + * rows back — rendering "no traffic yet" despite having recent impressions. + * + * The RPC used to return a row per campaign-day, which was said here to be + * safely under that cap and stopped being true as the account grew: 139 + * campaigns over 30 days is 2,731 rows, and the call was coming back + * `206 Partial Content, content-range 0-999/2731`. With no ORDER BY in the + * function, which two thirds got dropped was down to the join order — campaigns + * silently lost days off the end of their sparkline and five lost every row, + * rendering "no traffic yet" beside a row reading "Impressions: 24". + * + * So it now returns one jsonb array instead. One document is one row, and the + * cap cannot apply however many campaigns the account grows to. `data` is that + * array already parsed. See 20260902140100_ad_reporting_rpcs_on_rollups.sql, + * which is also where it stopped scanning raw events. */ export async function getCampaignDailySeries( supabase: SupabaseClient, diff --git a/supabase/migrations/20260902140000_ad_stats_rollups.sql b/supabase/migrations/20260902140000_ad_stats_rollups.sql new file mode 100644 index 00000000..1a9ce279 --- /dev/null +++ b/supabase/migrations/20260902140000_ad_stats_rollups.sql @@ -0,0 +1,256 @@ +-- Pre-aggregate ad delivery so the dashboards stop scanning raw events. +-- +-- /dashboard/ads renders no graphs, intermittently. The chart falls back to +-- "No delivery in this range." and every sparkline to "no traffic yet" because +-- ad_account_series and ad_campaign_daily_series are being cancelled by the 8s +-- statement_timeout on `authenticated`. Measured over 24h on 2026-09-02: +-- ad_account_series avg 1,453ms / max 8,039ms, ad_campaign_daily_series avg +-- 1,719ms / max 8,027ms, and 14 of ~150 reporting calls returned HTTP 500. +-- +-- This is NOT the RLS nested-loop from 20260901153000 -- that fix held, the +-- plan is a clean hash join. It is simply that every dashboard load aggregates +-- the whole event history from scratch: +-- +-- Seq Scan on ad_impressions (actual rows=177858) +-- Filter: ((NOT duplicate) AND (ts >= now() - '30 days')) +-- Rows Removed by Filter: 198394 +-- Buffers: shared hit=9767 -- the entire 78MB heap, every load +-- +-- and the page issues three such RPCs per render. ad_impressions is 376k rows +-- growing ~90k/day, so the cost rises with the archive rather than with the +-- window being asked for, and no index fixes that: the 30-day window selects +-- 47% of the table, far past the point an index scan can win. A covering +-- partial index (added below, and it is still worth having -- see the RPCs) was +-- measured at 157ms against 194ms for the seq scan, a 20% gain on a query that +-- needs to be 10x faster. +-- +-- So aggregate once, on a schedule, and let a page load read buckets instead of +-- events. The grain is chosen by what each surface actually plots: +-- +-- ad_stats_owner_hourly (owner_id, hour) ~24 rows/day +-- ad_stats_campaign_daily (campaign_id, day) ~135 rows/day +-- ad_stats_slot_daily (slot_id, day) ~27 rows/day +-- +-- A 30-day account query now reads ~720 rows where it read 246,506. +-- +-- Per-campaign is deliberately DAILY and not hourly: (campaign_id, hour) is +-- 2,997 distinct cells in a single day against 139 campaigns, which over the +-- archive is ~170k rows -- barely smaller than the raw table it replaces, so it +-- would buy nothing. Account-wide has no campaign dimension at all, which is +-- what makes hourly affordable there, and hourly is required there because the +-- 1W range plots 4-hour buckets. +-- +-- FRESHNESS. Rollup rows are only ever read for periods that have closed; the +-- current hour and the current UTC day are always read from raw. So a stale or +-- even a completely un-run refresh can never show wrong numbers for the live +-- edge, and the raw slices it reads are at most one hour / one day wide, which +-- is where the covering index earns its place. + +-- The reporting predicate is always (not duplicate) over a ts window, and the +-- columns wanted are few enough to ride along in the index. This serves the +-- narrow live-edge reads the RPCs below fall back to, and the refresh itself. +create index if not exists ad_impressions_reporting_idx + on public.ad_impressions (ts) include (campaign_id, slot_id, tier) + where not duplicate; + +create index if not exists ad_clicks_reporting_idx + on public.ad_clicks (ts) include (campaign_id, slot_id, tier, valid, charged_cents, publisher_earn_cents); + +-- --------------------------------------------------------------------------- +-- Rollup tables +-- --------------------------------------------------------------------------- +-- Column names mirror the RPC output they feed, including the split that has +-- caught this codebase out three times: `paid_*` and `free_*` are halves, never +-- totals. Anything reading one without the other is reading a network where +-- every fill is free backfill as though it were dead. + +create table if not exists public.ad_stats_owner_hourly ( + owner_id uuid not null, + hour timestamptz not null, + paid_impressions bigint not null default 0, + free_impressions bigint not null default 0, + valid_clicks bigint not null default 0, + free_clicks bigint not null default 0, + spent_cents bigint not null default 0, + primary key (owner_id, hour) +); + +create table if not exists public.ad_stats_campaign_daily ( + campaign_id uuid not null, + day date not null, + paid_impressions bigint not null default 0, + free_impressions bigint not null default 0, + valid_clicks bigint not null default 0, + free_clicks bigint not null default 0, + spent_cents bigint not null default 0, + primary key (campaign_id, day) +); + +create table if not exists public.ad_stats_slot_daily ( + slot_id uuid not null, + day date not null, + paid_impressions bigint not null default 0, + free_impressions bigint not null default 0, + valid_clicks bigint not null default 0, + free_clicks bigint not null default 0, + invalid_clicks bigint not null default 0, + earned_cents bigint not null default 0, + primary key (slot_id, day) +); + +-- Read exclusively through the security-definer RPCs below, which do their own +-- ownership filtering. RLS on with no policy is the intended posture: it denies +-- every direct PostgREST read while the definer functions (owned by the +-- migration role) still see the rows. +alter table public.ad_stats_owner_hourly enable row level security; +alter table public.ad_stats_campaign_daily enable row level security; +alter table public.ad_stats_slot_daily enable row level security; + +revoke all on public.ad_stats_owner_hourly from anon, authenticated; +revoke all on public.ad_stats_campaign_daily from anon, authenticated; +revoke all on public.ad_stats_slot_daily from anon, authenticated; + +-- --------------------------------------------------------------------------- +-- Refresh +-- --------------------------------------------------------------------------- + +/** + * Recompute every rollup for events at or after p_from and upsert the result. + * + * Idempotent, so re-running over a period that is already rolled up is a no-op + * in effect. Deliberately a full recompute of the touched periods rather than a + * delta: ad_clicks rows are updated after insert (a click is charged, or later + * invalidated), so counting only new rows would drift. Recomputing a bounded + * tail cannot. + * + * Periods with no events are simply absent; the RPCs zero-fill their own axis, + * as they always have. + */ +create or replace function public.ad_stats_rollup_refresh( + p_from timestamptz default (now() - interval '2 days') +) returns void +language plpgsql +security definer +set search_path to 'public' +as $fn$ +declare + -- Snap the window down to whole periods before recomputing. p_from is a plain + -- "two days ago" from the scheduler, which lands mid-day: filtering raw events + -- on it directly would recompute 08-31 from 13:40 onwards and then upsert that + -- partial count over the complete row already stored for that day. Every + -- period this function touches has to be recomputed in full or not at all. + v_from_hour timestamptz := date_trunc('hour', p_from); + v_from_day timestamptz := date_trunc('day', p_from at time zone 'UTC') at time zone 'UTC'; +begin + -- Account-wide hourly, advertiser side: an impression belongs to the owner of + -- the campaign that was shown. + insert into public.ad_stats_owner_hourly as t + (owner_id, hour, paid_impressions, free_impressions, valid_clicks, free_clicks, spent_cents) + select c.owner_id, + date_trunc('hour', e.ts), + sum(e.paid)::bigint, sum(e.free)::bigint, sum(e.clk)::bigint, + sum(e.fclk)::bigint, sum(e.spent)::bigint + from ( + select i.campaign_id, i.ts, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as spent + from public.ad_impressions i + where not i.duplicate and i.ts >= v_from_hour + union all + select cl.campaign_id, cl.ts, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + where cl.ts >= v_from_hour + ) e + join public.ad_campaigns c on c.id = e.campaign_id + group by 1, 2 + on conflict (owner_id, hour) do update set + paid_impressions = excluded.paid_impressions, + free_impressions = excluded.free_impressions, + valid_clicks = excluded.valid_clicks, + free_clicks = excluded.free_clicks, + spent_cents = excluded.spent_cents; + + insert into public.ad_stats_campaign_daily as t + (campaign_id, day, paid_impressions, free_impressions, valid_clicks, free_clicks, spent_cents) + select e.campaign_id, + (e.ts at time zone 'UTC')::date, + sum(e.paid)::bigint, sum(e.free)::bigint, sum(e.clk)::bigint, + sum(e.fclk)::bigint, sum(e.spent)::bigint + from ( + select i.campaign_id, i.ts, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as spent + from public.ad_impressions i + where not i.duplicate and i.ts >= v_from_day + union all + select cl.campaign_id, cl.ts, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + where cl.ts >= v_from_day + ) e + where e.campaign_id is not null + group by 1, 2 + on conflict (campaign_id, day) do update set + paid_impressions = excluded.paid_impressions, + free_impressions = excluded.free_impressions, + valid_clicks = excluded.valid_clicks, + free_clicks = excluded.free_clicks, + spent_cents = excluded.spent_cents; + + -- Publisher side. invalid_clicks is its own bucket on purpose: a refused + -- click is not delivery, and folding it into free_clicks would put a + -- double-digit CTR on the earnings page. + insert into public.ad_stats_slot_daily as t + (slot_id, day, paid_impressions, free_impressions, valid_clicks, free_clicks, invalid_clicks, earned_cents) + select e.slot_id, + (e.ts at time zone 'UTC')::date, + sum(e.paid)::bigint, sum(e.free)::bigint, sum(e.clk)::bigint, + sum(e.fclk)::bigint, sum(e.iclk)::bigint, sum(e.earned)::bigint + from ( + select i.slot_id, i.ts, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as iclk, 0 as earned + from public.ad_impressions i + where not i.duplicate and i.ts >= v_from_day + union all + select cl.slot_id, cl.ts, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when not cl.valid and cl.tier <> 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.publisher_earn_cents, 0) else 0 end + from public.ad_clicks cl + where cl.ts >= v_from_day + ) e + where e.slot_id is not null + group by 1, 2 + on conflict (slot_id, day) do update set + paid_impressions = excluded.paid_impressions, + free_impressions = excluded.free_impressions, + valid_clicks = excluded.valid_clicks, + free_clicks = excluded.free_clicks, + invalid_clicks = excluded.invalid_clicks, + earned_cents = excluded.earned_cents; +end; +$fn$; + +revoke all on function public.ad_stats_rollup_refresh(timestamptz) from anon, authenticated; + +-- Backfill the whole archive once. p_from covers every event ever recorded. +select public.ad_stats_rollup_refresh('epoch'::timestamptz); + +-- Close out finished hours/days. Ten minutes is well inside the one-hour +-- staleness the read path tolerates, and each run only touches two days. +select cron.schedule( + 'ad-stats-rollup', + '*/10 * * * *', + $cron$select public.ad_stats_rollup_refresh(now() - interval '2 days')$cron$ +); + diff --git a/supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql b/supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql new file mode 100644 index 00000000..7b9b35ad --- /dev/null +++ b/supabase/migrations/20260902140100_ad_reporting_rpcs_on_rollups.sql @@ -0,0 +1,404 @@ +-- Point the five ad reporting RPCs at the rollups from 20260902140000. +-- +-- Every function keeps its name, arguments and output columns, so nothing that +-- reads them has to change except ad_campaign_daily_series (see below). What +-- changes is where the numbers come from: +-- +-- closed periods -> ad_stats_* rollup rows +-- the live edge -> raw ad_impressions / ad_clicks +-- +-- "Closed" means strictly before the current hour (account series) or the +-- current UTC day (everything else). The live edge is therefore at most one +-- hour or one day wide no matter how long the requested window is, which is +-- what makes the cost proportional to the number of buckets plotted rather +-- than to the size of the archive. +-- +-- The split is exact, not approximate. When p_since lands mid-period the +-- leading partial period is read from raw too, so a window starting at +-- 13:24 still reports 13:24 onwards and not 14:00 onwards. That is the reason +-- for first_full_hour / first_full_day below rather than a plain date_trunc. +-- +-- Semantics are unchanged and deliberately restated here because they have +-- been misread three times: `impressions` and `clicks` are the PAID and VALID +-- halves, `free_impressions` / `free_clicks` are the others, and a caller that +-- reads one without the other sees zero on a network where every fill is free +-- backfill. + +-- --------------------------------------------------------------------------- +-- Account-wide series +-- --------------------------------------------------------------------------- +create or replace function public.ad_account_series( + p_since timestamptz default null, + p_bucket_seconds integer default 86400 +) returns table( + bucket timestamptz, impressions bigint, free_impressions bigint, + clicks bigint, free_clicks bigint, spent_cents bigint +) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + secs int := greatest(coalesce(p_bucket_seconds, 86400), 60); + step interval := make_interval(secs => secs); + uid uuid := auth.uid(); + this_hour timestamptz := date_trunc('hour', now()); + first_full_hour timestamptz; +begin + if uid is null then + return; + end if; + + -- Buckets finer than the rollup grain (the 1H/4H/1D tabs, at 60/300/1800s) + -- have to come from raw events. They only ever span the last 24 hours, so + -- the ts index makes that a narrow read rather than the full-archive scan + -- this function used to do for every range alike. + if secs < 3600 then + return query + with ev as ( + select date_bin(step, i.ts, timestamptz 'epoch') as b, + case when i.tier = 'free' then 0 else 1 end as paid, + case when i.tier = 'free' then 1 else 0 end as free, + 0 as clk, 0 as fclk, 0 as spent + from public.ad_impressions i + join public.ad_campaigns c on c.id = i.campaign_id and c.owner_id = uid + where not i.duplicate and (p_since is null or i.ts >= p_since) + union all + select date_bin(step, cl.ts, timestamptz 'epoch'), 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + join public.ad_campaigns c on c.id = cl.campaign_id and c.owner_id = uid + where (p_since is null or cl.ts >= p_since) + ) + select b, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(spent)::bigint + from ev group by b order by b; + return; + end if; + + first_full_hour := case + when p_since is null then null + when p_since = date_trunc('hour', p_since) then p_since + else date_trunc('hour', p_since) + interval '1 hour' + end; + + return query + with ev as ( + select date_bin(step, r.hour, timestamptz 'epoch') as b, + r.paid_impressions as paid, r.free_impressions as free, + r.valid_clicks as clk, r.free_clicks as fclk, r.spent_cents as spent + from public.ad_stats_owner_hourly r + where r.owner_id = uid + and r.hour < this_hour + and (first_full_hour is null or r.hour >= first_full_hour) + union all + select date_bin(step, i.ts, timestamptz 'epoch'), + case when i.tier = 'free' then 0 else 1 end, + case when i.tier = 'free' then 1 else 0 end, + 0, 0, 0 + from public.ad_impressions i + join public.ad_campaigns c on c.id = i.campaign_id and c.owner_id = uid + where not i.duplicate + and (p_since is null or i.ts >= p_since) + and (i.ts >= this_hour + or (first_full_hour is not null and i.ts < first_full_hour)) + union all + select date_bin(step, cl.ts, timestamptz 'epoch'), 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + join public.ad_campaigns c on c.id = cl.campaign_id and c.owner_id = uid + where (p_since is null or cl.ts >= p_since) + and (cl.ts >= this_hour + or (first_full_hour is not null and cl.ts < first_full_hour)) + ) + select b, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(spent)::bigint + from ev group by b order by b; +end; +$$; + +-- --------------------------------------------------------------------------- +-- Per-campaign totals for a window +-- --------------------------------------------------------------------------- +create or replace function public.ad_campaign_totals( + p_since timestamptz default null +) returns table( + campaign_id uuid, impressions bigint, free_impressions bigint, + clicks bigint, free_clicks bigint, spent_cents bigint +) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + first_full_day date; + first_full_start timestamptz; +begin + if uid is null then + return; + end if; + + first_full_day := case + when p_since is null then null + when p_since = ((p_since at time zone 'UTC')::date::timestamp at time zone 'UTC') + then (p_since at time zone 'UTC')::date + else (p_since at time zone 'UTC')::date + 1 + end; + first_full_start := (first_full_day::timestamp at time zone 'UTC'); + + return query + with owned as ( + select id from public.ad_campaigns where owner_id = uid + ), + ev as ( + select r.campaign_id, + r.paid_impressions as paid, r.free_impressions as free, + r.valid_clicks as clk, r.free_clicks as fclk, r.spent_cents as spent + from public.ad_stats_campaign_daily r + where r.campaign_id in (select id from owned) + and r.day < today + and (first_full_day is null or r.day >= first_full_day) + union all + select i.campaign_id, + case when i.tier = 'free' then 0 else 1 end, + case when i.tier = 'free' then 1 else 0 end, + 0, 0, 0 + from public.ad_impressions i + where i.campaign_id in (select id from owned) + and not i.duplicate + and (p_since is null or i.ts >= p_since) + and (i.ts >= today_start + or (first_full_start is not null and i.ts < first_full_start)) + union all + select cl.campaign_id, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end + from public.ad_clicks cl + where cl.campaign_id in (select id from owned) + and (p_since is null or cl.ts >= p_since) + and (cl.ts >= today_start + or (first_full_start is not null and cl.ts < first_full_start)) + ) + select ev.campaign_id, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(spent)::bigint + from ev group by ev.campaign_id; +end; +$$; + +-- --------------------------------------------------------------------------- +-- Per-campaign daily series +-- --------------------------------------------------------------------------- +-- Now returns a single jsonb array instead of a set of rows, because the set +-- had outgrown PostgREST's 1000-row response cap: 139 campaigns over 30 days is +-- 2,731 rows, and the request was coming back `206 Partial Content, +-- content-range 0-999/2731`. The function has no ORDER BY, so *which* two +-- thirds were dropped was down to the join order -- campaigns silently lost +-- recent days off their sparkline, and five lost every row and rendered +-- "no traffic yet" beside a row reading "Impressions: 24". +-- +-- One jsonb document is one row, so the cap cannot apply however many campaigns +-- the account grows to. +-- +-- The set-returning version has to go before the jsonb one can be created: +-- `create or replace` cannot change a function's return type, and leaving the +-- old signature in place would also let it win overload resolution for an +-- integer argument. +drop function if exists public.ad_campaign_daily_series(integer); + +create function public.ad_campaign_daily_series( + days integer default 30 +) returns jsonb +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + n int := greatest(coalesce(days, 30), 1); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + from_day date; +begin + if uid is null then + return '[]'::jsonb; + end if; + from_day := today - (n - 1); + + return coalesce(( + with owned as ( + select id from public.ad_campaigns where owner_id = uid + ), + ev as ( + select r.campaign_id, r.day, + (r.paid_impressions + r.free_impressions) as impressions, + r.valid_clicks as clicks, r.spent_cents as spent + from public.ad_stats_campaign_daily r + where r.campaign_id in (select id from owned) + and r.day >= from_day and r.day < today + union all + select i.campaign_id, today, 1::bigint, 0::bigint, 0::bigint + from public.ad_impressions i + where i.campaign_id in (select id from owned) + and not i.duplicate and i.ts >= today_start + union all + select cl.campaign_id, today, 0::bigint, + case when cl.valid then 1 else 0 end::bigint, + case when cl.valid then coalesce(cl.charged_cents, 0) else 0 end::bigint + from public.ad_clicks cl + where cl.campaign_id in (select id from owned) + and cl.ts >= today_start + ) + -- The group-by has to finish before jsonb_agg runs, or aggregating over + -- grouped rows yields one single-element array per group instead of one + -- array of every day. + select jsonb_agg(row_to_json(g)) + from ( + select campaign_id, + day, + sum(impressions)::bigint as impressions, + sum(clicks)::bigint as clicks, + sum(spent)::bigint as spent_cents + from ev + group by campaign_id, day + -- A rollup row exists for any activity in the cell, including a day whose + -- only events were free or invalid clicks. The set-returning version this + -- replaces joined impressions to VALID clicks, so it never emitted such a + -- cell, and emitting it now would add all-zero points the caller already + -- zero-fills for itself. Keep the output identical. + having sum(impressions) > 0 or sum(clicks) > 0 + ) g + ), '[]'::jsonb); +end; +$$; + +-- --------------------------------------------------------------------------- +-- Publisher side +-- --------------------------------------------------------------------------- +create or replace function public.ad_slot_totals( + p_since timestamptz default null +) returns table( + slot_id uuid, impressions bigint, free_impressions bigint, clicks bigint, + free_clicks bigint, invalid_clicks bigint, earned_cents bigint +) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + first_full_day date; + first_full_start timestamptz; +begin + if uid is null then + return; + end if; + + first_full_day := case + when p_since is null then null + when p_since = ((p_since at time zone 'UTC')::date::timestamp at time zone 'UTC') + then (p_since at time zone 'UTC')::date + else (p_since at time zone 'UTC')::date + 1 + end; + first_full_start := (first_full_day::timestamp at time zone 'UTC'); + + return query + with owned as ( + select id from public.ad_slots where owner_id = uid + ), + ev as ( + select r.slot_id, r.paid_impressions as paid, r.free_impressions as free, + r.valid_clicks as clk, r.free_clicks as fclk, + r.invalid_clicks as iclk, r.earned_cents as earned + from public.ad_stats_slot_daily r + where r.slot_id in (select id from owned) + and r.day < today + and (first_full_day is null or r.day >= first_full_day) + union all + select i.slot_id, + case when i.tier = 'free' then 0 else 1 end, + case when i.tier = 'free' then 1 else 0 end, + 0, 0, 0, 0 + from public.ad_impressions i + where i.slot_id in (select id from owned) + and not i.duplicate + and (p_since is null or i.ts >= p_since) + and (i.ts >= today_start + or (first_full_start is not null and i.ts < first_full_start)) + union all + select cl.slot_id, 0, 0, + case when cl.valid then 1 else 0 end, + case when not cl.valid and cl.tier = 'free' then 1 else 0 end, + case when not cl.valid and cl.tier <> 'free' then 1 else 0 end, + case when cl.valid then coalesce(cl.publisher_earn_cents, 0) else 0 end + from public.ad_clicks cl + where cl.slot_id in (select id from owned) + and (p_since is null or cl.ts >= p_since) + and (cl.ts >= today_start + or (first_full_start is not null and cl.ts < first_full_start)) + ) + select ev.slot_id, sum(paid)::bigint, sum(free)::bigint, sum(clk)::bigint, + sum(fclk)::bigint, sum(iclk)::bigint, sum(earned)::bigint + from ev group by ev.slot_id; +end; +$$; + +create or replace function public.ad_slot_daily_series( + days integer default 30 +) returns table(slot_id uuid, day date, clicks bigint, earned_cents bigint) +language plpgsql +stable +security definer +set search_path to 'public' +as $$ +declare + uid uuid := auth.uid(); + n int := greatest(coalesce(days, 30), 1); + today date := (now() at time zone 'UTC')::date; + today_start timestamptz := (today::timestamp at time zone 'UTC'); + from_day date; +begin + if uid is null then + return; + end if; + from_day := today - (n - 1); + + return query + with owned as ( + select id from public.ad_slots where owner_id = uid + ), + ev as ( + select r.slot_id, r.day, r.valid_clicks as clk, r.earned_cents as earned + from public.ad_stats_slot_daily r + where r.slot_id in (select id from owned) + and r.day >= from_day and r.day < today + union all + select cl.slot_id, today, 1::bigint, + coalesce(cl.publisher_earn_cents, 0)::bigint + from public.ad_clicks cl + where cl.slot_id in (select id from owned) + and cl.valid and cl.ts >= today_start + ) + select ev.slot_id, ev.day, sum(clk)::bigint, sum(earned)::bigint + from ev group by ev.slot_id, ev.day + -- Same reason as ad_campaign_daily_series above: this function has only ever + -- reported days on which a valid click landed, and ad_stats_slot_daily also + -- carries days that saw impressions alone. Without this the 30-day call goes + -- from 38 rows to 1,481, nearly all of them zero. + having sum(clk) > 0; +end; +$$; + diff --git a/tests/ads-campaign-daily-series.test.ts b/tests/ads-campaign-daily-series.test.ts new file mode 100644 index 00000000..5a53ac8b --- /dev/null +++ b/tests/ads-campaign-daily-series.test.ts @@ -0,0 +1,108 @@ +import { describe, expect, it } from "vitest"; +import { getCampaignDailySeries } from "@/lib/ads/series"; + +// ad_campaign_daily_series used to return one row per campaign-day. At 139 +// campaigns over 30 days that is 2,731 rows, and PostgREST caps a response at +// 1,000: the call came back `206 Partial Content, content-range 0-999/2731`. +// The function has no ORDER BY, so which two thirds survived was down to the +// join order — campaigns lost days off the end of their sparkline and five lost +// every row, rendering "no traffic yet" next to a row reading "Impressions: 24". +// +// It returns a single jsonb array now, which is one row whatever the campaign +// count. These tests pin the parsing of that shape, because the wire format is +// the only thing that changed and every ad surface reads this loader. + +const CAMPAIGNS = ["c1", "c2"]; + +/** Minimal stand-in: getCampaignDailySeries only calls .rpc. */ +const clientReturning = (data: unknown): any => ({ + rpc: async () => ({ data, error: null }), +}); +const failingClient = (): any => ({ + rpc: async () => ({ data: null, error: { message: "boom" } }), +}); + +const today = () => new Date().toISOString().slice(0, 10); + +describe("getCampaignDailySeries over the jsonb payload", () => { + it("reads a jsonb array straight through, no envelope to unwrap", async () => { + const { data, failed } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: today(), impressions: 24, clicks: 2, spent_cents: 0 }, + ]), + CAMPAIGNS, + 7, + ); + + expect(failed).toBe(false); + const point = data.get("c1")!.find((p) => p.date === today())!; + expect(point.impressions).toBe(24); + expect(point.clicks).toBe(2); + }); + + it("keeps every campaign's axis zero-filled, including ones with no rows", async () => { + const { data } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: today(), impressions: 5, clicks: 0, spent_cents: 0 }, + ]), + CAMPAIGNS, + 7, + ); + + // c2 sent nothing back. It still gets a full axis of zeros rather than + // being absent, so the row renders a flat sparkline instead of throwing. + expect(data.get("c2")).toHaveLength(7); + expect(data.get("c2")!.every((p) => p.impressions === 0)).toBe(true); + expect(data.get("c1")).toHaveLength(7); + }); + + it("takes bigint counts that arrived as JSON strings", async () => { + // jsonb renders bigint as a number, but a driver or a proxy that widens it + // to a string must not silently zero the sparkline. + const { data } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: today(), impressions: "48", clicks: "3", spent_cents: "150" }, + ]), + CAMPAIGNS, + 7, + ); + + const point = data.get("c1")!.find((p) => p.date === today())!; + expect(point.impressions).toBe(48); + expect(point.clicks).toBe(3); + expect(point.spentCents).toBe(150); + }); + + it("ignores days outside the requested window instead of misfiling them", async () => { + const { data } = await getCampaignDailySeries( + clientReturning([ + { campaign_id: "c1", day: "2020-01-01", impressions: 999, clicks: 9, spent_cents: 0 }, + ]), + CAMPAIGNS, + 7, + ); + + expect(data.get("c1")!.every((p) => p.impressions === 0)).toBe(true); + }); + + it("treats an empty array as a quiet window, not a failure", async () => { + const { data, failed } = await getCampaignDailySeries(clientReturning([]), CAMPAIGNS, 7); + expect(failed).toBe(false); + expect(data.get("c1")!.every((p) => p.impressions === 0)).toBe(true); + }); + + it("flags a failed call, so a zero-filled axis is never read as real", async () => { + const { data, failed } = await getCampaignDailySeries(failingClient(), CAMPAIGNS, 7); + // The zero-fill stays — one dead panel should not take the page down — but + // `failed` is what lets MiniTrend say "unavailable" instead of asserting + // "no traffic yet" about a campaign it never managed to read. + expect(failed).toBe(true); + expect(data.get("c1")).toHaveLength(7); + }); + + it("does not call out at all when there are no campaigns", async () => { + const { data, failed } = await getCampaignDailySeries(failingClient(), [], 7); + expect(failed).toBe(false); + expect(data.size).toBe(0); + }); +});