diff --git a/app/(app)/projects/[id]/uptime/add-monitor-form.tsx b/app/(app)/projects/[id]/uptime/add-monitor-form.tsx new file mode 100644 index 00000000..3afb1c5a --- /dev/null +++ b/app/(app)/projects/[id]/uptime/add-monitor-form.tsx @@ -0,0 +1,163 @@ +"use client"; + +import { useState, useTransition } from "react"; +import { useRouter } from "next/navigation"; +import { createMonitor, type CreateMonitorInput } from "@/app/actions/monitors"; + +type MonitorType = "http" | "keyword" | "ssl" | "tcp"; + +const TYPE_HINT: Record = { + http: "URL to request — up when it returns 2xx/3xx.", + keyword: "URL + a keyword that must appear (or not) in the response body.", + ssl: "Host — alerts when the TLS cert nears expiry.", + tcp: "host:port — up when the port accepts a connection.", +}; + +export function AddMonitorForm({ projectId }: { projectId: string }) { + const router = useRouter(); + const [open, setOpen] = useState(false); + const [type, setType] = useState("http"); + const [pending, start] = useTransition(); + const [error, setError] = useState(null); + + function submit(formData: FormData) { + setError(null); + const input: CreateMonitorInput = { + projectId, + type, + name: String(formData.get("name") ?? ""), + target: String(formData.get("target") ?? ""), + intervalS: Number(formData.get("intervalS") ?? 60), + alertEmail: String(formData.get("alertEmail") ?? ""), + keyword: String(formData.get("keyword") ?? ""), + match: (String(formData.get("match") ?? "present") as "present" | "absent"), + }; + start(async () => { + const res = await createMonitor(input); + if (!res.ok) { + setError(res.error ?? "Failed."); + return; + } + setOpen(false); + router.refresh(); + }); + } + + if (!open) { + return ( + + ); + } + + return ( +
+
+ + +
+ + + + {type === "keyword" && ( +
+ + +
+ )} + +
+ + +
+ + {error &&

{error}

} + +
+ + +
+
+ ); +} diff --git a/app/(app)/projects/[id]/uptime/monitor-actions.tsx b/app/(app)/projects/[id]/uptime/monitor-actions.tsx new file mode 100644 index 00000000..5cdecb67 --- /dev/null +++ b/app/(app)/projects/[id]/uptime/monitor-actions.tsx @@ -0,0 +1,53 @@ +"use client"; + +import { useTransition } from "react"; +import { useRouter } from "next/navigation"; +import { setMonitorEnabled, deleteMonitor } from "@/app/actions/monitors"; + +export function MonitorActions({ + projectId, + monitorId, + enabled, + name, +}: { + projectId: string; + monitorId: string; + enabled: boolean; + name: string; +}) { + const router = useRouter(); + const [pending, start] = useTransition(); + + function toggle() { + start(async () => { + await setMonitorEnabled(projectId, monitorId, !enabled); + router.refresh(); + }); + } + function remove() { + if (!window.confirm(`Delete monitor "${name}"?`)) return; + start(async () => { + await deleteMonitor(projectId, monitorId); + router.refresh(); + }); + } + + return ( +
+ + +
+ ); +} diff --git a/app/(app)/projects/[id]/uptime/page.tsx b/app/(app)/projects/[id]/uptime/page.tsx new file mode 100644 index 00000000..2e293d53 --- /dev/null +++ b/app/(app)/projects/[id]/uptime/page.tsx @@ -0,0 +1,188 @@ +// Per-project "Uptime" tab (uptime-monitoring-prd.md §3–§7). Lists monitors +// with their current up/down state and recent incidents; the worker runs the +// actual checks and sends down/recovery alerts. Degrades to empty states if the +// migration isn't applied yet. +import { notFound } from "next/navigation"; +import { createClient } from "@/lib/supabase/server"; +import { AddMonitorForm } from "./add-monitor-form"; +import { MonitorActions } from "./monitor-actions"; + +interface MonitorRow { + id: string; + name: string; + type: string; + target: string; + enabled: boolean; + current_state: "up" | "down" | "unknown"; + last_checked_at: string | null; + last_error: string | null; + last_response_ms: number | null; +} + +interface IncidentRow { + id: string; + monitor_id: string; + started_at: string; + ended_at: string | null; + cause: string | null; + duration_s: number | null; +} + +const STATE_COLOR: Record = { + up: "var(--color-pass)", + down: "var(--color-fail)", + unknown: "var(--color-muted)", +}; +const STATE_LABEL: Record = { + up: "Up", + down: "Down", + unknown: "Pending", +}; + +function fmt(ts: string | null): string { + return ts ? new Date(ts).toLocaleString() : "—"; +} +function dur(s: number | null): string { + if (s == null) return "—"; + if (s < 60) return `${s}s`; + if (s < 3600) return `${Math.round(s / 60)}m`; + return `${(s / 3600).toFixed(1)}h`; +} + +export default async function ProjectUptimePage({ + params, +}: { + params: Promise<{ id: string }>; +}) { + const { id } = await params; + const supabase = await createClient(); + + const { data: project } = await supabase + .from("projects") + .select("id") + .eq("id", id) + .maybeSingle(); + if (!project) notFound(); + + const { data: monitorsData } = await supabase + .from("monitors") + .select( + "id, name, type, target, enabled, current_state, last_checked_at, last_error, last_response_ms", + ) + .eq("project_id", id) + .order("created_at", { ascending: true }); + const monitors = (monitorsData ?? []) as MonitorRow[]; + + const { data: incidentsData } = await supabase + .from("monitor_incidents") + .select("id, monitor_id, started_at, ended_at, cause, duration_s") + .eq("project_id", id) + .order("started_at", { ascending: false }) + .limit(15); + const incidents = (incidentsData ?? []) as IncidentRow[]; + const nameOf = new Map(monitors.map((m) => [m.id, m.name])); + + return ( +
+
+
+

Uptime Monitors

+

+ We check each target on its interval and email you the moment it goes + down — and again when it recovers. HTTP, keyword, SSL-expiry, and TCP. +

+
+ +
+ + {/* Monitors */} + {monitors.length === 0 ? ( +
+ No monitors yet. Add one to start watching a URL, host, or port. +
+ ) : ( +
+ + + + + + + + + + + + {monitors.map((m) => ( + + + + + + + + ))} + +
StateMonitorLast checkLatency
+ + ● {STATE_LABEL[m.current_state]} + + {!m.enabled && ( + (paused) + )} + +
{m.name}
+
+ {m.type} · {m.target} +
+ {m.current_state === "down" && m.last_error && ( +
{m.last_error}
+ )} +
{fmt(m.last_checked_at)} + {m.last_response_ms != null ? `${m.last_response_ms} ms` : "—"} + + +
+
+ )} + + {/* Incidents */} +
+

Recent incidents

+ {incidents.length === 0 ? ( +
+ No incidents. 🎉 +
+ ) : ( +
    + {incidents.map((i) => ( +
  • +
    + {nameOf.get(i.monitor_id) ?? "monitor"} + {i.cause} +
    +
    +
    {fmt(i.started_at)}
    +
    + {i.ended_at ? `resolved · down ${dur(i.duration_s)}` : "ongoing"} +
    +
    +
  • + ))} +
+ )} +
+
+ ); +} diff --git a/app/actions/monitors.ts b/app/actions/monitors.ts new file mode 100644 index 00000000..66743b9c --- /dev/null +++ b/app/actions/monitors.ts @@ -0,0 +1,82 @@ +"use server"; + +import { revalidatePath } from "next/cache"; +import { createClient } from "@/lib/supabase/server"; + +export interface CreateMonitorInput { + projectId: string; + name: string; + type: "http" | "keyword" | "ssl" | "tcp"; + target: string; + intervalS?: number; + alertEmail?: string; + keyword?: string; + match?: "present" | "absent"; + expectedStatus?: number; + port?: number; + warnDays?: number; +} + +export async function createMonitor( + input: CreateMonitorInput, +): Promise<{ ok: boolean; error?: string }> { + const supabase = await createClient(); + const { + data: { user }, + } = await supabase.auth.getUser(); + if (!user) return { ok: false, error: "Not signed in." }; + + const name = input.name.trim(); + const target = input.target.trim(); + if (!name || !target) return { ok: false, error: "Name and target are required." }; + + const config: Record = {}; + if (input.type === "keyword") { + if (!input.keyword?.trim()) return { ok: false, error: "Keyword is required." }; + config.keyword = input.keyword.trim(); + config.match = input.match ?? "present"; + } + if (input.type === "http" && input.expectedStatus) config.expected_status = input.expectedStatus; + if (input.type === "tcp" && input.port) config.port = input.port; + if (input.type === "ssl" && input.warnDays) config.warn_days = input.warnDays; + + const { error } = await supabase.from("monitors").insert({ + project_id: input.projectId, + name, + type: input.type, + target, + config, + interval_s: input.intervalS && input.intervalS >= 60 ? input.intervalS : 60, + alert_email: (input.alertEmail || user.email || "").trim() || null, + }); + if (error) return { ok: false, error: `Could not create monitor: ${error.message}` }; + + revalidatePath(`/projects/${input.projectId}/uptime`); + return { ok: true }; +} + +export async function setMonitorEnabled( + projectId: string, + monitorId: string, + enabled: boolean, +): Promise<{ ok: boolean; error?: string }> { + const supabase = await createClient(); + const { error } = await supabase + .from("monitors") + .update({ enabled }) + .eq("id", monitorId); + if (error) return { ok: false, error: error.message }; + revalidatePath(`/projects/${projectId}/uptime`); + return { ok: true }; +} + +export async function deleteMonitor( + projectId: string, + monitorId: string, +): Promise<{ ok: boolean; error?: string }> { + const supabase = await createClient(); + const { error } = await supabase.from("monitors").delete().eq("id", monitorId); + if (error) return { ok: false, error: error.message }; + revalidatePath(`/projects/${projectId}/uptime`); + return { ok: true }; +} diff --git a/components/project-tabs-nav.tsx b/components/project-tabs-nav.tsx index d816ee50..92151338 100644 --- a/components/project-tabs-nav.tsx +++ b/components/project-tabs-nav.tsx @@ -54,6 +54,12 @@ const TABS: ProjectTab[] = [ href: (id) => `/projects/${id}/security`, matches: (p, id) => p.startsWith(`/projects/${id}/security`), }, + { + id: "uptime", + label: "Uptime", + href: (id) => `/projects/${id}/uptime`, + matches: (p, id) => p.startsWith(`/projects/${id}/uptime`), + }, { id: "autoblog", label: "Autoblog", diff --git a/lib/uptime.ts b/lib/uptime.ts new file mode 100644 index 00000000..51d12523 --- /dev/null +++ b/lib/uptime.ts @@ -0,0 +1,299 @@ +// Uptime monitoring engine (uptime-monitoring-prd.md §3–§7). Runs due checks +// (HTTP / keyword / SSL / TCP), advances the up/down state machine with +// multi-failure confirmation, opens/closes incidents, and sends down/recovery +// email alerts. Called from worker/index.ts on a short interval. +import net from "node:net"; +import tls from "node:tls"; +import type { SupabaseClient } from "@supabase/supabase-js"; +import type { Resend } from "resend"; + +export interface MonitorRow { + id: string; + project_id: string; + name: string; + type: "http" | "keyword" | "ssl" | "tcp"; + target: string; + config: Record; + interval_s: number; + timeout_s: number; + fail_threshold: number; + recover_threshold: number; + current_state: "up" | "down" | "unknown"; + consecutive_failures: number; + consecutive_successes: number; + alert_email: string | null; +} + +interface CheckResult { + ok: boolean; + responseMs: number | null; + statusCode: number | null; + error: string | null; +} + +const FROM = process.env.RESEND_FROM ?? "CrawlProof "; + +async function checkHttp(m: MonitorRow): Promise { + const timeout = m.timeout_s * 1000; + const ctrl = new AbortController(); + const t = setTimeout(() => ctrl.abort(), timeout); + const started = Date.now(); + try { + const res = await fetch(m.target, { + method: "GET", + redirect: "follow", + signal: ctrl.signal, + headers: { "user-agent": "CrawlProof-Uptime/1.0" }, + }); + const responseMs = Date.now() - started; + const expect = Number(m.config.expected_status ?? 0); + const ok = expect ? res.status === expect : res.status >= 200 && res.status < 400; + + if (m.type === "keyword") { + const body = await res.text(); + const keyword = String(m.config.keyword ?? ""); + const mode = m.config.match === "absent" ? "absent" : "present"; + const present = keyword.length > 0 && body.includes(keyword); + const kwOk = mode === "present" ? present : !present; + return { + ok: ok && kwOk, + responseMs, + statusCode: res.status, + error: kwOk ? null : `keyword "${keyword}" ${mode === "present" ? "missing" : "present"}`, + }; + } + return { + ok, + responseMs, + statusCode: res.status, + error: ok ? null : `unexpected status ${res.status}`, + }; + } catch (e) { + return { ok: false, responseMs: null, statusCode: null, error: (e as Error).message }; + } finally { + clearTimeout(t); + } +} + +function parseHostPort(target: string, defaultPort: number): { host: string; port: number } { + let s = target.trim(); + try { + if (s.includes("://")) { + const u = new URL(s); + return { host: u.hostname, port: u.port ? Number(u.port) : defaultPort }; + } + } catch { + /* fall through to host:port parsing */ + } + const [host, port] = s.split(":"); + return { host, port: port ? Number(port) : defaultPort }; +} + +function checkTcp(m: MonitorRow): Promise { + const { host, port } = parseHostPort(m.target, Number(m.config.port ?? 0) || 80); + const started = Date.now(); + return new Promise((resolve) => { + const socket = net.connect({ host, port, timeout: m.timeout_s * 1000 }); + const done = (ok: boolean, error: string | null) => { + socket.destroy(); + resolve({ ok, responseMs: ok ? Date.now() - started : null, statusCode: null, error }); + }; + socket.once("connect", () => done(true, null)); + socket.once("timeout", () => done(false, "connection timed out")); + socket.once("error", (e) => done(false, e.message)); + }); +} + +function checkSsl(m: MonitorRow): Promise { + const { host, port } = parseHostPort(m.target, 443); + const warnDays = Number(m.config.warn_days ?? 14); + const started = Date.now(); + return new Promise((resolve) => { + const socket = tls.connect( + { host, port, servername: host, timeout: m.timeout_s * 1000 }, + () => { + const cert = socket.getPeerCertificate(); + const responseMs = Date.now() - started; + if (!cert || !cert.valid_to) { + socket.destroy(); + return resolve({ ok: false, responseMs, statusCode: null, error: "no certificate" }); + } + const days = Math.floor((new Date(cert.valid_to).getTime() - Date.now()) / 86_400_000); + socket.destroy(); + resolve({ + ok: days > warnDays, + responseMs, + statusCode: null, + error: days > warnDays ? null : `cert expires in ${days}d (warn ${warnDays}d)`, + }); + }, + ); + socket.once("timeout", () => { + socket.destroy(); + resolve({ ok: false, responseMs: null, statusCode: null, error: "tls connection timed out" }); + }); + socket.once("error", (e) => { + socket.destroy(); + resolve({ ok: false, responseMs: null, statusCode: null, error: e.message }); + }); + }); +} + +function runCheck(m: MonitorRow): Promise { + switch (m.type) { + case "http": + case "keyword": + return checkHttp(m); + case "tcp": + return checkTcp(m); + case "ssl": + return checkSsl(m); + } +} + +async function sendAlert( + resend: Resend, + m: MonitorRow, + kind: "down" | "up", + detail: string, +): Promise { + if (!m.alert_email) return; + const down = kind === "down"; + const subject = `${down ? "🔴 DOWN" : "🟢 RECOVERED"}: ${m.name}`; + const html = ` +
+

${down ? "Monitor is DOWN" : "Monitor recovered"}

+

${m.name} (${m.type})

+

${m.target}

+

${detail}

+

${new Date().toISOString()} · CrawlProof Uptime

+
`; + try { + await resend.emails.send({ from: FROM, to: m.alert_email, subject, html }); + } catch (e) { + console.warn("[uptime] alert send failed", (e as Error).message); + } +} + +async function processOne( + supabase: SupabaseClient, + resend: Resend | null, + m: MonitorRow, +): Promise<"up" | "down" | "same"> { + const result = await runCheck(m); + + await supabase.from("monitor_checks").insert({ + monitor_id: m.id, + ok: result.ok, + response_ms: result.responseMs, + status_code: result.statusCode, + error: result.error, + }); + + let failures = m.consecutive_failures; + let successes = m.consecutive_successes; + let state = m.current_state; + let transition: "up" | "down" | "same" = "same"; + + if (result.ok) { + successes += 1; + failures = 0; + if (state !== "up" && successes >= m.recover_threshold) { + const wasDown = state === "down"; + state = "up"; + if (wasDown) transition = "up"; + } + } else { + failures += 1; + successes = 0; + if (state !== "down" && failures >= m.fail_threshold) { + state = "down"; + transition = "down"; + } + } + + await supabase + .from("monitors") + .update({ + current_state: state, + consecutive_failures: failures, + consecutive_successes: successes, + last_checked_at: new Date().toISOString(), + last_error: result.error, + last_response_ms: result.responseMs, + }) + .eq("id", m.id); + + if (transition === "down") { + await supabase.from("monitor_incidents").insert({ + monitor_id: m.id, + project_id: m.project_id, + cause: result.error ?? "check failed", + }); + if (resend) await sendAlert(resend, m, "down", result.error ?? "Check failed."); + } else if (transition === "up") { + // Close the open incident and stamp its duration. + const { data: open } = await supabase + .from("monitor_incidents") + .select("id, started_at") + .eq("monitor_id", m.id) + .is("ended_at", null) + .order("started_at", { ascending: false }) + .limit(1) + .maybeSingle(); + if (open) { + const durationS = Math.round((Date.now() - new Date(open.started_at).getTime()) / 1000); + await supabase + .from("monitor_incidents") + .update({ ended_at: new Date().toISOString(), duration_s: durationS }) + .eq("id", open.id); + if (resend) { + await sendAlert(resend, m, "up", `Back up after ${durationS}s of downtime.`); + } + } else if (resend) { + await sendAlert(resend, m, "up", "Back up."); + } + } + + return transition; +} + +// Claim + check all due monitors. Safe on a short interval; claims each row by +// pushing due_at forward before checking so overlapping sweeps don't double-run. +export async function processDueMonitors( + supabase: SupabaseClient, + resend: Resend | null, +): Promise<{ checked: number; down: number; up: number }> { + const { data: due } = await supabase + .from("monitors") + .select( + "id, project_id, name, type, target, config, interval_s, timeout_s, fail_threshold, recover_threshold, current_state, consecutive_failures, consecutive_successes, alert_email", + ) + .eq("enabled", true) + .lte("due_at", new Date().toISOString()) + .limit(50); + + const monitors = (due ?? []) as MonitorRow[]; + if (monitors.length === 0) return { checked: 0, down: 0, up: 0 }; + + // Claim: push due_at forward immediately. + await Promise.all( + monitors.map((m) => + supabase + .from("monitors") + .update({ due_at: new Date(Date.now() + m.interval_s * 1000).toISOString() }) + .eq("id", m.id), + ), + ); + + const results = await Promise.allSettled(monitors.map((m) => processOne(supabase, resend, m))); + let down = 0; + let up = 0; + for (const r of results) { + if (r.status === "fulfilled") { + if (r.value === "down") down++; + else if (r.value === "up") up++; + } + } + return { checked: monitors.length, down, up }; +} diff --git a/supabase/migrations/20260707120000_uptime_monitors.sql b/supabase/migrations/20260707120000_uptime_monitors.sql new file mode 100644 index 00000000..88f8bf0a --- /dev/null +++ b/supabase/migrations/20260707120000_uptime_monitors.sql @@ -0,0 +1,77 @@ +-- Uptime monitoring V1 (uptime-monitoring-prd.md §3–§7): HTTP/keyword/SSL/TCP +-- monitors with an up/down state machine and incident tracking. The worker +-- runs due checks; down/recovery alerts go out by email. + +create table if not exists public.monitors ( + id uuid primary key default gen_random_uuid(), + project_id uuid not null references public.projects(id) on delete cascade, + name text not null, + type text not null check (type in ('http', 'keyword', 'ssl', 'tcp')), + target text not null, + config jsonb not null default '{}', + interval_s int not null default 60, + timeout_s int not null default 10, + fail_threshold int not null default 2, + recover_threshold int not null default 1, + enabled boolean not null default true, + current_state text not null default 'unknown' + check (current_state in ('up', 'down', 'unknown')), + consecutive_failures int not null default 0, + consecutive_successes int not null default 0, + last_checked_at timestamptz, + last_error text, + last_response_ms int, + alert_email text, + due_at timestamptz not null default now(), + created_at timestamptz not null default now() +); +create index if not exists monitors_due_idx on public.monitors(enabled, due_at); +create index if not exists monitors_project_idx on public.monitors(project_id); + +create table if not exists public.monitor_incidents ( + id uuid primary key default gen_random_uuid(), + monitor_id uuid not null references public.monitors(id) on delete cascade, + project_id uuid not null references public.projects(id) on delete cascade, + started_at timestamptz not null default now(), + ended_at timestamptz, + cause text, + duration_s int +); +create index if not exists monitor_incidents_monitor_idx + on public.monitor_incidents(monitor_id, started_at desc); + +create table if not exists public.monitor_checks ( + id bigint generated always as identity primary key, + monitor_id uuid not null references public.monitors(id) on delete cascade, + checked_at timestamptz not null default now(), + ok boolean not null, + response_ms int, + status_code int, + error text +); +create index if not exists monitor_checks_monitor_idx + on public.monitor_checks(monitor_id, checked_at desc); + +alter table public.monitors enable row level security; +alter table public.monitor_incidents enable row level security; +alter table public.monitor_checks enable row level security; + +-- Owner-scoped, matching public.project_repos / public.port_scans. +create policy "monitors owner select" on public.monitors for select + using (project_id in (select id from public.projects where owner_id = auth.uid())); +create policy "monitors owner insert" on public.monitors for insert + with check (project_id in (select id from public.projects where owner_id = auth.uid())); +create policy "monitors owner update" on public.monitors for update + using (project_id in (select id from public.projects where owner_id = auth.uid())); +create policy "monitors owner delete" on public.monitors for delete + using (project_id in (select id from public.projects where owner_id = auth.uid())); + +create policy "monitor_incidents owner select" on public.monitor_incidents for select + using (project_id in (select id from public.projects where owner_id = auth.uid())); + +create policy "monitor_checks owner select" on public.monitor_checks for select + using (monitor_id in ( + select m.id from public.monitors m + join public.projects p on p.id = m.project_id + where p.owner_id = auth.uid() + )); diff --git a/worker/index.ts b/worker/index.ts index 56ba9e32..f5af4f04 100644 --- a/worker/index.ts +++ b/worker/index.ts @@ -37,6 +37,7 @@ import { getOrMintInstallationToken } from "../lib/github/installations"; import { listInstallationRepos } from "../lib/github/app"; import { processUserAlerts } from "../lib/alerts/worker"; import { processDuePortScans } from "../lib/prober-queue"; +import { processDueMonitors } from "../lib/uptime"; const supabaseUrl = process.env.NEXT_PUBLIC_SUPABASE_URL!; const supabaseKey = process.env.SUPABASE_SERVICE_ROLE_KEY!; @@ -1271,6 +1272,16 @@ setInterval( () => portScanSweep().catch((e) => console.error("[worker] port scan sweep", e)), 5_000, ); +// Uptime monitors: run due HTTP/keyword/SSL/TCP checks, advance the up/down +// state machine, and send down/recovery alerts (uptime-monitoring-prd.md §4). +async function uptimeSweep() { + const r = await processDueMonitors(supabase, resend); + if (r.down || r.up) console.log("[worker] uptime sweep", r); +} +setInterval( + () => uptimeSweep().catch((e) => console.error("[worker] uptime sweep", e)), + 15_000, +); // Bind to loopback by default so the worker isn't reachable from the public // internet when colocated with the app. Override with WORKER_BIND=0.0.0.0 to