diff --git a/AGENTS.md b/AGENTS.md index faa5b15b5..59f08ef36 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,13 +29,13 @@ One binary: Eighteen internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): - **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers -- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive`/`longestGapWindow` for the two settings folded over every tenant served, and `defaultSetting`/`onDefaultAdopt` for the one resource a process still has one of, the MQ, which follows tenant `0`; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it +- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive`/`longestGapWindow` for the two settings folded over every tenant served, and `defaultSetting`/`onDefaultAdopt` for the one resource a process still has one of, the MQ, which follows tenant `0`; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index). Every key leads with the tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for a result and its singleflight, `...
.` for a namespace — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` advances the tenant version that leads every namespace key of one tenant, orphaning every cached query keyed by its tables in one step (a pipe result names no table and keeps its TTL, [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the table-keyed cache (the structured-query results) of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) - **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config - **`chsql/`** — dependency-free ClickHouse SQL helpers shared by `query`/`policy` (avoids an import cycle): `QuoteIdent` (backtick-quote every identifier) + `BindUnsafe` (reject names with a literal `?`) - **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability — boot is the validator, there is no dry run -- **`dedupe/`** — `Deduplicator` interface → `Embedded` (Pebble), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, the seam a shared backend later slots into; `Managed` opens its store through a function, `Embedded(dir)` for Pebble), rooted at `data_dir//dedupe` whatever the directory's shape (the four files are tenant `0`; an earlier layout's `data_dir/pebble` is moved there at boot when `data_dir/0/dedupe` is absent, and left alone and warned about when both exist), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its data kept on disk otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7) +- **`dedupe/`** — `Deduplicator` interface → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) - **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`), so non-admin callers never reach the proxy. A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, a topic without one is refused, and a pre-tenant subject reads as tenant `0`'s) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds the byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload) and `Stats` (the system gauges' source). Subjects, prefixes, wildcards, token encoding, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`); `internal/app` constructs it and hands everything else a `mq.Broker` @@ -43,9 +43,9 @@ Eighteen internal packages under `internal/` (plus `internal/testutil/` for shar - **`pipes/`** — Named query pipes: `NamedQuery` type + `BindParams` + `Source` (read per request; `settings.Store` in production, `Static(q...)` in tests) - **`policy/`** — Hasura-style access control, **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant splits by operation into `SelectPermissions` (columns, row `filter`, aggregations, the `max_*` limits) and `InsertPermissions` (columns, `check`) — so a field only one side honors does not exist on the other. `Evaluate()` resolves ONE operation and leaves the other side **nil** (`Select *ResolvedSelect` / `Insert *ResolvedInsert`), which every accessor fails closed on — nil is "not resolved", distinct from an empty side, which is "unrestricted" (what the admin return builds). Claim templating (`{{ jwt.claim.path }}`) resolves during that call. Policies come from `Source`, a `func() *Policy` read per call (`settings.Store.Policy` in production, `Static(p)` in tests) - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing -- **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes -- **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes; not `nats` or `pebble` in any letter case, the entries `data_dir` keeps for itself — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper folds over the tenants served (`longestGapWindow`); each served tenant has a schema registry of its own (story 6) +- **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes +- **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) +- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper folds over the tenants served (`longestGapWindow`); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -58,7 +58,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. 6. **Dead Letter Queue** — failed batch inserts publish to `WAVEHOUSE_DLQ` (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format` — a pre-v2 envelope carries none — or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. 7. **Auth: always on, fail-loud, decoupled from authz (security)** — the JWT middleware always runs (no `auth.enabled`/`dev_mode` flag); it verifies with HMAC **or** JWKS (not both), with accepted `alg` pinned to the active verifier and checked before any key is used (rejects `alg:none` and cross-family confusion). No/invalid/expired token → empty role → policy `default_role`, with the bad-token reason stashed so a denying gate returns a loud `401`, not a bare `403`; the one token outcome that never reaches `default_role` is a verifier still fetching its JWKS (`auth.ErrVerifierPending` → `503` + `Retry-After`, `api.refuseUnverifiable`). Elevated access needs a valid granted role. **Sanctioned exception:** a configured non-JWT operator key (`auth.operator_key`; presented via `Authorization: Operator ` or the `X-Operator-Key` alias) deliberately couples authN+authZ — a constant-time match authorizes a full-access platform operator (stamps the admin role plus an operator bit) independent of the verifier (see #11). Detail: architecture.md § `api/` + `internal/auth`; see also #11, §Security Considerations. -8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's Pebble store via `dedupe.Managed`, one per tenant in `dedupe.Stores`; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key, overridable per table. +8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key, overridable per table. 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. 10. **Active Sweeper** — purges NATS messages that are both ACKed (written to CH) and older than the gap window; SSE gap-fill uses `DeliverByStartTime`, no in-process ring buffer. 11. **Hasura-style access control: fail-closed (security)** — `policy.IsAdmin` (role == `admin_role`, **exact case-sensitive**, default `"admin"`) is the single admin check, shared by `Evaluate`/`ResolveRole`/`Validate`/the `/v1/ops` gate/`RoleAllowed`. Empty/absent role matches nothing (no `"*"` wildcard); `Validate` rejects empty role keys; a `nil` policy (deleted) denies **everyone incl. admin** via a role — a total lockout for token-based callers, so recovery is writing `policies.json` and reloading, never an implicit admin grant (**exception:** the operator key's `auth.IsOperator` bit passes the `/v1/ops` gate even under a `nil` policy — a deliberate break-glass that can `POST /v1/ops/settings/reload` over HTTP, see #7). Over a nested settings directory the `/v1/ops` gate reads no policy at all — those routes reach every tenant, so the operator key alone passes and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, not from what was wired. `default_role` is the one sanctioned roleless exception (`ResolveRole` maps empty → it pre-eval); `default_role == admin_role` is permitted but dev-only and loudly warned (`policy.DefaultRoleGrantsAdmin`). Preserve when touching `internal/policy` (policy twin of #13; see #159). Detail: architecture.md § `policy/`. diff --git a/CHANGELOG.md b/CHANGELOG.md index eacf23992..5bf580196 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,11 +10,12 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Added +- **A tenant removed or rejected at runtime has its open streams ended** (`internal/stream/{hub,subscriber,bucket}.go` (+ tests), `internal/api/stream.go` (+ tests), `internal/api/router.go`, `internal/settings/{registry,tree}.go` (+ tests), `internal/ingest/worker.go` (+ tests), `internal/app/wire.go` (+ tests), `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/settings-directory.mdx`, `AGENTS.md`): story 3 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). A `GET /v1/stream` used to outlive its tenant: the hub read no policy for it and withheld every row while the keepalive wheel held the connection open, so the client could not tell it from a quiet table. After every reload the hub now evicts the subscribers of each tenant no longer served, removed or rejected alike (`Hub.Prune`, through the close-once `Subscriber.Evict`), and the handler ends the stream: a gap-fill in progress included, and a stream `TenantMW` admitted just before the reload but registered just after it, which the handler checks for as it registers. The client's reconnect gets `404` (removed) or `503` (rejected); the SDK stops on the first and retries the second, resuming from `Last-Event-ID` once the folder is back — except in a browser going cross-origin where tenant `0` is not served or its CORS list does not admit the page, which cannot read either refusal (both are decorated from tenant `0`'s list) and re-dials as after a dropped connection. A flat directory never stops serving tenant `0`, so nothing changes there. Removing a tenant is deleting its folder, then reloading the whole directory, the last folder included: a server started with tenant folders reads the emptied directory as no folder left, where it read as a change of shape and the reload was rejected whole, leaving that tenant served; at boot an empty directory still reads as the four files, missing. Reloading the deleted folder by name leaves its tenant rejected. `GET /v1/health` keeps resolving a tenant, deliberately: its `404` tells a caller with no token no more than every tenant route's does, since they all answer before authenticating, and resolving answers a served tenant's ping from that tenant's own CORS list. A batch queued for a tenant with no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and meets its DLQ switch once, whole, logged once per batch rather than twice per row; a tenant no longer served reads the switch as on, so its queued rows are parked under its own subject rather than dropped, inserted into another tenant's ClickHouse, or left unacked to hold the ack floor that the one shared stream's purge waits on. Over a nested directory `/livez` no longer keeps naming a tenant that stopped being served before any tenant completed a first discovery: the diagnostic goes back to `no tenant has completed a first discovery yet`. - **One ClickHouse pool per tuple and one schema registry per tenant** (`internal/chconn/chconn.go` (+ tests), `internal/discovery/discovery.go` (+ tests), `internal/app/discoveries.go` (new), `internal/app/{app,wire}.go` (+ tests), `internal/api/{schema,ingest,structured_query,pipes,query,health,errors,clickhouse_exec}.go` (+ tests), `internal/stream/hub.go`, `internal/ingest/worker.go`, `internal/cache/{cache,local,version_manager}.go` (+ tests), `internal/testutil/{testutil,mocks}.go`, `tests/integration/{setup,tenants,boot_resilience,query_limits}_test.go`, `clients/ts/src/{schema,table,sql,client,types}.ts` (+ tests), `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/{settings-directory,configuration,access-control,reverse-proxy}.mdx`, `docs/src/content/docs/sdk/{admin,reference,queries}.md`, `AGENTS.md`): the second slice of story 6 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files beyond the three noted at the end. The process opens one native pool per distinct `clickhouse.addr` / `database` / `username` / password / `tls` tuple among the served tenants (`chconn.Identity`, `chconn.Pools`), shared by the tenants naming it and sized to their largest `max_open_conns` and `max_idle_conns`; `http_port`, `http_scheme`, `headers` and `query_timeout` stay each tenant's own. Every reload reconciles the pools: a new tuple opens (never dials), a tenant whose tuple changed is repointed, a tuple no tenant names closes after the longest `query_timeout` among the tenants it had, and a pool whose largest ask changed is resized with the same grace. The boot config's `clickhouse.max_total_conns` now bounds the open pools together: boot is refused naming the sum and the ceiling; at a reload a resize above it is refused with the pool kept at its size, and a tuple that cannot be opened — the ceiling, a certificate file that cannot be read, or options the driver refuses — leaves its tenants on the pool they had (the keep-previous-wiring rule of the connection ceiling) or on none when they had none; both are logged and the next reload retries. A tenant on no pool fails closed: `503` with `Retry-After: 30` on `POST /v1/query`, `GET/POST /v1/pipes/{name}`, `POST /v1/ops/query` and `POST /v1/ops/schema/refresh`, ahead of the cache. Each served tenant gets a `discovery.SchemaRegistry` of its own over its pool, kept fresh by its own loop — the boot retry until the first success, then `schema.refresh_interval` with the first refresh at a random point within the interval so tenants adopted together do not refresh together — created and stopped from the settings reload and stopped under `App.Close`; `wavehouse_schema_refresh_failures_total{tenant}` counts a loop's failed attempts. `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the strict `?tenant=` the pipe reads take (absent is tenant `0`, `400` malformed, `404` unknown, `503` rejected); the SDK sends it as the `tenant` option of `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()` and `wh.sql()`. Over a nested directory `/livez` (and `/v1/health`) is `503` with the latest discovery failure, naming its tenant, while no tenant has completed a first discovery, then `200` for the rest of the process lifetime; `/readyz` pings every open pool at once, is ready at the first answer, and names every pool that did not answer when none does — one tenant's ClickHouse outage is its log line and counter, never a probe failure. The ingest worker inserts each batch into its own tenant's ClickHouse — the tenant the message's topic names, through that tenant's HTTP wiring (`chconn.Pools.Target`); a tenant on no pool takes the failure path an unreachable ClickHouse takes — and its cache invalidation fans out to the tenants on the same address and database as the batch's tenant, whatever their user or `tls` block (they read the same tables), rather than to every known tenant, and a tenant adopted after an absence — rejected or removed, so out of that fan-out — or moved to another address or database has its cached structured-query results orphaned in one step (`Cache.InvalidateTenant`, a tenant generation in every version key), so a repaired folder never serves query rows cached before the inserts it missed (a pipe result names no table, so neither this nor any insert invalidates it: it stays until its TTL expires). Three changes reach the single-tenant directory too: a table lookup before the first discovery is a `503` with `Retry-After: 5` (`schema not loaded yet`) rather than a `404`, on `POST /v1/ingest`, `POST /v1/query` and `GET /v1/ops/schema` (the list included, where `[]` would read as no tables); the first periodic refresh fires at a random point within the interval rather than a full interval after boot; and a reload that moves `clickhouse.addr` or `clickhouse.database` now orphans the structured-query results cached before it, where they were served until their TTL. Until [#529](https://github.com/Wave-RF/WaveHouse/issues/529) every tenant's user authenticates with `WH_CH_PASSWORD`, so the tuple is in effect the address, database, username and `tls` block. - **One token verifier per tenant, built off the boot and reload paths** (`internal/auth/auth.go` (+ tests), `internal/api/router.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `go.mod`, `docs/src/content/docs/sdk/{reference,streaming}.md`, `docs/src/content/docs/{architecture,deployment,api}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `SECURITY.md`): story 9 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). Over a nested settings directory each tenant's folder now wires that tenant's verifier (`auth.jwks_url`, `auth.role_claim`), so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its identity provider's key set — under any other tenant's header it is refused as invalid — where before one verifier, tenant `0`'s, accepted a token under any header. `auth.Config` is now the boot-config half alone (the HMAC secret and the operator key, shared by every tenant) and the new `auth.Wiring` a tenant's half; `NewAuthenticator` builds no verifier, `Reconfigure(id, wiring)` gives a tenant one — swapped atomically when its wiring changed, kept when it did not — `Prune` drops the verifiers of the tenants a reload stopped serving, rejected or removed alike (no work runs for a tenant that is not served; a folder adopted again gets a fresh verifier), and `Close` stops every JWKS refresh as a component of `App.Close`. This holds on every reload because the registry's `AfterAdopt` hooks run after every reload it applied, empty list included, so a per-tenant reload that rejects a folder drops its verifier then rather than at the next adoption. The middleware reads the request tenant's verifier through an injected `auth.TenantSource` — the store `api.TenantMW` resolved names its tenant with `settings.Store.Tenant()`, one read of the context — a tenant-exempt route (the ops tree) verifies as tenant `0`, and a tenant with no verifier fails closed. The operator key stamps the request tenant's `admin_role`, read from that tenant's policy, rather than tenant `0`'s. A JWKS key set is fetched on its own goroutine: the verifier is in place at once and *pending* until a set has been stored — the first fetch retried with backoff from one second to a minute, a stored set then kept fresh by the library hourly and, rate-limited, on an unknown key id — so neither boot nor a reload (which holds the registry's lock) waits on the endpoint, and **an unreachable JWKS no longer refuses boot** — it logs (`jwks refresh failed; no token validates until it succeeds`) and that tenant alone is affected. A token checked against a pending verifier is a new outcome, `auth.ErrVerifierPending`: every `/v1` route answers it `503 {"error": "token verifier not ready: …"}` with `Retry-After: 30` (`api.refuseUnverifiable`) rather than evaluate the request under the `default_role`, which could accept its data under a lesser role while another pod holding the keys would have served it as its own; a request without a token, and the operator key, are unaffected. Every fetch goes through one client that caps the response at 1 MiB, refusing a larger one as unreachable with `jwks response exceeds 1048576 bytes` as the logged cause. Nothing changes for a flat directory beyond that boot rule: its one tenant gets exactly the verifier it had. The boot warning for the secretless posture (no `auth.jwt_secret` and no `jwks_url`, whose token refusal landed in [#607](https://github.com/Wave-RF/WaveHouse/pull/607)) is now per tenant — each served tenant with no `jwks_url` while the boot secret is unset — where one line that any JWKS tenant silenced used to stand for the whole directory. - **ClickHouse TLS, HTTP-interface headers, pool sizes and a connection ceiling** (`internal/settings/{settings,validate,store}.go` + seed, `internal/chconn/chconn.go`, `internal/ingest/worker.go`, `internal/api/query.go`, `internal/config/config.go`, `internal/app/wire.go`, `deployments/compose/settings/config.json`, `deployments/compose/standalone.yaml`, `config.yaml`, `docs/src/content/docs/{settings-directory,configuration,reverse-proxy}.mdx`, `docs/src/content/docs/{architecture,deployment}.md`): the tenant-agnostic first slice of story 6 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `config.json`'s `clickhouse` block gains a `tls` block (`enabled`, `ca_file`, `cert_file`, `key_file`, `insecure_skip_verify`, `server_name`), a `headers` map for the HTTP interface, and `max_open_conns` / `max_idle_conns`. **Every key is required, so an existing `config.json` must add them**; the seed values (`tls.enabled: false` with the other `tls` keys empty, `headers: {}`, `10` / `5`) change nothing. `tls.enabled` switches the native hop to TLS, `http_scheme` stays the HTTP hop's switch, and the material applies to whichever hop uses TLS: the driver gets the TLS config and the pool sizes, the ingest worker and the raw-SQL proxy get the TLS config and the headers, set ahead of their own so the credentials win (naming `X-ClickHouse-User`, `X-ClickHouse-Key` or `Authorization` is a validation error, and so are two spellings of one name). Validation checks shape only — the paths are not opened, so `wavehouse validate` runs anywhere — warns when `insecure_skip_verify` is on or when only one of the two hops is on TLS (each carries the credentials in the clear without it), and a certificate file that cannot be read or parsed refuses boot or leaves a reload's connection unchanged; the files are read when the connection is built, so a file replaced in place needs a restart. Boot config gains the optional `clickhouse.max_total_conns` (`WH_CH_MAX_TOTAL_CONNS`, `0` = no ceiling): a settings pool above it refuses boot, naming both numbers in the error, and a reload that raises the pool above it is refused and logged (the reload still reports adopted), leaving the connection as it was. -- **One dedupe store per tenant, rooted at `//dedupe`** (`internal/dedupe/stores.go` (new, + tests), `internal/dedupe/{managed,embedded}.go` (+ tests), `internal/settings/registry.go` (+ tests), `internal/settings/tree_test.go`, `internal/tenant/tenant.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/observability/metrics.go` (+ tests), `internal/config/config.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `deployments/Dockerfile{,.goreleaser}`, `docs/src/content/docs/{deployment,architecture}.md`, `docs/src/content/docs/{configuration,settings-directory}.mdx`, `AGENTS.md`): story 7 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). Each tenant has a Pebble store of its own at `//dedupe` whatever the settings directory's shape — the four files are tenant `0`, so a standalone deployment's store now lives at `/0/dedupe`, and boot moves an earlier layout's `/pebble` there once, seen ids included (both present is left alone and warned about, the new one in use; a failed move refuses boot rather than open an empty store and let duplicates through in silence), so an implicit tenant `0` becomes an explicit `0` folder with nothing to restructure. A tenant's store follows its own folder's `dedupe.enabled` rather than tenant `0`'s: `dedupe.Stores` holds one `dedupe.Managed` per tenant, built on first use through a `dedupe.Factory` (`func(tenant.ID) *Managed`) so a shared backend can later put the tenant in the key without the callers changing, and one reconcile hook on the registry sets every store to what the registry says — open exactly when the tenant is served with its switch on, closed when it is switched off, rejected, or removed, its data left on disk for the folder that restores it (the seen ids are still there). The ingest handler picks the tenant's store off the request's `settings.Store`, which names its tenant (`Store.Tenant()`), so the same event id is first seen under each tenant that sends it and a duplicate only within the tenant that sent it before. A store that fails to open follows the registry's rule for the shape: a flat directory still refuses boot, a nested one fails closed for that tenant alone (its ingest answers `500 dedupe failed`, the next reload retries) while every other tenant carries on. The registry's `AfterAdopt` hooks now run after every reload it applied, with an empty list for one that only rejected or removed a tenant, so a gone tenant's store closes on that reload rather than at the next adoption (and the shared keepalive wheel is recomputed on it too). The Pebble system gauges report the process's figures summed across the tenants' stores, and `RegisterSystemMetrics` takes a stats function for them. `App.Close` releases the stores as one component — one Pebble close per tenant with dedupe on, in turn — inside the same 5s release budget. `dedupe.Managed` now opens its store through a function (`dedupe.Embedded(dir)` for Pebble), so the wiring's factory is the one place Pebble is named and a shared backend behind the same switch semantics is a wiring change. `dedupe.Stores` edits its map under its lock and does the stores' I/O outside it, so one tenant's close, or a scrape waiting on a store, never stalls another tenant's lookup. `nats` and `pebble`, in any letter case, are no longer tenant ids — `data_dir` keeps those names for itself, and a tenant's state lives at `/` — so a nested folder so named is skipped with a finding like any other name that is not a tenant id, and a header or `?tenant=` naming one is a `400`. +- **One dedupe store per tenant, every tenant's seen ids in one Pebble instance** (`internal/dedupe/stores.go` (new, + tests), `internal/dedupe/{dedupe,managed,embedded}.go` (+ tests), `internal/settings/registry.go` (+ tests), `internal/api/ingest.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/observability/metrics.go` (+ tests), `internal/testutil/mocks.go`, `internal/config/config.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/{deployment,architecture}.md`, `docs/src/content/docs/{configuration,settings-directory}.mdx`, `AGENTS.md`): stories 7 and 3 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). Each tenant has a store of its own whatever the settings directory's shape — the four files are tenant `0` — and every tenant's store is a share of one Pebble instance at `/pebble`, each key led by its tenant, so a thousand tenants cost one instance's goroutines, open files and heap rather than a thousand. Where the stores live is the embedded implementation's call (`dedupe.NewEmbedded`, handed `data_dir` once); the instance is open while any tenant's store is, so a server with dedupe off opens nothing, and an implicit tenant `0` becomes an explicit `0` folder with nothing to restructure. A tenant's store follows its own folder's `dedupe.enabled` rather than tenant `0`'s: `dedupe.Stores` holds one `dedupe.Managed` per tenant, built on first use through a `dedupe.Factory` (`func(tenant.ID) *Managed`; `dedupe.Embedded.Tenant` in production), and one reconcile hook on the registry sets every store to what the registry says — open exactly when the tenant is served with its switch on, closed when it is switched off, rejected, or removed, its data left on disk for the folder that restores it (the seen ids are still there). The ingest handler picks the tenant's store off the request's `settings.Store`, which names its tenant (`Store.Tenant()`), so the same event id is first seen under each tenant that sends it and a duplicate only within the tenant that sent it before. An instance that fails to open follows the registry's rule for the shape: a flat directory still refuses boot, a nested one fails closed for every tenant with dedupe on (their ingest answers `500 dedupe failed`, the next reload retries) while the process carries on. The registry's `AfterAdopt` hooks now run after every reload it applied, with an empty list for one that only rejected or removed a tenant, so a gone tenant's store closes on that reload rather than at the next adoption (and the shared keepalive wheel is recomputed on it too). The Pebble system gauges report the one instance's figures (`dedupe.Embedded.Stats`; `Stats` leaves the per-tenant `Deduplicator`, `Managed` and `Stores`), and `RegisterSystemMetrics` takes a stats function for them. `App.Close` releases the stores as one component — the last store's close closes the instance — inside the same 5s release budget. `dedupe.Managed` now opens its store through a function, so every backend gets the same switch semantics. `dedupe.Stores` edits its map under its lock and does the stores' I/O outside it, so one tenant's close never stalls another tenant's lookup. - **CORS is decided per tenant** (`internal/api/{router,tenant}.go`, `internal/app/{app,wire}.go`, `docs/src/content/docs/{deployment,api,architecture}.md`, `docs/src/content/docs/settings-directory.mdx`, `AGENTS.md`): story 10 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files. On a `/v1` route outside `/v1/ops/*` the response is decorated from the `cors.allowed_origins` of the tenant the request names (absent meaning tenant `0`), read per request off that tenant's adopted settings, so a reload of one tenant's folder changes its next response and nobody else's, and with the tenant tied to the URL no tenant's list widens another tenant's routes. The preflight is decided the same way — a browser sends no `X-Tenant-ID` on an `OPTIONS`, so over a nested directory the fronting proxy has to set it on the preflight as on every other request, and a preflight naming no tenant is tenant `0`'s; that proxy fixes the tenant from the host or path and overwrites whatever the client sent, since a browser caches a successful preflight by URL, not by header value, and a tenant a page could choose by header would let one tenant's cached preflight release writes at another. The tenant-exempt routes, which ignore the header, and a request naming a tenant that is not served (the pre-auth `400`/`404`/`503`) are answered from tenant `0`'s list, read through the registry rather than as last adopted, so with no tenant `0` being served — a nested directory without a `0` folder, or with one a reload rejected — they carry no CORS headers at all. `api.Dependencies.CORSOrigins` is store-keyed like the other per-tenant getters (`func(*settings.Store) []string`, `(*settings.Store).CORSOrigins` in production), and the tenant middleware's header read is shared with the CORS resolver (`requestTenant`). - **A nested settings directory serves one tenant per folder, failing closed per tenant** (`internal/settings/{tree,registry,store,validate,watch}.go` (`tree.go` new, + tests), `internal/api/{tenant,router,pipes,settings}.go`, `internal/app/{app,wire}.go`, `cmd/wavehouse/validate.go`, `clients/ts/src/{pipes,settings,types,index}.ts`, `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{deployment,api,architecture}.md`, `docs/src/content/docs/{settings-directory,reverse-proxy,access-control}.mdx`, `docs/src/content/docs/sdk/{admin,pipes,reference,streaming}.md`): story 2 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files — the same findings byte for byte, the same boot refusal, the same keep-previous on a rejected reload, the same watcher and `SIGHUP`, the same `wavehouse validate` exit codes. `settings.Validate` now reads the directory's shape off its entries — any of the four file names makes it flat, otherwise a folder makes it nested — and checks either one: each tenant folder goes through the per-directory checks (now `ValidateDir`), its name through the tenant-id grammar, and its findings carry the folder (`acme/policies.json`). The shapes never mix: a flat directory rejects a folder and a nested one rejects a loose file. `settings.Open` returns the `Registry`, which now owns `Reload`, the new `ReloadTenant`, the `AfterAdopt` hooks (handed the tenants a reload adopted), and the watcher; a `Store` is a passive holder of one tenant's document. A nested directory fails closed per tenant, at boot and on reload alike: a folder with an error finding stops its tenant being served — the tenant routes answer a bare `503 {"error": "tenant settings are invalid"}`, since they resolve before authentication and the findings quote the settings — while every other tenant carries on, with no previous-snapshot fallback; a request already admitted finishes on the document it started with, and the tenant keeps one `*Store` across the rejection. A whole-directory reload mirrors the folders (a new one is served, a removed one is a `404`), and a finding about the directory itself — a loose file, an unreadable directory, a changed shape — refuses boot and rejects a reload whole, leaving every tenant as it was. A nested directory gets no watcher; `SIGHUP` reloads the whole tree in both shapes. `POST /v1/ops/settings/reload` and the admin pipe reads (`GET /v1/ops/pipes[/{name}]`) take an optional `?tenant=`, parsed strictly — a query string that does not parse, or an empty, repeated, or malformed `tenant`, is a `400`, never a read of the default tenant — with `404` for an unknown tenant; absent means the whole directory on the reload and tenant `0` on the reads. The reload response keeps its shape: over a nested directory `adopted: false` with a `422` can mean adopted in part, the rejected folders being the ones with an error among their `findings`. Over a nested directory the `/v1/ops/*` gate admits the operator key alone — those routes reach every tenant — and an admin-role token gets `403`; `api.NewRouter` decides that from the registry's shape, whatever policy source was wired. Booting a nested directory with no `auth.operator_key` therefore leaves `SIGHUP` as the only reload, and boot warns about it; a whole-directory reload that drops a tenant names it in the log; and `Registry.Watch` refuses a nested directory itself. The resources a process still has one of (ClickHouse connection, dedupe store, MQ byte budget, auth verifier) follow the settings tenant `0` last adopted, their reload hooks running only when tenant `0` is adopted, so another tenant's reload never moves them and a `0` folder that a reload rejects or removes leaves all of them as they were; a nested directory without a `0` folder boots with them unconfigured (no ClickHouse address, `/livez` degraded) and says so once at boot, and the async paths (ingest worker, sweeper, stream hub, schema refresh) stay wired to tenant `0` until story 5. The SSE keepalive wheel is the one shared resource that weighs every tenant: it runs at the shortest `stream.keepalive_interval` among the tenants being served ([#597](https://github.com/Wave-RF/WaveHouse/issues/597) tracks honoring each tenant's own). The SDK's `wh.pipes.list()`, `wh.pipes.get()`, and `wh.settings.reload()` take a `tenant` option (the new `OpsRequestOptions`) sent as `?tenant=`, since `options.headers` cannot set a query parameter. @@ -75,7 +76,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Fixed -- **An insert invalidates a table's cached results under every tenant the directory holds** (`internal/app/wire.go` (+ tests), `internal/settings/registry.go` (+ tests), `AGENTS.md`, `docs/src/content/docs/{deployment,architecture,ingest-pipeline}.md`): until [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6 gives each tenant its own ClickHouse, every tenant reads the same tables, but the ingest worker — which writes every event as tenant `0`'s until story 5 — bumped only tenant `0`'s cache namespaces after an insert, so another tenant's cached query could answer stale rows for up to its TTL (an hour at most). The cache the worker invalidates through now fans each bumped namespace out to every tenant the registry knows (the new `Registry.Known`), the named one and a rejected one included — a rejected tenant comes back into service with the entries it has, so leaving it out would let a folder repaired inside a TTL serve pre-insert rows; reads are untouched, so a tenant is still never served another's cached rows. The residual is a folder removed and restored inside a TTL: the registry forgets a removed tenant, and what becomes of its cache is story 3's. Raised by CodeRabbit on #602. +- **An insert invalidates a table's cached results under every tenant the directory holds** (`internal/app/wire.go` (+ tests), `internal/settings/registry.go` (+ tests), `AGENTS.md`, `docs/src/content/docs/{deployment,architecture,ingest-pipeline}.md`): until [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6 gives each tenant its own ClickHouse, every tenant reads the same tables, but the ingest worker — which writes every event as tenant `0`'s until story 5 — bumped only tenant `0`'s cache namespaces after an insert, so another tenant's cached query could answer stale rows for up to its TTL (an hour at most). The cache the worker invalidates through now fans each bumped namespace out to every tenant the registry knows (the new `Registry.Known`), the named one and a rejected one included — a rejected tenant comes back into service with the entries it has, so leaving it out would let a folder repaired inside a TTL serve pre-insert rows; reads are untouched, so a tenant is still never served another's cached rows. The residual, a folder removed and restored inside a TTL, is closed since #610: a tenant back on a pool after an absence has its structured-query results orphaned at once (`Cache.InvalidateTenant`). Raised by CodeRabbit on #602. - **The `?token=` strip no longer repairs a query string that does not parse** (`internal/auth/auth.go`): `bearerToken` removed a query-string token by parsing the query, deleting `token`, and re-encoding what was left — and `url.ParseQuery` skips a pair it cannot read, so the re-encoding erased that pair. A handler that parses the query strictly in order to refuse a malformed one would then see a clean query: `GET /v1/ops/pipes?tenant=acme;x=1&token=…` would have answered `200` with the default tenant's pipes. The token is read exactly as before and a query that parses is rewritten exactly as before; a query that does not parse now loses its token pairs and nothing else, byte for byte. Pinned through `api.NewRouter` with the real authenticator, since a handler-level test never runs the middleware that rewrote the URL. - **SSE gap-fill re-reads the policy per replayed row** (`internal/stream/hub.go`): `ReplayProjector` captured the policy once when the replay began, so a policy adopted mid-fill — a revoked grant, say — applied only after the fill ended. It now reads it per event, as `Broadcast` does on the live path. diff --git a/clients/ts/src/stream/sse.ts b/clients/ts/src/stream/sse.ts index 350c1dd1f..e4efebe18 100644 --- a/clients/ts/src/stream/sse.ts +++ b/clients/ts/src/stream/sse.ts @@ -344,11 +344,14 @@ export class SSETransport> implements StreamTranspor // is transient even though the stream still ends (#469). Note WaveHouse // doesn't reject here *for authentication* — the endpoint is ungated and // answers a bad token with a reduced view, which is why `auth` is re-read - // per attempt (#239 tracks enforcing expiry). The one 4xx it raises itself - // is a 400 on this route for a missing or empty `table` - // (internal/api/stream.go); a 404/405 means the request missed the route - // entirely (router.go's chi handlers), and anything else comes from - // something in front — a gateway, or a proxy. + // per attempt (#239 tracks enforcing expiry). The 4xx it raises itself are + // a 400 for a missing or empty `table` (internal/api/stream.go) or a + // malformed `X-Tenant-ID`, and a 404 `unknown tenant` for a tenant it does + // not serve (internal/api/tenant.go) — never held, or removed, which is + // what a stream the server ended with its tenant reconnects into; a + // rejected tenant's 503 is retried like any 5xx. Any other 404/405 means + // the request missed the route entirely (router.go's chi handlers), and + // anything else comes from something in front — a gateway, or a proxy. return { liveMs: 0, terminal: !error.retryable }; } diff --git a/config.yaml b/config.yaml index 4be44b821..53a435029 100644 --- a/config.yaml +++ b/config.yaml @@ -2,14 +2,10 @@ # declare (a typo, or a tunable that moved to the settings directory) refuses # to boot and is named — whether it arrives as a YAML key here or as a WH_* # variable no key binds. -# Root for embedded state. NATS lives at /nats; each tenant's Pebble -# dedupe store (when its dedupe is enabled) at //dedupe — -# tenant 0's for a settings directory of the four files. An earlier layout's -# /pebble is moved there at boot when /0/dedupe is absent; -# with both present, boot uses /0/dedupe, leaves the old directory -# alone, and warns; a move that fails refuses boot. In a container, this MUST -# be on a host-backed volume — the relative default is for local binary use -# only. +# Root for embedded state. NATS lives at /nats; Pebble, holding every +# tenant's dedupe store while any tenant has dedupe enabled, at +# /pebble. In a container, this MUST be on a host-backed volume — +# the relative default is for local binary use only. data_dir: ./data server: diff --git a/deployments/Dockerfile b/deployments/Dockerfile index 816d6c2f1..bb74ecd7f 100644 --- a/deployments/Dockerfile +++ b/deployments/Dockerfile @@ -25,7 +25,7 @@ RUN --mount=type=cache,target=/go/pkg/mod \ CGO_ENABLED=0 go build -tags="${BUILD_TAGS}" -ldflags="-s -w" -o /bin/wavehouse ./cmd/wavehouse # Create the parent state and settings directories owned by the nonroot -# user (UID 65532 in distroless). The binary creates `nats/` and `/dedupe/` +# user (UID 65532 in distroless). The binary creates `nats/` and `pebble/` # subdirectories itself when NATS/Pebble open their stores — pre-creating # them here would buy nothing and obscures intent. Named-volume copy-up runs # against `/app/data` regardless; bind mounts mask the image dir entirely diff --git a/deployments/Dockerfile.goreleaser b/deployments/Dockerfile.goreleaser index 3d9ad8dfc..8d22201cb 100644 --- a/deployments/Dockerfile.goreleaser +++ b/deployments/Dockerfile.goreleaser @@ -4,7 +4,7 @@ # variant in lockstep with deployments/Dockerfile (source-build) so the two # images expose the same /app filesystem layout to operators. # -# Only the parents are pre-created — the binary mkdirs `nats/` and `/dedupe/` +# Only the parents are pre-created — the binary mkdirs `nats/` and `pebble/` # subdirs itself when NATS/Pebble open their stores. Named-volume copy-up # runs against `/app/data`; bind mounts mask the image dir entirely and # require the host directory to be writable by UID 65532. diff --git a/deployments/compose/standalone.yaml b/deployments/compose/standalone.yaml index 5c3a5bb75..d2fea9fee 100644 --- a/deployments/compose/standalone.yaml +++ b/deployments/compose/standalone.yaml @@ -17,9 +17,8 @@ services: ports: - "8080:8080" environment: - # NATS lives at /app/data/nats, each tenant's Pebble dedupe store at - # /app/data//dedupe (tenant 0's for the four-file settings - # directory) — subdirs are convention, not config. The Dockerfile + # NATS lives at /app/data/nats, Pebble (every tenant's dedupe store) at + # /app/data/pebble — subdirs are convention, not config. The Dockerfile # pre-creates /app/data owned by the nonroot user; bind-mount any # persistent volume here. WH_DATA_DIR: /app/data diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 119be82e9..75311090c 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -111,7 +111,7 @@ Status code: `503 Service Unavailable` The boot-degraded response lets an operator `curl /livez` to learn why the gateway isn't ready to serve traffic yet, instead of grepping a restart-loop log. The binary is bound on `:8080` and serves diagnostics, but is not yet accepting ingest/query traffic. Schema discovery retries with exponential backoff (2s → 60s); once a Refresh succeeds, `/livez` flips to `200` and stays there for the rest of the process lifetime — transient ClickHouse blips after that point are reflected in `/readyz`, not `/livez`. -Over a [nested settings directory](/deployment#the-nested-settings-directory) the probe reads every tenant together: `/livez` is `503` while **no** tenant has completed a first discovery — the diagnostic names the tenant whose attempt it reports (`schema discovery: tenant acme: …`), and reads `no tenant has completed a first discovery yet` before any attempt or when the directory serves no tenant — and `200` from the first tenant's success on, for the rest of the process lifetime. A tenant whose ClickHouse is unreachable after that is a log line and the `wavehouse_schema_refresh_failures_total{tenant}` counter, never a probe failure. A tenant that has not completed its own first discovery answers `503` (`schema not loaded yet`) on its schema-aware routes until it does; one whose ClickHouse goes down after that answers query errors, as a single-tenant server does. +Over a [nested settings directory](/deployment#the-nested-settings-directory) the probe reads every tenant together: `/livez` is `503` while **no** tenant has completed a first discovery — the diagnostic names the tenant whose attempt it reports (`schema discovery: tenant acme: …`), and reads `no tenant has completed a first discovery yet` before any attempt, when the directory serves no tenant, and once the tenant it named stops being served — and `200` from the first tenant's success on, for the rest of the process lifetime. A tenant whose ClickHouse is unreachable after that is a log line and the `wavehouse_schema_refresh_failures_total{tenant}` counter, never a probe failure. A tenant that has not completed its own first discovery answers `503` (`schema not loaded yet`) on its schema-aware routes until it does; one whose ClickHouse goes down after that answers query errors, as a single-tenant server does. --- @@ -152,7 +152,7 @@ Status code: `503 Service Unavailable` ### `GET /v1/health` — Liveness ping (public, content-free) -Returns **`200 OK` with an empty body** once the gateway is past boot, or **`503 Service Unavailable`** (also empty) while boot-time schema discovery is still failing. Like every `/v1` route outside `/v1/ops/*` it [resolves a tenant](/deployment#multi-tenant-deployments) first, so a malformed or unknown `X-Tenant-ID` answers `400`/`404` before the probe runs, and — over a [nested settings directory](/deployment#the-nested-settings-directory) — a tenant whose settings folder was rejected answers a `503` that carries the usual JSON error body rather than this route's empty one, as does a request carrying a token, with no valid operator key, while that tenant's [JWKS has not been fetched yet](#authentication) (`Retry-After: 30`; the SDK sends its token on this ping too). No authentication required and no response body — the caller only branches on the status code, so there's nothing to JSON-encode or cache per request. +Returns **`200 OK` with an empty body** once the gateway is past boot, or **`503 Service Unavailable`** (also empty) while boot-time schema discovery is still failing. Like every `/v1` route outside `/v1/ops/*` it [resolves a tenant](/deployment#multi-tenant-deployments) first, so a malformed or unknown `X-Tenant-ID` answers `400`/`404` before the probe runs, and — over a [nested settings directory](/deployment#the-nested-settings-directory) — a tenant whose settings folder was rejected answers a `503` that carries the usual JSON error body rather than this route's empty one, as does a request carrying a token, with no valid operator key, while that tenant's [JWKS has not been fetched yet](#authentication) (`Retry-After: 30`; the SDK sends its token on this ping too). No authentication required and no response body — the caller only branches on the status code, so there's nothing to JSON-encode or cache per request. Resolving the tenant makes the ping an unauthenticated answer to whether a tenant is served, deliberately: every tenant route gives an unknown tenant the same `404` before authenticating, since authenticating takes that tenant's own verifier, so the ping reveals nothing the others don't. It also answers a served tenant's ping from that tenant's own CORS list, where a tenant-exempt route answers from tenant `0`'s, which a nested directory need not have. This is what the SDK's `wh.sys.health()` calls, and the endpoint to use when choosing among multiple servers in a distributed setup. It mirrors `/livez` under the hood but is intentionally a `/v1` API route rather than a Kubernetes probe path: an operator may filter the bare probe paths (`/livez`, `/readyz`, `/healthz`) out at the reverse proxy since they're internal probes, so the SDK relies on `/v1/health`, which is documented public API surface meant to stay reachable. It does **not** ping ClickHouse — readiness-based load balancing is the proxy/LB's job (via `/readyz`), not the client's. @@ -612,7 +612,7 @@ Opens a persistent SSE connection for real-time event streaming. Supports histor | ------ | ----------- | | `Last-Event-ID` | RFC 3339 timestamp of the last received event. If present, overrides the `since` query parameter for automatic reconnection (standard `EventSource` behavior). | -**Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. +**Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back, while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. **Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event, the row slot is that column as published (`null` where the record omitted it, which ClickHouse replaces with the column's default on insert). @@ -873,7 +873,7 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the DLQ NATS stream (`WAVEHOUSE_DLQ`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format` (a pre-v2 message has no `format` field at all, which is how it presents here), or `columns` and `row` that do not pair — is parked without ever reaching a table batch, which is what an operator sees after upgrading across the wire change without draining first. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. +When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the DLQ NATS stream (`WAVEHOUSE_DLQ`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format` (a pre-v2 message has no `format` field at all, which is how it presents here), or `columns` and `row` that do not pair — is parked without ever reaching a table batch, which is what an operator sees after upgrading across the wire change without draining first. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Use `GET /v1/ops/dlq/stats` to monitor DLQ depth. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 0e1c64760..9d1dc6363 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -82,7 +82,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ingest.go** — Accepts `POST /v1/ingest?table={table}` in three body shapes: one flat JSON object, a JSON array of them, or NDJSON. The **required** `Content-Type` chooses the format *family* — `application/json` versus the four NDJSON spellings — and within the JSON family the body's first non-whitespace byte picks array versus single object; the bytes never choose the family. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type — a comma inside a *quoted* parameter value is data, so `application/json; a=", application/x-ndjson; b="` is accepted. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer and runs the per-format record readers over those bytes, so the `413` lands before any record is processed and peak memory per request is O(body) rather than O(record). Then it validates each record against the discovered schema, optional dedup, and publishes each row through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row missing the configured `id_field` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` (the Go-side type conversion lives in the structured-query / pipes path, not here). -- **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). +- **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` ahead of the cache, so nothing cached before is served. - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for the per-table parked counts (optionally one table) and the total. A dead-letter queue that does not exist (`mq.ErrNoDeadLetterQueue`) reads as empty; any other failure to read it is a 500. The queue itself is `internal/mq`'s. - **health.go** — Liveness (`/livez`), readiness (`/readyz`), and a content-free `Online` ping (`/v1/health`, the SDK's public liveness check); `/healthz` is a permanent alias of `/livez`, and `/health`/`/ready` are deprecated aliases. All three consult an optional `BootState` so they can return 503 while boot-time schema discovery is still failing in the retry loop (see `internal/app`; over a nested directory, while no tenant's has succeeded); once `BootState.Set(nil)` fires, `/livez` returns 200 and stays there. `/readyz` additionally runs a `Ping` each call — `chconn.Pools.Ping` in production: every open pool at once, ready at the first answer, every pool's error joined when none answers; `/v1/health` deliberately does not. @@ -90,14 +90,14 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one resource a process still has one of, the MQ byte budget, follows the default tenant: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`; the zero value when a nested directory has never served a tenant `0`, warned about once at boot), and `onDefaultAdopt` runs its hook only after a reload that adopted it, so another tenant's reload never moves it and a `0` folder that a reload rejects or removes leaves it as it was — like the one setting read per request that follows tenant `0`, the ops gate's admin role. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. Two settings are shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)); and the sweeper keeps the longest `stream.gap_window_minutes` (`longestGapWindow`, read every sweep), since the ingest queue is one stream and a purge is one bound over it — a stream per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 5b) gives each its own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over a factory that opens each tenant's embedded Pebble store at `data_dir//dedupe` whatever the directory's shape (the four files are tenant `0`; an earlier layout's `data_dir/pebble` is moved to tenant `0`'s once by `moveLegacyDedupeStore`, both present is left alone and warned about, and a failed move refuses boot) — the factory being the one place Pebble is named, so a shared backend behind the same switch is a wiring change — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its data left on disk when the tenant is switched off, rejected, or removed. A store that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for that tenant alone over a nested one. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. The `mq.max_bytes_gb` hook only hands the adopted budget to `mq.Broker.SetMaxBytes` under the App's stop context; how it is split across the streams, the time bounds, and the rollback are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one resource a process still has one of, the MQ byte budget, follows the default tenant: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`; the zero value when a nested directory has never served a tenant `0`, warned about once at boot), and `onDefaultAdopt` runs its hook only after a reload that adopted it, so another tenant's reload never moves it and a `0` folder that a reload rejects or removes leaves it as it was — like the one setting read per request that follows tenant `0`, the ops gate's admin role. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. Two settings are shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)); and the sweeper keeps the longest `stream.gap_window_minutes` (`longestGapWindow`, read every sweep), since the ingest queue is one stream and a purge is one bound over it — a stream per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 5b) gives each its own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. The `mq.max_bytes_gb` hook only hands the adopted budget to `mq.Broker.SetMaxBytes` under the App's stop context; how it is split across the streams, the time bounds, and the rollback are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)) lives next to the keepalive primitives it shares. One abstraction per file. -- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row (`ResolvedPermissions.RowVisible`, evaluated against the full event via the type-aware comparison seeded from the schema registry — `policy.ColumnSpec`: numeric columns compare numerically, `String` bytewise, `DateTime`/`DateTime64` as instants through the same parser ingest canonicalization uses (`discovery.Column.TimeParser` — one grammar, so filter constants and canonicalized payloads can't disagree on the instant), everything else admits byte-equality only and fails ordering/`!=` closed, so a missing schema can never downgrade the comparison to a leak). Each row withheld this way increments `wavehouse_sse_rows_withheld_total`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and caching the per-table column-kind lookup across the replay loop. -- **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and an `Evicted()` channel is the seam the slow-consumer follow-up closes to disconnect a wedged consumer. +- **hub.go** — `Hub`, the event fan-out. Subscribers register under `(mq.Topic, role)` — one tenant's table, so a subscriber never receives another tenant's rows for a table of the same name — and each event is evaluated under its own tenant's policy (the `PolicySource` read with the topic's tenant; a gap-fill and the opening schema frame read the connection's); `Broadcast` decodes each event once, applies each subscribed role's column policy once, builds one SSE frame per role, and fans it to every member of that role's `Bucket` — prepending a per-connection `event: schema` frame wherever that connection's announced column list has drifted, and withholding the row if the announcement cannot be queued — collapsing the prior per-subscriber `unmarshal → evaluate → filter → marshal` into one pass per distinct `(role, table)` output shape (the [#294](https://github.com/Wave-RF/WaveHouse/issues/294) lever; the measured ceiling was ~2 270 deliveries/s from re-projecting per subscriber). That schema-before-row guarantee is the LIVE path's: `ReplayProjector` tracks drift in its own state and the two are not reconciled ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). The column projection is claims-independent, so it is shared across a role's whole bucket; the role's row-level `filter` predicate is not — it is resolved against each subscriber's JWT claims, so for a role that carries a filter `Broadcast` keeps the shared column projection but delivers it only to the subscribers whose claims admit each row (`ResolvedPermissions.RowVisible`, evaluated against the full event via the type-aware comparison seeded from the schema registry — `policy.ColumnSpec`: numeric columns compare numerically, `String` bytewise, `DateTime`/`DateTime64` as instants through the same parser ingest canonicalization uses (`discovery.Column.TimeParser` — one grammar, so filter constants and canonicalized payloads can't disagree on the instant), everything else admits byte-equality only and fails ordering/`!=` closed, so a missing schema can never downgrade the comparison to a leak). Each row withheld this way increments `wavehouse_sse_rows_withheld_total`. This is the [#319](https://github.com/Wave-RF/WaveHouse/issues/319) fix that closes the query/stream row-level-security drift; roles without a filter keep the pure once-per-role fast path. `ReplayProjector` shares the same projection and per-connection row check for the handler's gap-fill, reading the policy per replayed event as `Broadcast` does, and caching the per-table column-kind lookup across the replay loop. `Prune(served)` evicts the subscribers of every tenant a reload stopped serving, removed or rejected alike, so their streams end rather than outlive the tenant with every row withheld. +- **subscriber.go** — `Subscriber`, the per-connection handle. It carries the connection's JWT claims, fixed at construction (`NewSubscriber(claims, metrics)`, no setter) — the claims the `Hub` resolves a role's row-level `filter` against, and immutability is what makes the fan-out's unsynchronized claims read race-free structurally. It owns a single ready-to-write outbound queue of `Frame`s (each tagged with its `kind`, so the handler labels the write where it happens): producers — the keepalive wheel and the event `Hub` — fan frames in with `Send` (non-blocking; a full queue drops, and `Send` itself counts the drop by frame kind, so no producer can forget to), and the handler drains `Frames()` to the client verbatim. The queue is sized for buffering live events (cap 64, up from the keepalive-only cap 1; #152 will make it a knob), and `Evict` closes its `Evicted()` channel, once, for the handler to end the stream: the `Hub`'s `Prune` does for a tenant no longer served, and the slow-consumer follow-up will for a wedged consumer. - **bucket.go** — `Bucket`, the reusable fan-out primitive: a concurrency-safe set of subscribers. `Push` fans one `Frame` to every member fire-and-forget — the keepalive wheel's ring is its only caller now that both `Hub` paths iterate `Snapshot`, since the schema announcement is per connection even where the projection is shared per role; `Snapshot` exposes the members so the event `Hub` can evaluate row visibility per subscriber before sending (drop counting lives in `Send` itself). The `Hub` holds one `Bucket` per `(topic, role)` so a projected frame is built once and sent to every member instead of re-projected per subscriber. - **heartbeat.go** — The keepalive wheel (`Heartbeater`). A single process-wide ticker fans a minimal `:` comment across the ring of `Bucket`s, waking ~1/N of live streams per tick so the writes don't synchronize. The effective per-connection keepalive period is `stream.keepalive_interval` in the settings directory (the wheel ticks every `keepalive_interval ÷ keepalive_buckets`, so one rotation spans the interval; a reload calls `Reconfigure`, which rebuilds the ring in place with every live subscriber carried over); the owning handler goroutine does the actual write, so the shared ticker never touches a `ResponseWriter` directly. - **metrics.go** — `Metrics`, the SSE instrument set: `wavehouse_sse_active_streams` (open streams), `wavehouse_sse_stream_duration_seconds` (lifetime), `wavehouse_sse_frames_sent_total` / `wavehouse_sse_bytes_sent_total` (labeled by `kind`: `keepalive`, `event`, `replay`, `schema`), `wavehouse_sse_dropped_frames_total` (frames dropped to a full subscriber queue — the slow-consumer signal that was silent before #294), and `wavehouse_sse_rows_withheld_total` (rows withheld from a subscriber by the row-level-security filter, labeled by table and role — the signal that separates "no matching rows" from "a fail-closed filter is withholding everything"). Nil-safe, so the handler holds one unconditionally and tests skip wiring it; one shared instance records the handler's write sites, each `Subscriber`'s queue-full drops (counted inside `Send`, by frame kind), and the `Hub`'s row-withheld counts. Separate from `observability.RegisterSystemMetrics`, which covers only the NATS/Pebble system gauges. Streams are observed through these metrics rather than per-event traces (the router excludes `/v1/stream` from the HTTP tracer). @@ -123,9 +123,9 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `dedupe/` — Deduplication (Optional) - **dedupe.go** — `Deduplicator` interface: `CheckAndMark(ctx, eventID) (bool, error)`. -- **embedded.go** — Uses [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store). Key = event ID, within one tenant's store: the tenant is the store, not part of the key. `Embedded(dir)` is the opener a `Managed` takes. +- **embedded.go** — `Embedded`, the [Pebble](https://github.com/cockroachdb/pebble) (embedded key-value store) implementation: every tenant's seen ids in one instance at `data_dir/pebble` ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 3), key = tenant id, a NUL, event id — no tenant id holds a NUL, so no two tenants' keys meet. `NewEmbedded(dataDir)` opens nothing; `Tenant(id)` is the `Factory` a `Stores` takes, building the tenant's `Managed` over its share of the instance, which opens with the first tenant store switched on and closes with the last one switched off. `Stats` reports the instance's figures for the system gauges, nil while it is closed. - **managed.go** — `Managed` wraps one store — opened through the function `NewManaged` takes, so the switch semantics are the same for every backend — behind the hot-reloadable `dedupe.enabled` switch: `Apply(enabled)` opens or closes it, idempotently, and in-flight `CheckAndMark` calls are serialized against the swap, so flipping the key is a reload, not a restart. `CheckAndMark` returns `ErrDisabled` while switched off (the ingest handler publishes un-deduped and counts it — a reload-window race, not a mode) and `ErrUnavailable` while switched on but not open (ingest fails closed). -- **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — the seam a shared backend slots into later by putting the tenant in the key, with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Stats()` sums the open stores for the system gauges; `Close()` closes every store. `internal/app` drives it from the registry's `AfterAdopt` hook. +- **stores.go** — `Stores` is one `Managed` per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7), built on first use through a `Factory` (`func(tenant.ID) *Managed`) — whether tenants share a backend is the factory's business (`Embedded.Tenant` puts them all in one Pebble instance), with nothing that holds the `Stores` changing. `For(id)` returns a tenant's store, built closed so a tenant adopted a moment ago answers `ErrDisabled` rather than having no store; `Retain(keep)` closes and forgets the stores of tenants no longer served, touching nothing on disk; `Close()` closes every store. `internal/app` drives it from the registry's `AfterAdopt` hook. ### `discovery/` — Schema Discovery & Validation @@ -136,7 +136,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `ingest/` — Ingest Pipeline, DLQ & Sweeping -- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line, with `columns` naming its positions (the table's insertable columns — a computed one cannot be named in an `INSERT`); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow`. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy. On a bulk-insert failure the batch is re-inserted row by row; rows that succeed are acked, and only the rows that fail again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. +- **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line, with `columns` naming its positions (the table's insertable columns — a computed one cannot be named in an `INSERT`); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow`. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy. On a bulk-insert failure the batch is re-inserted row by row — except a batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling), which no row could pass and `parkBatch` takes to the DLQ switch whole, logging once per batch rather than twice per row; rows that succeed are acked, and only the rows that fail again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. - **compact.go** — `EncodeCompactRow`, the positional row encoder every published row goes through, rendering one record over the table's **insertable** columns in declaration order. Serialization only: it validates nothing and judges no value. - **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: the longest `stream.gap_window_minutes` among the tenants being served — `internal/app`'s `longestGapWindow`). Finding the purge point is `internal/mq`'s (`purge.go`). @@ -154,7 +154,7 @@ The **only** package that imports NATS/JetStream — a `depguard` rule in `.gola - **provider.go** — `InitProvider(ctx, serviceName, ProviderConfig)` wires the OTel pipeline. Each output is independently gated; the W3C TraceContext + Baggage propagator is always installed (cheap, harmless when traces are off). Returns `(shutdown, promHandler http.Handler, err)` — `promHandler` is non-nil only when `PrometheusEnabled` is true and reads from a *private* `prometheus.Registry` to avoid leaking the process/Go collectors that `prometheus.DefaultRegisterer` auto-registers. OTLP-metrics push (`MetricsEnabled`) and Prometheus exposition (`PrometheusEnabled`) are independent: either, both, or neither may be set, and any combination produces a single MeterProvider feeding the active readers. The Endpoint field is only dialed by the OTLP exporters (traces / metrics-OTLP / logs); Prometheus-only operation leaves it untouched. Provider init in `internal/app` runs whenever `otel.enabled` OR `prometheus.enabled` is true, so Prometheus-only operation (Alloy/scrape, no collector) is a first-class mode. - **logger.go** — `NewLogger(component, level, isJSON, otlpSampleRate)` produces a slog logger that fans out to stdout (always 100%) and the OTLP log exporter (DEBUG/INFO sampled at `otlpSampleRate`, WARN/ERROR always 100% as a non-configurable safety floor). `TraceHandler` injects `trace_id`/`span_id` from the active span when one exists. `otlpSamplerFn` is exposed (lowercase) for unit testing the per-level rate logic without driving through the slogmulti middleware. -- **metrics.go** — `RegisterSystemMetrics(mqStats, pebbleStats)` registers observable gauges for embedded NATS connections, in-msgs, and Pebble dedupe storage stats. Both are functions read on every scrape — `mqStats` a `func() (MQStats, error)` (`mq.Broker.Stats` in production; nil skips the MQ gauges), `pebbleStats` a `func() map[string]int64` (`dedupe.Stores.Stats`, the figures summed across the tenants' stores; nil, or a nil map while no store is open, skips the Pebble gauges) — this package never holds the NATS server or a store. Wired in `internal/app` after the providers are up. +- **metrics.go** — `RegisterSystemMetrics(mqStats, pebbleStats)` registers observable gauges for embedded NATS connections, in-msgs, and Pebble dedupe storage stats. Both are functions read on every scrape — `mqStats` a `func() (MQStats, error)` (`mq.Broker.Stats` in production; nil skips the MQ gauges), `pebbleStats` a `func() map[string]int64` (`dedupe.Embedded.Stats`, the one instance's figures; nil, or a nil map while it is closed, skips the Pebble gauges) — this package never holds the NATS server or a store. Wired in `internal/app` after the providers are up. - **tracer.go** — W3C TraceContext propagation over message headers (`InjectHeaders` / `ExtractHeaders` on a plain `map[string][]string`, the shape NATS and HTTP headers share) — `internal/mq` injects on every publish and extracts onto the delivered `Message.Ctx` on the `Subscribe` path; no consumer reads it yet (the SSE hub bridge forwards the bytes and starts no span), and the ingest worker's `Consumer` path skips extraction entirely (see `embedded.go` above), so nothing downstream of the queue is linked to the originating request span. The package's design invariants — stdout always 100%, WARN+ERROR always export at 100%, gRPC exporters dial lazily so unreachable collectors never block startup, private Prometheus registry — are documented in AGENTS.md "Key Design Decisions" #15 and must be preserved by anything touching this package. @@ -184,13 +184,13 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi - **finding.go** — `Finding` / `Severity`: errors make the directory invalid, warnings don't block adoption. The JSON shape is part of the ops API (`POST /v1/ops/settings/reload` returns them). - **tree.go** — `Validate(root)` reads the directory's shape off its entries and checks either one: a root holding any of the four file names is flat and goes through `ValidateDir` untouched; otherwise a root holding a folder is nested, each folder name going through `tenant.Parse` and each folder through `ValidateDir`, with the folder leading every finding's `File` (`acme/policies.json`). The `Tree` it returns is nil when the finding is about the root itself — it cannot be listed, or a nested root holds a loose file or an entry that cannot be stat'ed. - **store.go** — `Store` is a passive holder: one tenant's adopted document behind an atomic pointer, swapped by the registry, which stamps it with the id of the tenant it created the store for (`Tenant()`, how a handler names its tenant to a per-tenant resource). Consumers read typed accessors per call (`ClickHouse()`, `Auth()`, `DedupeFor(table)`, `DLQFor(table)`, `Keepalive()`, …) rather than holding values. -- **registry.go** — `Registry` maps a tenant id to its `Store` and owns everything that changes one. `Open` validates and adopts at boot; `Reload` re-validates the whole directory and `ReloadTenant` one tenant's folder, serialized with each other; `AfterAdopt` hooks run after every reload the registry applied, with the tenants it adopted — none when it only rejected or removed one, which a consumer holding a resource per tenant needs to hear of too; `For(id)` and `All()` see only the tenants being served, `Known()` every tenant held, rejected ones included, and `Resolve(id)` tells a rejected tenant from an unknown one. The shape is fixed at `Open`. Flat: an invalid directory refuses boot, and a rejected reload keeps the previous snapshot. Nested: fail closed per tenant — a folder with an error finding stops being served (the store keeps its document for requests already admitted, and gets the next good one) while the rest carry on; a whole-directory reload mirrors the folders; and a finding about the directory itself refuses boot or rejects the reload whole, leaving every tenant as it was. The tenant map is replaced whole by a reload, so a lookup is one lock-free load. +- **registry.go** — `Registry` maps a tenant id to its `Store` and owns everything that changes one. `Open` validates and adopts at boot; `Reload` re-validates the whole directory and `ReloadTenant` one tenant's folder, serialized with each other; `AfterAdopt` hooks run after every reload the registry applied, with the tenants it adopted — none when it only rejected or removed one, which a consumer holding a resource per tenant needs to hear of too; `For(id)` and `All()` see only the tenants being served, `Known()` every tenant held, rejected ones included, and `Resolve(id)` tells a rejected tenant from an unknown one. The shape is fixed at `Open`. Flat: an invalid directory refuses boot, and a rejected reload keeps the previous snapshot. Nested: fail closed per tenant — a folder with an error finding stops being served (the store keeps its document for requests already admitted, and gets the next good one) while the rest carry on; a whole-directory reload mirrors the folders, down to none (an emptied directory is not a change of shape); and a finding about the directory itself refuses boot or rejects the reload whole, leaving every tenant as it was. The tenant map is replaced whole by a reload, so a lookup is one lock-free load. - **watch.go** — `Registry.Watch`, which `internal/app` starts for a flat directory only: fsnotify on the *directory* (not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost), debounced into one reload; reloads once as soon as the watch exists so an edit between the boot read and the watch is never missed. `SIGHUP` and the reload endpoint funnel through the same serialized `Reload`. - **seed.go** / **seed/** — The embedded (`go:embed`) starter directory with every key at its default. The binary carries no compiled defaults: `wavehouse bootstrap [dir]` writes this seed, and the compose stack and e2e fixture ship copies of it. ### `tenant/` — Tenant Identifier -- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes, and not `nats` or `pebble` in any letter case, the entries `data_dir` keeps for itself beside the tenants' own directories. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper folds over the tenants served; each served tenant has a schema registry of its own, built with its id (story 6). +- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper folds over the tenants served; each served tenant has a schema registry of its own, built with its id (story 6). ### `chconn/` — ClickHouse Connection Pools @@ -241,6 +241,8 @@ Ingest worker pipeline (StartIngestWorker): ClickHouse 26.5; see /ingest-pipeline for the basic-vs-best_effort divergence) → On success: DoubleAck messages → On failure: re-insert row by row; each row that fails again → DLQ output (dlq.{tenant}.{table}), then Ack to prevent infinite retry + (a batch whose tenant has no ClickHouse connection skips the row-by-row pass + and meets the DLQ switch whole — parkBatch) (Insert-only pipeline. The wire format `EventMessage` carries only {table_name, scope, received_timestamp, format, columns, row}; non-insert mutations diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index 44bc80222..a9c9de1db 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -35,7 +35,7 @@ This page is boot config only — what the platform operator owns (wiring, lifec | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | -| `data_dir` | `WH_DATA_DIR` | `./data` | Root directory for embedded state. NATS JetStream lives at `/nats`; each tenant's Pebble dedupe store (when its dedupe is enabled) at `//dedupe` — tenant `0`'s for [the four files](/settings-directory), one per folder of [a nested directory](/deployment#the-nested-settings-directory); an earlier layout's `/pebble` is moved to tenant `0`'s at boot when `/0/dedupe` is absent — with both present, boot uses `/0/dedupe`, leaves the old directory alone, and warns; a move that fails refuses boot rather than start an empty store. Subdirectory names are conventions, not config — one knob, one mount. **In a container this MUST resolve to a host-backed volume**; the relative default is for local binary use. WaveHouse logs a startup `WARN` when the directory is missing or empty (no prior state). See [Persistent Storage](/deployment#persistent-storage-required-for-containers). | +| `data_dir` | `WH_DATA_DIR` | `./data` | Root directory for embedded state. NATS JetStream lives at `/nats`; Pebble, holding every tenant's dedupe store while any tenant has dedupe enabled, at `/pebble`. Subdirectory names are conventions, not config — one knob, one mount. **In a container this MUST resolve to a host-backed volume**; the relative default is for local binary use. WaveHouse logs a startup `WARN` when the directory is missing or empty (no prior state). See [Persistent Storage](/deployment#persistent-storage-required-for-containers). | ### Server @@ -180,8 +180,7 @@ Every key, with its default. Save the YAML as `config.yaml` next to the binary ( ```yaml -data_dir: ./data # nats → ./data/nats, dedupe → ./data//dedupe - # (./data/0/dedupe for the four-file settings directory) +data_dir: ./data # nats → ./data/nats, pebble → ./data/pebble server: port: 8080 diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index b92c333cf..020c676c6 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -169,15 +169,15 @@ WH_SETTINGS_DIR=/etc/wavehouse/settings WaveHouse keeps all embedded state under a single configurable root, `WH_DATA_DIR` (yaml: `data_dir`). Subdirectories are convention, not config: - `/nats` — embedded NATS JetStream. Holds in-flight events between an ingest POST and the ingest worker → ClickHouse flush, plus the `stream.gap_window_minutes` window (settings directory) of history that powers SSE gap-fill across restarts. -- `//dedupe` — one Pebble dedup KV per tenant: `/0/dedupe` for [the four-file settings directory](/settings-directory), one per folder of [a nested one](#the-nested-settings-directory). Only used while that tenant's `dedupe.enabled` is `true` in its `config.json` (opened and closed on reload). An earlier layout kept the one store at `/pebble`; boot moves it to `/0/dedupe` once, seen ids included, when that directory is absent; with both present it uses `/0/dedupe`, leaves the old directory alone, and warns. A move that fails — a separately mounted or read-only `/pebble`, say — refuses boot, the failure being the boot error (`dedupe store relocation: …`), rather than start an empty store and let duplicates through in silence. +- `/pebble` — the Pebble dedup KV: one instance shared by every tenant, each key led by its tenant. Only used while some tenant's `dedupe.enabled` is `true` in its `config.json` (opened and closed on reload). -In a Docker / Podman / Kubernetes deployment, **`data_dir` must resolve to a host-backed volume**. The reference compose file `deployments/compose/standalone.yaml` sets `WH_DATA_DIR=/app/data` and binds a `wavehouse-data:/app/data` volume — copy that pattern. The bundled Dockerfiles pre-create `/app/data` and `/app/settings` owned by the nonroot user (UID 65532); the binary creates the `nats/` and `/dedupe/` subdirectories under `/app/data` itself on first run. +In a Docker / Podman / Kubernetes deployment, **`data_dir` must resolve to a host-backed volume**. The reference compose file `deployments/compose/standalone.yaml` sets `WH_DATA_DIR=/app/data` and binds a `wavehouse-data:/app/data` volume — copy that pattern. The bundled Dockerfiles pre-create `/app/data` and `/app/settings` owned by the nonroot user (UID 65532); the binary creates the `nats/` and `pebble/` subdirectories under `/app/data` itself on first run. If `data_dir` resolves into the container's writable overlay layer instead, **JetStream state is wiped on every restart**: in-flight events are lost, gap-fill stops bridging restarts, and disk usage accumulates inside `/var/lib/docker` instead of the volume the operator chose. Beyond persistence, the *speed* of that volume matters: JetStream `fsync`s every event to `/nats` before the ingest endpoint returns `200`, so the volume's `fsync` latency is your ingest latency floor. Managed cloud block storage handles this without thinking; commodity or virtualized substrates (ZFS without a SLOG, qcow2-on-`ext4`, spinning disks) can stall ingest with multi-second `fsync` tails. See [Durability & Storage](/durability) to measure yours before going live. -WaveHouse runs a simple existence check on startup and logs a `WARN` if `/nats` (or `/0/dedupe`, when tenant `0`'s dedupe is on) is missing or empty: +WaveHouse runs a simple existence check on startup and logs a `WARN` if `/nats` (or `/pebble`, when dedupe is on) is missing or empty: ```text wrap=false WARN data directory does not exist — starting with no prior state. @@ -185,7 +185,7 @@ WARN data directory does not exist — starting with no prior state. persisting; verify your mount. ``` -On a first-ever run this is expected. On every subsequent run it should be silent — so when this warning *does* fire after a redeploy, that's the most direct signal that the persistent volume isn't actually persisting. Another tenant's store is not checked this way: its first enable always starts it fresh, and a volume that did not persist shows in the NATS check. +On a first-ever run this is expected. On every subsequent run it should be silent — so when this warning *does* fire after a redeploy, that's the most direct signal that the persistent volume isn't actually persisting. ### Distroless Permission Traps (named volume vs bind mount) @@ -318,7 +318,7 @@ Until `startupProbe` succeeds, kubelet doesn't run `livenessProbe` or `readiness `SIGTERM` or `SIGINT` begins a graceful stop in three bounded phases whose budgets add up: 1. **Drain**, within [`server.shutdown_timeout`](/configuration#server) (default 10s). The listener stops accepting, every open [SSE stream](/api#get-v1stream--server-sent-events-stream) is ended at once, gap-fill in progress included (clients reconnect and resume from `Last-Event-ID`), and in-flight requests and the ingest worker's in-hand batches finish. A settings reload caught mid-hook gives up too. Whatever is still open at the deadline is force-closed. -2. **Release**, within a fixed 5s. The stores (embedded NATS, the Pebble dedupe stores — one per tenant with dedupe on, closed in turn as one step — the cache, ClickHouse) close; one still closing at the deadline is abandoned, the ones after it are left to the exit, and both are named in the log. +2. **Release**, within a fixed 5s. The stores (embedded NATS, the Pebble dedupe store, the cache, ClickHouse) close; one still closing at the deadline is abandoned, the ones after it are left to the exit, and both are named in the log. 3. **Flush**, within a fixed 3s. Telemetry is flushed last, on its own budget, so the lines the release logged reach the collector even when a close was slow. Only the drain scales with the deployment's workload, so it is the one operators tune; the other two are constants. @@ -345,7 +345,7 @@ X-Tenant-ID: 0 A request without the header, or with an empty one, resolves to tenant `0`, the default tenant. A settings directory that holds the four files itself defines that one tenant, so any other id is unknown; [a nested settings directory](#the-nested-settings-directory) defines one tenant per folder. Setting the header on every request is the client's or the fronting proxy's job; WaveHouse never derives it from the token. -A tenant id is 1–64 characters of ASCII letters, digits, `_`, and `-`. It is a string, not a number, so a long numeric id keeps every digit. `nats` and `pebble`, in any letter case, are not tenant ids: a tenant's own state lives at `/` ([Persistent Storage](#persistent-storage-required-for-containers)), and `data_dir` keeps those two names for itself. +A tenant id is 1–64 characters of ASCII letters, digits, `_`, and `-`. It is a string, not a number, so a long numeric id keeps every digit. | Status | Body | When | | ------ | ---- | ---- | @@ -381,17 +381,17 @@ settings/ That is the layout a control plane writes. Each folder's `clickhouse` block is its tenant's own ClickHouse, so a tenant answers queries once its first schema discovery against that ClickHouse succeeds (until then its schema-aware routes answer `503`, `schema not loaded yet`); what tenant `0`'s folder still supplies for the whole process — the message queue's budget, the token verifier of the routes that name no tenant, their CORS list — is listed under "What a tenant's folder decides", below. -The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe stores need no restructuring: the four files' store already lives at `/0/dedupe` (see [Persistent Storage](#persistent-storage-required-for-containers)), which is where a `0` folder's store goes. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id (`nats` and `pebble` are not, in any letter case: `data_dir` keeps those names for itself) is a finding of its own — that folder is skipped, and the rest of the directory still loads. +The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe store needs no restructuring: it keys every tenant's seen ids by tenant, and the four files are tenant `0`, as a `0` folder is. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id is a finding of its own — that folder is skipped, and the rest of the directory still loads. -**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. +**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. -**Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. +**Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. That is how a tenant is removed: delete its folder, then reload the whole directory. Its open streams end, its routes answer `404`, and its queued rows are parked on the DLQ under its own subject; nothing it stored is deleted, so restoring the folder restores the tenant, seen ids included. Reloading a deleted folder by name instead leaves its tenant rejected, answering `503`. The last folder can be removed the same way, with two catches, since `wavehouse validate` and boot both read an emptied directory as the four files missing: `validate` exits `1`, so a writer that gates each reload on it has to skip the check for that one reload, and a server restarted before a folder is written back refuses to boot. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. **The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` reads the queue the whole process shares and ignores the parameter. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store at `//dedupe`, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does — so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. The process still has one message queue, and its budget, `mq.max_bytes_gb`, follows tenant `0`'s folder. The queue is shared but addressed per tenant: an event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool too, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. Two settings weigh every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`; and the sweeper, which keeps the longest `stream.gap_window_minutes` among them, since every tenant's events share one message-queue stream and a purge is one bound over it. +**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. The process still has one message queue, and its budget, `mq.max_bytes_gb`, follows tenant `0`'s folder. The queue is shared but addressed per tenant: an event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool too, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. Two settings weigh every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`; and the sweeper, which keeps the longest `stream.gap_window_minutes` among them, since every tenant's events share one message-queue stream and a purge is one bound over it. -**What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. `mq.max_bytes_gb` stays as tenant `0` last adopted it; tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its data staying on disk for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its `GET /v1/stream` subscribers have every row withheld, since the hub reads no policy for it — the other tenants' events are untouched. The sweeper keeps the longest gap window among the tenants still served, so tenant `0`'s history is purged at theirs, and with no tenant left being served the window is zero, which purges the acknowledged history gap-fill replays. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. +**What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. `mq.max_bytes_gb` stays as tenant `0` last adopted it; tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. The sweeper keeps the longest gap window among the tenants still served, so tenant `0`'s history is purged at theirs, and with no tenant left being served the window is zero, which purges the acknowledged history gap-fill replays. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. ### Upgrading behind a proxy that already sends `X-Tenant-ID` @@ -446,7 +446,7 @@ Message-queue subjects now lead with the tenant: `ingest.{tenant}.{table}` and ` ## Dead Letter Queue (DLQ) -A failed batch insert is retried row by row; while the tenant's `dlq.enabled` is `true` for the table (the seed default — a hot-reloadable [settings directory](/settings-directory#dead-letter-queue) key, overridable per table), the rows that fail again are published to the `WAVEHOUSE_DLQ` NATS stream under subjects `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) instead of retrying forever. Monitor DLQ depth via `GET /v1/ops/dlq/stats`. +A failed batch insert is retried row by row; while the tenant's `dlq.enabled` is `true` for the table (the seed default — a hot-reloadable [settings directory](/settings-directory#dead-letter-queue) key, overridable per table), the rows that fail again are published to the `WAVEHOUSE_DLQ` NATS stream under subjects `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) instead of retrying forever. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for, such as by the connection ceiling — skips the row-by-row retry, which no row of it could pass: its tenant's switch is read once for the whole batch, and a tenant no longer served has no switch to read, so its batch is always parked. Monitor DLQ depth via `GET /v1/ops/dlq/stats`. ## Observability diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index dd90bd37d..870133cac 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -13,7 +13,7 @@ It is deliberately detailed: this is a hot, concurrency-heavy path, and the goro | File | Contents | | --- | --- | -| `worker.go` | `StartIngestWorker`, the `dispatchLoop`, `parseMsg` (+ `rejectPoison` for an envelope it cannot read), the per-tenant-table `tableBatcher`/`tableLoop`, `flushTable` (splits a batch per column list via `groupByColumns`) and `flushGroup` (bulk insert with a row-by-row poison-isolation fallback), `insertToClickHouse` (into the batch's tenant's ClickHouse, `chconn.Pools.Target`), `handleSuccess` (acks, after `invalidate` bumps the tenant's cache namespaces — under every tenant on the same ClickHouse address and database, through the cache `internal/app` hands the worker, since they read the same tables), `sendToDLQ`/`parkOnDLQ` | +| `worker.go` | `StartIngestWorker`, the `dispatchLoop`, `parseMsg` (+ `rejectPoison` for an envelope it cannot read), the per-tenant-table `tableBatcher`/`tableLoop`, `flushTable` (splits a batch per column list via `groupByColumns`, or hands the whole batch of a tenant with no ClickHouse connection to `parkBatch`) and `flushGroup` (bulk insert with a row-by-row poison-isolation fallback), `insertToClickHouse` (into the batch's tenant's ClickHouse, `chconn.Pools.Target`), `handleSuccess` (acks, after `invalidate` bumps the tenant's cache namespaces — under every tenant on the same ClickHouse address and database, through the cache `internal/app` hands the worker, since they read the same tables), `sendToDLQ`/`parkOnDLQ` | | `compact.go` | `EncodeCompactRow` — renders one record as a `JSONCompactEachRow` line over the table's **insertable** columns, in declaration order. Serialization only: it validates nothing and judges no value | | `sweeper.go` | The **Active Sweeper** — every minute, asks the MQ to purge the events that are both written to ClickHouse and past the SSE gap window (the purge arithmetic below lives in `internal/mq/purge.go`) | | `types.go` | `EventMessage` wire format and the `BufferConsumerName` constant | @@ -22,7 +22,7 @@ The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the q ## High-level shape -One process consumes a single durable JetStream consumer and fans events out to a goroutine per tenant table — the tenant is the subject's leading token. Each tenant's table batches independently and POSTs to ClickHouse over the HTTP interface (`JSONCompactEachRow`). On a bulk-insert failure the batch is re-inserted row by row, so a single poison row can't sink it: clean rows ack, and only the rows that fail again go to the dead-letter stream. An envelope the worker cannot *read* — malformed JSON, an unknown row `format` (what a pre-v2 message looks like), or columns and a row that don't pair — never reaches a table loop at all: `parseMsg` parks it on the same dead-letter stream, or, where the DLQ is off for the table, acks and drops it rather than redelivering a message that can never insert. A separate sweeper reclaims stream storage. +One process consumes a single durable JetStream consumer and fans events out to a goroutine per tenant table — the tenant is the subject's leading token. Each tenant's table batches independently and POSTs to ClickHouse over the HTTP interface (`JSONCompactEachRow`). On a bulk-insert failure the batch is re-inserted row by row, so a single poison row can't sink it: clean rows ack, and only the rows that fail again go to the dead-letter stream. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and meets the dead-letter switch once, whole; a tenant no longer served has no switch to read, so its batch is parked. An envelope the worker cannot *read* — malformed JSON, an unknown row `format` (what a pre-v2 message looks like), or columns and a row that don't pair — never reaches a table loop at all: `parseMsg` parks it on the same dead-letter stream, or, where the DLQ is off for the table, acks and drops it rather than redelivering a message that can never insert. A separate sweeper reclaims stream storage. ```mermaid flowchart LR diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 2a034bfa7..9555bf797 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -120,11 +120,11 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `clickhouse.max_idle_conns` | `5` | Idle native connections kept open (`>= 1`). | | `auth.jwks_url` | `""` | JWKS endpoint (absolute `http(s)` URL). When set, JWKS is the **sole** verifier and `jwt_secret` is ignored. See [Authentication](#authentication). | | `auth.role_claim` | `role` | Dot-separated JWT claim path the role is read from (e.g. `app_metadata.role`). | -| `dedupe.enabled` | `false` | Turn deduplication on; a reload opens or closes this tenant's Pebble store — see [Deduplication](#deduplication). | +| `dedupe.enabled` | `false` | Turn deduplication on; a reload opens or closes this tenant's store — see [Deduplication](#deduplication). | | `dedupe.id_field` | `event_id` | Dedup key field — see [Deduplication](#deduplication). | | `dedupe.require_id` | `false` | Reject rows missing the id field — see [Deduplication](#deduplication). | | `dedupe.tables.
.{id_field, require_id}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | -| `dlq.enabled` | `true` | Park rows that still fail after row-by-row isolation on the `WAVEHOUSE_DLQ` stream (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | +| `dlq.enabled` | `true` | Park poison rows — those that still fail after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the `WAVEHOUSE_DLQ` stream (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | | `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that a structured query's relative time range is truncated to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | | `query.default_max_rows` | `10000` | Fallback result `LIMIT` (`>= 1`) applied to a structured query when the caller and policy specify none. A result-**shaping** default, not a resource limit — server-wide limits (memory, rows scanned, execution time) belong in ClickHouse, see [Server-side resource limits](/configuration#server-side-resource-limits). | @@ -185,7 +185,7 @@ What stays in boot config is only what cannot change under a running process — Every dedupe knob lives here — there are no boot-config keys for it. The switch and its fields are resolved per record from one snapshot (table override → global value): -- `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens this tenant's embedded Pebble store at `//dedupe` — `/0/dedupe` for a directory of the four files — or closes it, so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`500 dedupe failed`) until the next reload or restart — the files asked for dedupe, so publishing un-deduped is not a fallback. At boot a failed open refuses to start, like every other store. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and leaves its data on disk for the folder that restores it; and a tenant whose store fails to open, at boot or on reload, fails closed alone — its ingest answers `500 dedupe failed` until a reload opens it — while every other tenant carries on. +- `dedupe.enabled` (seed default `false`) — turns deduplication on. Hot-reloadable: a reload that flips it opens or closes this tenant's store in the embedded Pebble instance at `/pebble`, so no restart is needed; seen ids persist across an off/on cycle. If the store fails to open on a reload, the failure is logged and ingest fails closed (`500 dedupe failed`) until the next reload or restart — the files asked for dedupe, so publishing un-deduped is not a fallback. At boot a failed open refuses to start, like every other store. A record that lands in the instant of the flip itself is published un-deduped: if the settings already say on but the store is not yet open, it's counted by `wavehouse_ingest_dedupe_disabled_total`; in the reverse case (settings already say off, store still open) the handler skips dedupe like any other disabled record and nothing is counted. That counter should only ever tick during a reload, so a steadily climbing rate means the store and the settings have come apart. Over [a nested directory](/deployment#the-nested-settings-directory) every tenant's seen ids live in that one instance, each key led by its tenant, and it is open while any tenant's switch is on: each tenant's store follows its own folder's `dedupe.enabled` the same way; a tenant's seen ids are never another's; a rejected or removed folder closes its tenant's store and keeps its seen ids for the folder that restores it; and if that instance fails to open, at boot or on reload, every tenant with dedupe on fails closed — its ingest answers `500 dedupe failed` until a reload opens it — while the tenants with dedupe off carry on. - `dedupe.id_field` (seed default `event_id`) — JSON field name in the ingest body used as the dedup key. - `dedupe.require_id` (seed default `false`) — controls what happens to a row missing `id_field` (which can't be deduped, so idempotency wouldn't apply to it). Such a row is always logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total`, in both modes. `false`: it is then published un-deduped. `true` rejects it instead (`400` for a single insert; a per-record failure in a batch) — a server-side tripwire for producers that must guarantee the id. - `dedupe.tables.
.{id_field, require_id}` — per-table overrides; each entry overrides only the fields it names and inherits the rest. @@ -210,11 +210,13 @@ The `auth` block is the verifier wiring, minus the secrets. `jwks_url` (absolute ## Dead Letter Queue -A failed batch insert is retried row by row; a row that fails again on its own is a poison row. `dlq.enabled` (seed default `true`) decides what happens to it, resolved per table (`dlq.tables.
.enabled` → global) at the moment of the failure, so a reload applies to the next poison row: +A failed batch insert is retried row by row; a row that fails again on its own is a poison row. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the retry, which no row of it could pass, and every row of it is a poison row. `dlq.enabled` (seed default `true`) decides what happens to it, resolved per table (`dlq.tables.
.enabled` → global) at the moment of the failure, so a reload applies to the next poison row: -- `true` — the row is published to the `WAVEHOUSE_DLQ` NATS stream under `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) with the ClickHouse error in its headers, and its original is acked. Inspect it with `GET /v1/ops/dlq/stats` (admin-only). +- `true` — the row is published to the `WAVEHOUSE_DLQ` NATS stream under `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) with the failure in its headers, and its original is acked. Inspect it with `GET /v1/ops/dlq/stats` (admin-only). - `false` — the row is left unacked, so NATS redelivers it and it retries until it inserts or the switch is flipped back. For every row the worker **can read**, nothing is ever dropped either way — the choice is *park it* versus *keep retrying*. **One exception, new in this release:** an envelope the worker cannot read *at all* — malformed JSON, an unknown `format` (what a pre-v2 in-flight message looks like), or `columns` and `row` that do not pair — can never insert, so redelivering it forever would wedge the consumer. With the DLQ off for the table it is acked and **dropped**, logged at `ERROR` and counted by `wavehouse_ingest_poison_total` with `disposition="dropped"` (also labeled by `table` and `reason`; an envelope parked on the DLQ carries `disposition="parked"`). See [Ingest Pipeline](/ingest-pipeline) — and drain the ingest queue before upgrading. +For a tenant no longer served — its folder removed or rejected — there is no switch to read: its rows are always parked, so none of them sits unacked in the shared ingest queue, where it would stop the [Active Sweeper](/ingest-pipeline#the-active-sweeper) purging it. + The `WAVEHOUSE_DLQ` stream always exists (an empty stream costs nothing) and the stats endpoint is always registered — the switch is purely behavioral, which is what makes it safe to reload. ## Message Queue diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index fde3e46de..2ae205e37 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -10,7 +10,6 @@ import ( "net/http" "net/http/httptest" "net/url" - "path/filepath" "strings" "testing" "testing/iotest" @@ -687,17 +686,15 @@ func TestIngest_Policy_CheckIn_AbsentClaim_FailsClosed(t *testing.T) { testutil.AssertJSONErrorResponse(t, w) } -// Two tenants, one event id: each tenant's store is its own (#583 story 7), -// so the id is first seen under both and a duplicate only within the tenant -// that sent it before — through a handler holding nothing but the request's +// Two tenants, one event id: each tenant's store is its own (#583 story 7) +// though every tenant's seen ids share one Pebble instance (story 3), so the +// id is first seen under both and a duplicate only within the tenant that +// sent it before — through a handler holding nothing but the request's // store, the way internal/app wires it. func TestIngest_DedupIsTheTenants(t *testing.T) { t.Parallel() tenants := nestedTenants(t, map[string]string{"acme": fullConfig(100), "globex": fullConfig(100)}) - root := t.TempDir() - stores := dedupe.NewStores(func(id tenant.ID) *dedupe.Managed { - return dedupe.NewManaged(dedupe.Embedded(filepath.Join(root, id.String(), "dedupe"))) - }) + stores := dedupe.NewStores(dedupe.NewEmbedded(t.TempDir()).Tenant) t.Cleanup(func() { _ = stores.Close() }) for id := range tenants.All() { require.NoError(t, stores.For(id).Apply(true)) diff --git a/internal/api/router.go b/internal/api/router.go index c40dbcb9c..488c9742c 100644 --- a/internal/api/router.go +++ b/internal/api/router.go @@ -144,6 +144,14 @@ func NewRouter(deps Dependencies) http.Handler { // paths at the reverse proxy. AuthMW runs but never rejects, so no // token is required and there's no authz gate. Mirrors /livez under // the hood (200 past boot, 503 while degraded), no body. + // + // It resolves a tenant like the routes beside it, deliberately + // (#583 story 3): its 404 tells a caller with no token that a + // tenant is not served, but so does every tenant route's, since all + // of them answer before authenticating, which needs the tenant's + // verifier. Exempt, it would take its CORS answer from tenant 0's + // list, which a nested directory need not have; resolved, a served + // tenant's ping answers from that tenant's own. r.Get("/health", deps.Health.Online) r.Post("/ingest", deps.Ingest.Handle) diff --git a/internal/api/stream.go b/internal/api/stream.go index 884b89736..eda5d410a 100644 --- a/internal/api/stream.go +++ b/internal/api/stream.go @@ -10,6 +10,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/stream" + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // StreamHandler handles GET /v1/stream @@ -24,6 +25,11 @@ type StreamHandler struct { // the client reconnects and gap-fills via Last-Event-ID. A nil channel // never fires (a harness that serves the handler itself). Closing <-chan struct{} + // Served, when set, reports whether a tenant is still being served. A + // reload that stops serving one evicts its streams (Hub.Prune), but not a + // stream TenantMW admitted just before the reload and registered with the + // Hub just after it: this check, made after registering, ends that one. + Served func(tenant.ID) bool } func NewStreamHandler(hub *stream.Hub, replayer mq.Replayer) *StreamHandler { @@ -113,6 +119,11 @@ func (h *StreamHandler) Handle(w http.ResponseWriter, r *http.Request) { h.Hub.Add(topic, role, sub) defer h.Hub.Remove(topic, role, sub) + // The registry stops serving a tenant before its hooks run, so a reload + // either finds this subscriber to evict or is seen here. + if h.Served != nil && !h.Served(topic.Tenant) { + return + } // Gap fill from the MQ's retained messages (DeliverByStartTime, see // mq.Replayer). @@ -142,7 +153,7 @@ func (h *StreamHandler) Handle(w http.ResponseWriter, r *http.Request) { } return true } - replayCtx, cancelReplay := h.replayContext(r) + replayCtx, cancelReplay := h.replayContext(r, sub) if ts, err := time.Parse(time.RFC3339Nano, sinceStr); err == nil && h.Replayer != nil { h.replay(replayCtx, ts, topic, sendReplay) } else if err != nil { @@ -168,8 +179,9 @@ func (h *StreamHandler) Handle(w http.ResponseWriter, r *http.Request) { case <-h.Closing: return case <-sub.Evicted(): - // Marked for disconnection (slow consumer). The client reconnects and - // gap-fills via Last-Event-ID. Inert until the slow-consumer follow-up. + // Its tenant is no longer served (Hub.Prune). The client reconnects + // into that tenant's 404 or 503, and gap-fills via Last-Event-ID once + // the tenant is served again. return case f := <-sub.Frames(): // One byte-pump for every frame kind: keepalive comments from the wheel and @@ -186,20 +198,21 @@ func (h *StreamHandler) Handle(w http.ResponseWriter, r *http.Request) { } // replayContext is the gap-fill's context: the request's, cancelled early -// when the server begins shutting down. Shutdown never cancels a request -// context itself, so without this the consumer creation — an MQ round trip -// made before the replay loop's first check — could hold the drain. -func (h *StreamHandler) replayContext(r *http.Request) (context.Context, context.CancelFunc) { +// when the server begins shutting down or sub is evicted. Neither cancels a +// request context itself, so without this the consumer creation — an MQ +// round trip made before the replay loop's first check — could hold the +// drain, and a long gap-fill would run on for a tenant no longer served. +func (h *StreamHandler) replayContext(r *http.Request, sub *stream.Subscriber) (context.Context, context.CancelFunc) { ctx, cancel := context.WithCancel(r.Context()) - if h.Closing != nil { - go func() { - select { - case <-h.Closing: - cancel() - case <-ctx.Done(): - } - }() - } + go func() { + select { + case <-h.Closing: + case <-sub.Evicted(): + case <-ctx.Done(): + return + } + cancel() + }() return ctx, cancel } diff --git a/internal/api/stream_test.go b/internal/api/stream_test.go index a8b152a3a..e09fa845c 100644 --- a/internal/api/stream_test.go +++ b/internal/api/stream_test.go @@ -177,3 +177,82 @@ func TestSSE_SubscribesUnderTheRequestTenant(t *testing.T) { wg.Wait() assert.Equal(t, 0, hub.Len(mq.Topic{Tenant: "acme", Table: "clicks"}), "every subscriber is removed") } + +// blockingReplayer is a gap-fill that never catches up: it signals once it +// has started, then holds until its context ends. +type blockingReplayer struct{ started chan struct{} } + +func (b blockingReplayer) ReplaySince(ctx context.Context, _ mq.Topic, _ time.Time, _ func([]byte) bool) error { + close(b.started) + <-ctx.Done() + return ctx.Err() +} + +// A stream ends on its own once its tenant stops being served — removed or +// rejected by a reload — whether it is idle, mid-gap-fill, or was admitted by +// TenantMW before the reload and registered with the hub after it pruned. +func TestSSE_EndsWhenItsTenantIsNoLongerServed(t *testing.T) { + t.Parallel() + topic := mq.Topic{Tenant: tenant.Default, Table: "clicks"} + served := func(tenant.ID) bool { return true } + unserved := func(tenant.ID) bool { return false } + // handle runs the stream in the background; the request is never + // cancelled, so the handler returns only by ending the stream itself. + handle := func(t *testing.T, h *StreamHandler, lastEventID string) (<-chan struct{}, *httptest.ResponseRecorder) { + t.Helper() + ctx, cancel := context.WithCancel(context.Background()) + t.Cleanup(cancel) // only a stream that failed to end is still open here + req := httptest.NewRequestWithContext(ctx, http.MethodGet, "/v1/stream?table=clicks", nil) + if lastEventID != "" { + req.Header.Set("Last-Event-ID", lastEventID) + } + w := httptest.NewRecorder() + done := make(chan struct{}) + go func() { + defer close(done) + h.Handle(w, withTenant(req)) + }() + return done, w + } + ended := func(t *testing.T, done <-chan struct{}) { + t.Helper() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Fatal("the stream outlived its tenant") + } + } + + t.Run("idle", func(t *testing.T) { + t.Parallel() + hub := stream.NewHub(nil, nil, nil) + done, _ := handle(t, &StreamHandler{Hub: hub, Served: served}, "") + require.Eventually(t, func() bool { return hub.Len(topic) == 1 }, 5*time.Second, 5*time.Millisecond) + hub.Prune(unserved) + ended(t, done) + assert.Zero(t, hub.Len(topic)) + }) + t.Run("mid gap-fill", func(t *testing.T) { + t.Parallel() + hub := stream.NewHub(nil, nil, nil) + replayer := blockingReplayer{started: make(chan struct{})} + done, _ := handle(t, &StreamHandler{Hub: hub, Replayer: replayer, Served: served}, "2026-09-24T00:00:00Z") + select { + case <-replayer.started: + case <-time.After(5 * time.Second): + t.Fatal("the gap-fill never started") + } + hub.Prune(unserved) + ended(t, done) + }) + t.Run("no longer served when it registers", func(t *testing.T) { + t.Parallel() + hub := stream.NewHub(nil, nil, nil) + hb := stream.NewHeartbeater(time.Hour, 1) + done, w := handle(t, &StreamHandler{Hub: hub, Heartbeater: hb, Served: unserved}, "") + ended(t, done) + assert.Zero(t, hub.Len(topic), "the deferred Remove ran") + assert.Zero(t, hb.Len(), "it never reached the keepalive wheel") + assert.Contains(t, w.Body.String(), ": connected", "admitted, then ended") + }) +} diff --git a/internal/app/app.go b/internal/app/app.go index 33bf7dfc4..f853a76b3 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -105,8 +105,10 @@ type App struct { pools *chconn.Pools bootState *api.BootState discoveries *discoveries - // dedup is one store per tenant, each following its own folder's switch. + // dedup is one store per tenant, each following its own folder's switch, + // and dedupeStats the figures of the one Pebble instance they share. dedup *dedupe.Stores + dedupeStats func() map[string]int64 mq mq.Broker cache cache.Cache sseMetrics *stream.Metrics diff --git a/internal/app/app_test.go b/internal/app/app_test.go index ad1d503fd..fc5d27553 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -212,8 +212,8 @@ func TestNew_DedupeFollowsSettings(t *testing.T) { cfg := testConfig(t, dir) a := newApp(t, cfg, Options{}) assert.Equal(t, tt.enabled, a.dedup.For(tenant.Default).Open()) - _, err := os.Stat(filepath.Join(cfg.DataDir, "0", "dedupe")) - assert.Equal(t, tt.enabled, err == nil, "tenant 0's directory exists iff dedupe is on: the four files are tenant 0") + _, err := os.Stat(filepath.Join(cfg.DataDir, "pebble")) + assert.Equal(t, tt.enabled, err == nil, "the Pebble instance exists iff dedupe is on: a server with dedupe off opens nothing") }) } } @@ -466,26 +466,26 @@ func TestReload_NestedHooksFollowTheDefaultTenant(t *testing.T) { assert.Equal(t, "*", allowOrigin("/v1/health", "acme")) } -// One dedupe store per tenant over a nested directory (#583 story 7), rooted -// at data_dir//dedupe and following that tenant's own switch: opened -// by its folder's adoption, closed — the directory left as it is — once the -// folder is rejected or removed, and reopened over the same seen ids when -// the folder is back. Close releases every open store. +// One dedupe store per tenant over a nested directory (#583 story 7), each +// following its own tenant's switch, and every one a share of the one Pebble +// instance at data_dir/pebble (story 3): opened by its folder's adoption, +// closed — its seen ids kept — once the folder is rejected or removed, and +// reopened over the same seen ids when the folder is back. The instance is +// open while some tenant's store is, and Close releases it. func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": nil, "broken": invalidQuery}) cfg := testConfig(t, root) a := newApp(t, cfg, Options{}) ctx := t.Context() - dir := func(id string) string { return filepath.Join(cfg.DataDir, id, "dedupe") } acme, globex := a.dedup.For("acme"), a.dedup.For("globex") assert.True(t, acme.Open(), "acme's switch is on") - assert.DirExists(t, dir("acme")) assert.False(t, globex.Open(), "globex's is off") - assert.NoDirExists(t, dir("globex"), "a closed store creates nothing") - assert.NoDirExists(t, dir("broken"), "nor does a rejected tenant") - assert.NoDirExists(t, filepath.Join(cfg.DataDir, legacyDedupeDir), "the earlier layout's directory is never created") + assert.DirExists(t, filepath.Join(cfg.DataDir, "pebble"), "one instance for every tenant") + for _, id := range []string{"acme", "globex", "broken"} { + assert.NoDirExists(t, filepath.Join(cfg.DataDir, id), "and no directory of a tenant's own") + } dup, err := acme.CheckAndMark(ctx, "e1") require.NoError(t, err) assert.False(t, dup) @@ -493,7 +493,6 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { rewriteSettings(t, filepath.Join(root, "globex"), dedupeOn) a.tenants.Reload("test") assert.True(t, globex.Open(), "globex's reload opens globex's store") - assert.DirExists(t, dir("globex")) dup, err = globex.CheckAndMark(ctx, "e1") require.NoError(t, err) assert.False(t, dup, "an id acme has seen is new to globex") @@ -507,14 +506,16 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { assert.False(t, globex.Open()) assert.True(t, acme.Open()) - // A removed folder: the store closes, the directory stays as it is. + // A removed folder closes its store; with none left open, the instance + // closes too, its files staying where they are. require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) a.tenants.Reload("test") _, known = a.tenants.Resolve("acme") require.False(t, known) assert.False(t, acme.Open(), "a tenant the registry no longer holds has its store closed") - entries, err := os.ReadDir(dir("acme")) - require.NoError(t, err, "and its directory untouched") + assert.Nil(t, a.dedupeStats(), "no store open: the instance is closed") + entries, err := os.ReadDir(filepath.Join(cfg.DataDir, "pebble")) + require.NoError(t, err) assert.NotEmpty(t, entries) // Restoring the folder restores the tenant, seen ids included. @@ -530,99 +531,39 @@ func TestNew_NestedDedupeStoreFollowsEachTenant(t *testing.T) { assert.False(t, restored.Open(), "Close releases every open store") } -// seedLegacyStore writes a Pebble store at dir with ids seen, the way an -// earlier layout — or an older binary rolled back to — leaves one. -func seedLegacyStore(t *testing.T, dir string, ids ...string) { - t.Helper() - require.NoError(t, os.MkdirAll(filepath.Dir(dir), 0o750)) - d, err := dedupe.NewEmbedded(dir) - require.NoError(t, err) - for _, id := range ids { - _, err := d.CheckAndMark(t.Context(), id) - require.NoError(t, err) - } - require.NoError(t, d.Close()) -} - -// An earlier layout's data_dir/pebble is tenant 0's store: boot moves it to -// data_dir/0/dedupe once, whatever the switch says, so a standalone -// deployment keeps its seen ids across the upgrade. Both directories present -// — an older binary ran in between — leaves both, the new one in use. -func TestNew_MovesTheLegacyDedupeStore(t *testing.T) { - dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} - seen := func(t *testing.T, a *App, id string) bool { - t.Helper() - dup, err := a.dedup.For(tenant.Default).CheckAndMark(t.Context(), id) - require.NoError(t, err) - return dup - } - t.Run("seen ids survive the move", func(t *testing.T) { - cfg := testConfig(t, writeSettings(t, dedupeOn)) - seedLegacyStore(t, filepath.Join(cfg.DataDir, legacyDedupeDir), "e1") - a := newApp(t, cfg, Options{}) - assert.NoDirExists(t, filepath.Join(cfg.DataDir, legacyDedupeDir)) - assert.DirExists(t, filepath.Join(cfg.DataDir, "0", "dedupe")) - assert.True(t, seen(t, a, "e1"), "an id the old store had seen is still a duplicate") - assert.False(t, seen(t, a, "e2")) - }) - t.Run("moved even with dedupe off", func(t *testing.T) { - cfg := testConfig(t, writeSettings(t, nil)) - seedLegacyStore(t, filepath.Join(cfg.DataDir, legacyDedupeDir), "e1") - a := newApp(t, cfg, Options{}) - assert.NoDirExists(t, filepath.Join(cfg.DataDir, legacyDedupeDir)) - assert.DirExists(t, filepath.Join(cfg.DataDir, "0", "dedupe")) - assert.False(t, a.dedup.For(tenant.Default).Open(), "moved, not opened: the switch is off") - }) - t.Run("both present: the new one is in use, the old is left", func(t *testing.T) { - cfg := testConfig(t, writeSettings(t, dedupeOn)) - seedLegacyStore(t, filepath.Join(cfg.DataDir, legacyDedupeDir), "old") - seedLegacyStore(t, filepath.Join(cfg.DataDir, "0", "dedupe"), "new") - a := newApp(t, cfg, Options{}) - assert.DirExists(t, filepath.Join(cfg.DataDir, legacyDedupeDir), "deleting data is never boot's call") - assert.True(t, seen(t, a, "new")) - assert.False(t, seen(t, a, "old"), "the old store's ids are not merged in") - }) - t.Run("a failed move refuses boot", func(t *testing.T) { - guardGlobals(t) - cfg := testConfig(t, writeSettings(t, dedupeOn)) - seedLegacyStore(t, filepath.Join(cfg.DataDir, legacyDedupeDir), "e1") - // Tenant 0's directory cannot be created under a regular file. - require.NoError(t, os.WriteFile(filepath.Join(cfg.DataDir, "0"), nil, 0o600)) - _, err := New(t.Context(), Options{Config: cfg}) - require.ErrorContains(t, err, "dedupe store relocation") - }) -} - -// A store that cannot open follows the registry's own rule for the shape: a -// flat directory refuses boot, like every other store, and a nested one -// fails closed per tenant — that tenant's ingest answers 500 until a reload -// or a restart opens it, and every other tenant carries on. +// A Pebble instance that cannot open follows the registry's own rule for the +// shape: a flat directory refuses boot, like every other store, and a nested +// one fails closed for every tenant with dedupe on, since they share the +// instance — their ingest answers 500 until a reload or a restart opens it — +// while the process, and every tenant with dedupe off, carries on. func TestNew_DedupeOpenFailure(t *testing.T) { dedupeOn := map[string]any{"dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}} - // A regular file where the store's directory should be is what Pebble + // A regular file where the instance's directory should be is what Pebble // refuses to open. - block := func(t *testing.T, path string) { + block := func(t *testing.T, dataDir string) { t.Helper() - require.NoError(t, os.MkdirAll(filepath.Dir(path), 0o750)) - require.NoError(t, os.WriteFile(path, nil, 0o600)) + require.NoError(t, os.WriteFile(filepath.Join(dataDir, "pebble"), nil, 0o600)) } t.Run("flat refuses boot", func(t *testing.T) { guardGlobals(t) cfg := testConfig(t, writeSettings(t, dedupeOn)) - block(t, filepath.Join(cfg.DataDir, "0", "dedupe")) + block(t, cfg.DataDir) _, err := New(t.Context(), Options{Config: cfg}) require.ErrorContains(t, err, "dedupe open") }) - t.Run("nested fails closed per tenant", func(t *testing.T) { - root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": dedupeOn}) + t.Run("nested fails closed", func(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": dedupeOn, "globex": dedupeOn, "initech": nil}) cfg := testConfig(t, root) - block(t, filepath.Join(cfg.DataDir, "acme", "dedupe")) + block(t, cfg.DataDir) a := newApp(t, cfg, Options{}) - acme := a.dedup.For("acme") - assert.False(t, acme.Open()) - _, err := acme.CheckAndMark(t.Context(), "e1") - require.ErrorIs(t, err, dedupe.ErrUnavailable, "switched on but not open: that tenant's ingest fails closed") - assert.True(t, a.dedup.For("globex").Open(), "the tenant beside it is served") + for _, id := range []tenant.ID{"acme", "globex"} { + store := a.dedup.For(id) + assert.False(t, store.Open()) + _, err := store.CheckAndMark(t.Context(), "e1") + require.ErrorIs(t, err, dedupe.ErrUnavailable, "%s: switched on but not open, so its ingest fails closed", id) + } + _, err := a.dedup.For("initech").CheckAndMark(t.Context(), "e1") + require.ErrorIs(t, err, dedupe.ErrDisabled, "a tenant with dedupe off is as it would be anyway") }) } @@ -1170,33 +1111,121 @@ func TestClose_AbandonsAStuckCloseAtTheDeadline(t *testing.T) { assert.NotContains(t, err.Error(), "fine") } -func TestRun_StopEndsOpenStreams(t *testing.T) { - var lc net.ListenConfig - ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") - require.NoError(t, err) - cfg := testConfig(t, writeSettings(t, nil)) - a := newApp(t, cfg, Options{Listener: ln}) - baseURL, stop := runApp(t, a, ln) - +// openStream opens GET /v1/stream?table=events for tenant id ("" sends no +// header) and returns its body once the ": connected" preamble arrives. +func openStream(t *testing.T, baseURL, id string) io.Reader { + t.Helper() req, err := http.NewRequestWithContext(t.Context(), http.MethodGet, baseURL+"/v1/stream?table=events", nil) require.NoError(t, err) + if id != "" { + req.Header.Set(tenant.Header, id) + } resp, err := http.DefaultClient.Do(req) require.NoError(t, err) - defer func() { _ = resp.Body.Close() }() + t.Cleanup(func() { _ = resp.Body.Close() }) require.Equal(t, http.StatusOK, resp.StatusCode) buf := make([]byte, 64) n, err := resp.Body.Read(buf) require.NoError(t, err) require.Contains(t, string(buf[:n]), ": connected", "the stream is open") + return resp.Body +} - // A stream is a connection to close, not work to drain: the stop ends it - // at once rather than waiting out server.shutdown_timeout and then - // force-closing it anyway. - started := time.Now() - assert.NoError(t, stop()) - assert.Less(t, time.Since(started), time.Second, "the open stream held the stop for the drain budget") - _, err = io.ReadAll(resp.Body) - assert.NoError(t, err, "the server ended the stream cleanly") +// endsCleanly fails unless the server ends the stream within a second, and +// with the clean end of the response rather than a broken connection. +func endsCleanly(t *testing.T, stream io.Reader) { + t.Helper() + read := make(chan error, 1) + go func() { + _, err := io.ReadAll(stream) + read <- err + }() + select { + case err := <-read: + assert.NoError(t, err, "the server ended the stream cleanly") + case <-time.After(time.Second): + t.Fatal("the stream is still open") + } +} + +// A stream is a connection to close, not work to drain: the stop ends every +// open one at once rather than waiting out server.shutdown_timeout. A reload +// that stops serving a tenant — its folder rejected or removed — ends that +// tenant's streams the same way, and no other tenant's; the client's +// reconnect then meets the tenant's 503 or 404. A flat directory never stops +// serving tenant 0, so a reload it rejects leaves the stream open. +func TestRun_StopEndsOpenStreams(t *testing.T) { + start := func(t *testing.T, settingsDir string) (a *App, baseURL string, stop func() error) { + t.Helper() + var lc net.ListenConfig + ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + a = newApp(t, testConfig(t, settingsDir), Options{Listener: ln}) + baseURL, stop = runApp(t, a, ln) + return a, baseURL, stop + } + // registered waits for tenant id's stream to join the hub: openStream + // returns at the ": connected" preamble, which the handler writes first. + registered := func(t *testing.T, a *App, id tenant.ID) { + t.Helper() + topic := mq.Topic{Tenant: id, Table: "events"} + require.Eventually(t, func() bool { return a.hub.Len(topic) == 1 }, 5*time.Second, 5*time.Millisecond) + } + + t.Run("the stop", func(t *testing.T) { + _, baseURL, stop := start(t, writeSettings(t, nil)) + resp := openStream(t, baseURL, "") + started := time.Now() + assert.NoError(t, stop()) + assert.Less(t, time.Since(started), time.Second, "the open stream held the stop for the drain budget") + endsCleanly(t, resp) + }) + + t.Run("a reload that stops serving the tenant", func(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) + a, baseURL, stop := start(t, root) + defer func() { assert.NoError(t, stop()) }() + reconnect := func(id string) int { + req, err := http.NewRequestWithContext(t.Context(), http.MethodGet, baseURL+"/v1/stream?table=events", nil) + require.NoError(t, err) + req.Header.Set(tenant.Header, id) + resp, err := http.DefaultClient.Do(req) + require.NoError(t, err) + _ = resp.Body.Close() + return resp.StatusCode + } + acme, globex := openStream(t, baseURL, "acme"), openStream(t, baseURL, "globex") + registered(t, a, "acme") + registered(t, a, "globex") + + rewriteSettings(t, filepath.Join(root, "globex"), invalidQuery) + _, adopted, known := a.tenants.ReloadTenant("globex", "test") + require.True(t, known) + require.False(t, adopted) + endsCleanly(t, globex) + assert.Equal(t, 1, a.hub.Len(mq.Topic{Tenant: "acme", Table: "events"}), "acme's stream stays open") + assert.Equal(t, http.StatusServiceUnavailable, reconnect("globex"), "rejected: retried until its folder is fixed") + + require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) + a.tenants.Reload("test") + endsCleanly(t, acme) + assert.Equal(t, http.StatusNotFound, reconnect("acme"), "removed: the client stops") + }) + + t.Run("a reload the flat directory rejects", func(t *testing.T) { + dir := writeSettings(t, nil) + a, baseURL, stop := start(t, dir) + resp := openStream(t, baseURL, "") + registered(t, a, tenant.Default) + rewriteSettings(t, dir, invalidQuery) + _, adopted := a.tenants.Reload("test") + require.False(t, adopted) + // An evicted stream leaves the hub only once its handler returns. + assert.Never(t, func() bool { return a.hub.Len(mq.Topic{Tenant: tenant.Default, Table: "events"}) == 0 }, + 200*time.Millisecond, 10*time.Millisecond, "tenant 0 keeps its previous settings, and its stream") + assert.NoError(t, stop()) + endsCleanly(t, resp) + }) } // poolSettings is a config.json patch: the seed's clickhouse block pointed @@ -1356,15 +1385,44 @@ func TestReload_CeilingRefusesAThirdTupleThenOpensIt(t *testing.T) { } // A tenant the registry stops serving — its folder rejected, then removed — -// releases its pool and its schema registry; the tenant beside it keeps -// both; restoring the folder restores both. +// releases what it held: its pool, its schema registry and the loop +// refreshing it, its verifier, and its dedupe store, its seen ids kept. Once +// removed its routes answer 404, the ingest worker is handed no ClickHouse +// and a DLQ switch that reads on for it — its queued rows are parked — and a +// /livez diagnostic naming it goes back to the no-tenant line; the tenant +// beside it keeps its own. Restoring the folder restores the tenant over a +// fresh pool, registry and verifier, and an id it sent before the removal is +// still a duplicate. Its open streams end too: TestRun_StopEndsOpenStreams. func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { - root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) + jwks, _, fetches := jwksServer(t, "acme-1") + acmeSettings := authPatch(jwks.URL) + acmeSettings["dedupe"] = map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}} + root := writeNestedSettings(t, map[string]map[string]any{"acme": acmeSettings, "globex": nil}) a := newApp(t, testConfig(t, root), Options{}) - acme, acmeRegistry := a.pools.For("acme"), a.discoveries.For("acme") + acme, acmeRegistry, acmeDedup := a.pools.For("acme"), a.discoveries.For("acme"), a.dedup.For("acme") require.NotNil(t, acme) require.NotNil(t, acmeRegistry) require.NotNil(t, a.pools.For("globex")) + loops := *a.discoveries.cur.Load() + stopped := func(id tenant.ID) bool { + select { + case <-loops[id].done: + return true + case <-time.After(5 * time.Second): + return false + } + } + pipe := func(id string) string { + req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/pipes/nope", nil) + req.Header.Set(tenant.Header, id) + rec := httptest.NewRecorder() + a.Handler().ServeHTTP(rec, req) + return fmt.Sprintf("%d %s", rec.Code, rec.Body.String()) + } + dup, err := acmeDedup.CheckAndMark(t.Context(), "e1") + require.NoError(t, err) + require.False(t, dup) + require.Eventually(t, func() bool { return fetches.Load() > 0 }, 5*time.Second, 10*time.Millisecond, "acme's key set is fetched off the boot path") rewriteSettings(t, filepath.Join(root, "globex"), invalidQuery) _, adopted, known := a.tenants.ReloadTenant("globex", "test") @@ -1372,20 +1430,42 @@ func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { require.False(t, adopted) assert.Nil(t, a.pools.For("globex"), "a rejected tenant is on no pool") assert.Nil(t, a.discoveries.For("globex"), "and has no registry") + assert.True(t, stopped("globex"), "nor a loop refreshing one") assert.Same(t, acme, a.pools.For("acme")) assert.Same(t, acmeRegistry, a.discoveries.For("acme")) - // Adopted in part from here on: globex's folder stays rejected. + // Adopted in part from here on: globex's folder stays rejected. No + // tenant has completed a first discovery, and the diagnostic names acme. + a.discoveries.onAttempt("acme", errors.New("connection refused")) + require.Contains(t, get(t, a.Handler(), "/livez").Body.String(), "tenant acme") require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) a.tenants.Reload("test") assert.Nil(t, a.pools.For("acme")) assert.Nil(t, a.discoveries.For("acme")) - - require.NoError(t, os.Rename(writeSettings(t, nil), filepath.Join(root, "acme"))) + assert.True(t, stopped("acme")) + // A stopped loop's last attempt can land after the reload: its line must + // not come back. + a.discoveries.onAttempt("acme", errors.New("connection refused")) + assert.False(t, acmeDedup.Open(), "its dedupe store is closed") + assert.Contains(t, pipe("acme"), "404 {\"error\":\"unknown tenant: acme\"}") + assert.Empty(t, a.pools.Target("acme").URL, "the worker has no ClickHouse to insert its queued rows into") + assert.True(t, dlqFor(a.tenants)("acme", "events"), "and parks them") + livez := get(t, a.Handler(), "/livez") + assert.Equal(t, http.StatusServiceUnavailable, livez.Code) + assert.Contains(t, livez.Body.String(), "no tenant has completed a first discovery yet", "the diagnostic went with its tenant") + + fetched := fetches.Load() + require.NoError(t, os.Rename(writeSettings(t, acmeSettings), filepath.Join(root, "acme"))) a.tenants.Reload("test") + assert.Contains(t, pipe("acme"), "pipe not found", "served again") assert.NotNil(t, a.pools.For("acme")) - assert.NotNil(t, a.discoveries.For("acme"), "back, over a fresh registry") - assert.NotSame(t, acmeRegistry, a.discoveries.For("acme")) + assert.NotSame(t, acme, a.pools.For("acme"), "over a fresh pool") + assert.NotNil(t, a.discoveries.For("acme")) + assert.NotSame(t, acmeRegistry, a.discoveries.For("acme"), "and a fresh registry") + assert.Eventually(t, func() bool { return fetches.Load() > fetched }, 5*time.Second, 10*time.Millisecond, "and a fresh verifier, fetching the key set again") + dup, err = a.dedup.For("acme").CheckAndMark(t.Context(), "e1") + require.NoError(t, err) + assert.True(t, dup, "an id acme sent before the removal is still a duplicate") } // Over a nested directory the probes read every tenant together: /livez is diff --git a/internal/app/wire.go b/internal/app/wire.go index 313209ba2..009f9920e 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -153,13 +153,22 @@ func longestGapWindow(tenants *settings.Registry) time.Duration { return window } +// served reports whether the registry is serving tenant id: what the +// per-tenant resources — verifiers, dedupe stores, open streams — are pruned +// by once a reload removes or rejects their tenant. +func (a *App) served(id tenant.ID) bool { + _, ok := a.tenants.For(id) + return ok +} + // perTenant adapts a store accessor to the tenant-keyed getter the async // paths take: they hold a tenant id — the one each message's topic names // for the stream hub and the ingest worker (#583 story 5) — not a request's -// resolved store. A -// miss — a nested directory with no 0 folder, or with a rejected or removed -// one — is logged and read as T's zero value; what a removed tenant means to -// each async path is story 3's to decide. +// resolved store. A miss — a tenant no longer served, or a 0 a nested +// directory does not hold — is logged and read as T's zero value. By then a +// tenant a reload removed or rejected has had its streams ended (Hub.Prune) +// and its schema loop stopped (discoveries), so a miss is an event still in +// flight; the ingest worker reads its DLQ switch through dlqFor instead. func perTenant[T any](tenants *settings.Registry, get func(*settings.Store) T) func(tenant.ID) T { return func(id tenant.ID) T { store, ok := tenants.For(id) @@ -176,6 +185,10 @@ func perTenant[T any](tenants *settings.Registry, get func(*settings.Store) T) f // miss reads as DLQ on, not as the zero value perTenant would give: off lets // the worker drop a message it cannot read, and not knowing the tenant is no // reason to destroy its row. Parked, it survives until the tenant resolves. +// So a removed or rejected tenant's queued rows are parked under its own +// subject rather than left unacked for its return: an unacked row holds the +// ack floor, the sweeper stops purging, and the one shared stream fills +// toward mq.max_bytes_gb until every tenant's ingest answers 503. func dlqFor(tenants *settings.Registry) func(tenant.ID, string) bool { return func(id tenant.ID, table string) bool { store, ok := tenants.For(id) @@ -382,7 +395,9 @@ func queryTimeout(s *settings.Store) time.Duration { return s.ClickHouse().Query // refresh loop of its own (discoveries), and the boot state /livez reports: // 503 with the latest discovery failure while no tenant has completed a // first discovery, then 200 for the rest of the process lifetime — with one -// tenant, the rule there always was. Non-fatal either way. A flat +// tenant, the rule there always was. A failure goes with its tenant: once the +// tenant it names is no longer served, the diagnostic is the no-tenant one +// again. Non-fatal either way. A flat // directory's tenant 0 is refreshed synchronously here, as before, so the // port binds with the state known; a failure marks the binary degraded and // leaves the retry (backoff 2s → 60s) to its loop. A nested directory's @@ -402,7 +417,10 @@ func (a *App) wireDiscovery(ctx context.Context) { var ( mu sync.Mutex loaded bool + // failing is the tenant the degraded diagnostic names. + failing tenant.ID ) + noTenantLoaded := errors.New("schema discovery: no tenant has completed a first discovery yet") diagnostic := func(id tenant.ID, err error) error { if nested { return fmt.Errorf("schema discovery: tenant %s: %w", id, err) @@ -417,7 +435,10 @@ func (a *App) wireDiscovery(ctx context.Context) { slog.Warn("schema discovery retry failed", "tenant", id, "error", err) mu.Lock() defer mu.Unlock() - if !loaded { + // A loop a reload stopped may report one last attempt after its + // tenant has gone. + if !loaded && a.served(id) { + failing = id a.bootState.Set(diagnostic(id, err)) } }, @@ -434,7 +455,7 @@ func (a *App) wireDiscovery(ctx context.Context) { }) a.discoveries = d if nested { - a.bootState.Set(errors.New("schema discovery: no tenant has completed a first discovery yet")) + a.bootState.Set(noTenantLoaded) d.reconcile(a.tenants) } else { // A flat registry always serves tenant 0: Open refused boot otherwise. @@ -448,53 +469,43 @@ func (a *App) wireDiscovery(ctx context.Context) { } d.adopt(tenant.Default, reg) } - a.tenants.AfterAdopt(func([]tenant.ID) { d.reconcile(a.tenants) }) + a.tenants.AfterAdopt(func([]tenant.ID) { + d.reconcile(a.tenants) + mu.Lock() + defer mu.Unlock() + if !loaded && failing != "" && !a.served(failing) { + failing = "" + a.bootState.Set(noTenantLoaded) + } + }) a.add(component{name: "schema discovery", close: d.close}) } -// legacyDedupeDir is where the one store lived before #583 story 7 gave each -// tenant its own: tenant 0's, implicitly. tenant.Parse reserves the name, as -// it does the queue's nats, in any letter case, so no tenant's directory is -// ever this one — not on a case-insensitive filesystem either. -const legacyDedupeDir = "pebble" - -// wireDedupe builds the dedupe stores: one per tenant (#583 story 7), at -// data_dir//dedupe whatever the settings directory's shape — the -// four files are tenant 0 — each following its own tenant's hot-reloadable -// dedupe.enabled. Which store that is, is the factory's business alone; the -// embedded Pebble one is what a process chooses here, and an earlier -// layout's data_dir/pebble is moved to tenant 0's directory once -// (moveLegacyDedupeStore). One reconcile closure sets every store to what -// the registry says: open exactly when its tenant is served with the switch -// on, closed — its data left on disk — when the tenant is switched off, -// rejected, or removed. It is registered as the after-adopt hook BEFORE the -// boot apply (Apply is idempotent), so a reload landing between the two -// can't leave a tenant's settings saying "on" with its store still closed — -// either the hook sees it or the boot apply reads it. A failed open follows -// the registry's own rule for the shape: flat refuses boot, like every other -// store, and on reload logs and leaves the store closed — ingest then fails -// closed (500 "dedupe failed") rather than silently publishing un-deduped, -// since the files asked for dedupe; nested fails closed per tenant the same -// way at boot too, the next reload retrying, so one tenant's unopenable -// store never costs the others their process. +// wireDedupe builds the dedupe stores: one per tenant (#583 story 7), each +// following its own tenant's hot-reloadable dedupe.enabled, over the +// embedded Pebble implementation, which is handed data_dir and decides the +// rest: every tenant's seen ids in one instance there, open while any +// tenant's store is (dedupe.Embedded). One reconcile closure sets every +// store to what the registry says: open exactly when its tenant is served +// with the switch on, closed — its seen ids kept — when the tenant is +// switched off, rejected, or removed. It is registered as the after-adopt +// hook BEFORE the boot apply (Apply is idempotent), so a reload landing +// between the two can't leave a tenant's settings saying "on" with its store +// still closed — either the hook sees it or the boot apply reads it. An +// instance that cannot open follows the registry's own rule for the shape: +// flat refuses boot, like every other store, and on reload logs and leaves +// the store closed — ingest then fails closed (500 "dedupe failed") rather +// than silently publishing un-deduped, since the files asked for dedupe; +// nested fails closed the same way at boot too, for every tenant with +// dedupe on, the next reload retrying, so it never costs the process. func (a *App) wireDedupe() error { nested := a.tenants.Nested() - dir := func(id tenant.ID) string { return filepath.Join(a.cfg.DataDir, id.String(), "dedupe") } - if err := moveLegacyDedupeStore(a.cfg.DataDir, dir(tenant.Default)); err != nil { - return fmt.Errorf("dedupe store relocation: %w", err) - } - stores := dedupe.NewStores(func(id tenant.ID) *dedupe.Managed { return dedupe.NewManaged(dedupe.Embedded(dir(id))) }) - a.dedup = stores + embedded := dedupe.NewEmbedded(a.cfg.DataDir) + stores := dedupe.NewStores(embedded.Tenant) + a.dedup, a.dedupeStats = stores, embedded.Stats a.add(component{name: "dedupe", close: withoutContext(stores.Close)}) reconcile := func() error { - // The gone tenants' stores close first, so a tenant renamed only in - // letter case — one directory to a case-insensitive filesystem — - // never has both spellings open at once. - served := func(id tenant.ID) bool { - _, ok := a.tenants.For(id) - return ok - } - if err := stores.Retain(served); err != nil { + if err := stores.Retain(a.served); err != nil { slog.Error("dedupe store close failed", "error", err) } var errs []error @@ -502,15 +513,17 @@ func (a *App) wireDedupe() error { m := stores.For(id) enabled := store.DedupeEnabled() wasOpen := m.Open() - // Tenant 0's store is the one an earlier deployment had, so a - // fresh directory there is a first run or a lost volume. Another - // tenant's first enable always starts fresh, and a lost volume - // shows in the NATS check. - if enabled && !wasOpen && id == tenant.Default { - config.WarnIfFreshDataDir("pebble", dir(id)) + // The instance opens with the first store switched on: a fresh + // directory then is a first run or a lost volume. + if enabled && !embedded.Open() && len(errs) == 0 { + config.WarnIfFreshDataDir("pebble", embedded.Dir()) } if err := m.Apply(enabled); err != nil { - config.LogStorageInitError("dedupe", dir(id), err) + // The stores share the one instance, so a failure is every + // store's: logged once, not once per tenant. + if len(errs) == 0 { + config.LogStorageInitError("dedupe", embedded.Dir(), err) + } errs = append(errs, fmt.Errorf("tenant %s: %w", id, err)) continue } @@ -527,36 +540,6 @@ func (a *App) wireDedupe() error { return nil } -// moveLegacyDedupeStore moves an earlier layout's data_dir/pebble — tenant -// 0's store, implicitly — to dst, tenant 0's directory, once: a standalone -// deployment keeps its seen ids across the upgrade, and later across the -// move to a nested directory with an explicit 0 folder. Both present (an -// older binary ran in between and started a new store at the old path) is -// left alone and warned about: dst is the one in use, and deleting data is -// never boot's call. A failed move refuses boot rather than open an empty -// store and let duplicates through in silence. -func moveLegacyDedupeStore(dataDir, dst string) error { - src := filepath.Join(dataDir, legacyDedupeDir) - if _, err := os.Stat(src); err != nil { - if errors.Is(err, os.ErrNotExist) { - return nil - } - return err - } - if _, err := os.Stat(dst); err == nil { - slog.Warn("dedupe store already moved to tenant 0's directory; the old directory is unused and can be removed", "old", src, "path", dst) - return nil - } - if err := os.MkdirAll(filepath.Dir(dst), 0o750); err != nil { - return err - } - if err := os.Rename(src, dst); err != nil { - return err - } - slog.Info("dedupe store moved to tenant 0's directory", "old", src, "path", dst) - return nil -} - // wireMQ starts the MQ — the embedded NATS under data_dir/nats, the one // place the implementation is chosen; everything after it sees mq.Broker. // mq.max_bytes_gb is hot-reloadable: after each adoption the new budget is @@ -579,7 +562,7 @@ func (a *App) wireMQ() error { // provider and RegisterCallback silently no-ops, making this look // authoritative when it's actually doing nothing. if a.cfg.OTel.Enabled || a.cfg.Prometheus.Enabled { - if err := observability.RegisterSystemMetrics(broker.Stats, a.dedup.Stats); err != nil { + if err := observability.RegisterSystemMetrics(broker.Stats, a.dedupeStats); err != nil { slog.Error("failed to register system metrics", "error", err) } } @@ -629,9 +612,13 @@ func (a *App) wireSweeper() { // (drop counts) and the stream handler (write counts); the Hub that // projects/serializes each event once per (topic, role) and pushes it to // that role's subscribers; the MQ → Hub bridge; and the keepalive wheel. +// After every reload the Hub ends the open streams of each tenant no longer +// served, removed or rejected alike (Hub.Prune); the client reconnects into +// that tenant's 404 or 503 and gap-fills once it is served again. func (a *App) wireStreaming() { a.sseMetrics = stream.NewMetrics() a.hub = stream.NewHub(perTenant(a.tenants, (*settings.Store).Policy), a.discoveries.For, a.sseMetrics) + a.tenants.AfterAdopt(func([]tenant.ID) { a.hub.Prune(a.served) }) // Hub bridge: MQ → broadcast to connected SSE clients. The Hub decodes and // projects each event itself (skipping malformed payloads) under the @@ -776,10 +763,7 @@ func (a *App) wireAuth() func(http.Handler) http.Handler { authn.Reconfigure(id, wiring(store)) } } - authn.Prune(func(id tenant.ID) bool { - _, served := a.tenants.For(id) - return served - }) + authn.Prune(a.served) }) return authn.Middleware() } @@ -788,10 +772,10 @@ func (a *App) wireAuth() func(http.Handler) http.Handler { // triggers (these two and POST /v1/ops/settings/reload) funnel into the same // serialized Registry.Reload, and a rejected reload keeps the previous good // snapshot. They only start in Run, after New has registered every -// AfterAdopt hook (ClickHouse reconnect, dedupe stores, keepalive wheel, auth -// verifiers): the watcher reloads once as soon as its watch exists, and that -// reload must already drive every hook — a hook registered after the first -// reload could miss it. +// AfterAdopt hook (ClickHouse reconnect, dedupe stores, keepalive wheel, open +// streams, auth verifiers): the watcher reloads once as soon as its watch +// exists, and that reload must already drive every hook — a hook registered +// after the first reload could miss it. // // A nested directory gets no watcher (#583): whoever writes a tenant's // folder calls the reload route once the folder is complete, where a watcher @@ -857,6 +841,7 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { streamHandler := api.NewStreamHandler(a.hub, a.mq) streamHandler.Metrics = a.sseMetrics streamHandler.Heartbeater = a.heartbeater + streamHandler.Served = a.served // Closed when the API server begins shutting down, ending every open // stream at once (see serve). closing := make(chan struct{}) diff --git a/internal/config/config.go b/internal/config/config.go index a09530667..68b0314b6 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -11,11 +11,8 @@ import ( // Config is the top-level application configuration. type Config struct { // DataDir is the root for embedded state. NATS JetStream lives at - // `/nats`; each tenant's Pebble dedupe store (when its dedupe - // is enabled) at `//dedupe` — tenant 0's for the four - // files. An earlier layout's `/pebble` is moved there at boot - // when `/0/dedupe` is absent; with both present, boot uses the - // new one, leaves the old one alone, and warns; a failed move refuses boot. + // `/nats`; Pebble, holding every tenant's dedupe store while any + // tenant has dedupe enabled, at `/pebble`. // Subdirectory names are conventions, not config — one knob, one mount. // In a container this MUST resolve to a host-backed volume; the relative // `./data` default is fine for local binary use only. diff --git a/internal/dedupe/dedupe.go b/internal/dedupe/dedupe.go index 015942201..c9fdb7f46 100644 --- a/internal/dedupe/dedupe.go +++ b/internal/dedupe/dedupe.go @@ -8,8 +8,6 @@ type Deduplicator interface { // If not seen, it atomically marks the event as seen. CheckAndMark(ctx context.Context, eventID string) (isDuplicate bool, err error) - Stats() map[string]int64 - // Close releases resources held by the deduplicator. Close() error } diff --git a/internal/dedupe/embedded.go b/internal/dedupe/embedded.go index cd62ef1b6..6b3b5efaf 100644 --- a/internal/dedupe/embedded.go +++ b/internal/dedupe/embedded.go @@ -5,42 +5,122 @@ import ( "encoding/binary" "errors" "math" + "path/filepath" + "sync" "time" "github.com/cockroachdb/pebble" + + "github.com/Wave-RF/WaveHouse/internal/tenant" ) -// EmbeddedDeduplicator uses Pebble for local deduplication. -type EmbeddedDeduplicator struct { - db *pebble.DB +// Embedded is the embedded implementation: every tenant's seen ids in one +// Pebble instance at data_dir/pebble, each key led by its tenant (#583 story +// 3), so a thousand tenants cost one instance's goroutines, open files and +// heap rather than a thousand. The instance opens with the first tenant's +// store switched on and closes with the last one switched off: it is open +// exactly while some tenant has dedupe on, and a tenant switched off, +// rejected or removed keeps its seen ids for when it is back. +type Embedded struct { + dir string + + mu sync.Mutex // guards db and open + db *pebble.DB + open int // tenant stores open over db } -// NewEmbedded opens a Pebble database at the given path. -func NewEmbedded(dir string) (*EmbeddedDeduplicator, error) { - db, err := pebble.Open(dir, &pebble.Options{}) - if err != nil { - return nil, err - } - return &EmbeddedDeduplicator{db: db}, nil +// NewEmbedded returns the embedded implementation under dataDir. Nothing is +// opened until a tenant's store is. +func NewEmbedded(dataDir string) *Embedded { + return &Embedded{dir: filepath.Join(dataDir, "pebble")} +} + +// Dir is where the instance lives. +func (e *Embedded) Dir() string { return e.dir } + +// Open reports whether the instance is open: whether some tenant's store is. +func (e *Embedded) Open() bool { + e.mu.Lock() + defer e.mu.Unlock() + return e.db != nil +} + +// keySeparator ends the tenant at the front of every key. A tenant id has no +// NUL, so the first one in a key is this one, and no two tenants' keys meet. +const keySeparator = 0 + +// Tenant builds tenant id's store, closed, over its share of the instance — +// the Factory Stores takes. +func (e *Embedded) Tenant(id tenant.ID) *Managed { + prefix := append([]byte(id), keySeparator) + return NewManaged(func() (Deduplicator, error) { return e.acquire(prefix) }) } -// Embedded returns the opener of the Pebble store at dir, for NewManaged: -// the one place the wiring names Pebble. -func Embedded(dir string) func() (Deduplicator, error) { - return func() (Deduplicator, error) { - d, err := NewEmbedded(dir) +// acquire opens a tenant's store, and the instance with it when no other +// tenant's store holds it open. +func (e *Embedded) acquire(prefix []byte) (Deduplicator, error) { + e.mu.Lock() + defer e.mu.Unlock() + if e.db == nil { + db, err := pebble.Open(e.dir, &pebble.Options{}) if err != nil { return nil, err } - return d, nil + e.db = db } + e.open++ + return &tenantStore{e: e, db: e.db, prefix: prefix}, nil +} + +// release closes a tenant's store, and the instance with it when that was +// the last store open. +func (e *Embedded) release() error { + e.mu.Lock() + defer e.mu.Unlock() + e.open-- + if e.open > 0 { + return nil + } + err := e.db.Close() + e.db = nil + return err +} + +// Stats reports the instance's figures for the system gauges — one set, +// however many tenants' stores are open — or nil while it is closed, which +// the metrics scraper skips. +func (e *Embedded) Stats() map[string]int64 { + e.mu.Lock() + defer e.mu.Unlock() + if e.db == nil { + return nil + } + m := e.db.Metrics() + walSize := m.WAL.Size + if walSize > math.MaxInt64 { + walSize = math.MaxInt64 + } + return map[string]int64{ + "pebble_wal_size": int64(walSize), + "pebble_table_count": m.Total().NumFiles, + } +} + +// tenantStore is one tenant's store: its keys in the shared instance, which +// stays open while the store does. +type tenantStore struct { + e *Embedded + db *pebble.DB + prefix []byte + closed sync.Once } // CheckAndMark returns true if the event was already seen. -func (d *EmbeddedDeduplicator) CheckAndMark(_ context.Context, eventID string) (bool, error) { - key := []byte(eventID) +func (s *tenantStore) CheckAndMark(_ context.Context, eventID string) (bool, error) { + key := make([]byte, 0, len(s.prefix)+len(eventID)) + key = append(append(key, s.prefix...), eventID...) - _, closer, err := d.db.Get(key) + _, closer, err := s.db.Get(key) if err == nil { _ = closer.Close() return true, nil @@ -53,24 +133,16 @@ func (d *EmbeddedDeduplicator) CheckAndMark(_ context.Context, eventID string) ( val := make([]byte, 8) binary.BigEndian.PutUint64(val, uint64(time.Now().UnixNano())) - if err := d.db.Set(key, val, pebble.Sync); err != nil { + if err := s.db.Set(key, val, pebble.Sync); err != nil { return false, err } return false, nil } -func (d *EmbeddedDeduplicator) Close() error { - return d.db.Close() -} - -func (d *EmbeddedDeduplicator) Stats() map[string]int64 { - m := d.db.Metrics() - walSize := m.WAL.Size - if walSize > math.MaxInt64 { - walSize = math.MaxInt64 - } - return map[string]int64{ - "pebble_wal_size": int64(walSize), - "pebble_table_count": m.Total().NumFiles, - } +// Close releases the store's hold on the instance. Safe to call more than +// once: only the first releases. +func (s *tenantStore) Close() error { + var err error + s.closed.Do(func() { err = s.e.release() }) + return err } diff --git a/internal/dedupe/embedded_test.go b/internal/dedupe/embedded_test.go index 9851164be..de65a1512 100644 --- a/internal/dedupe/embedded_test.go +++ b/internal/dedupe/embedded_test.go @@ -3,64 +3,134 @@ package dedupe import ( "context" "os" - "path/filepath" "testing" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/tenant" ) -func newEmbedded(t *testing.T) *EmbeddedDeduplicator { +// switchedOn returns tenant id's store over e, switched on, and switches it +// off at cleanup. +func switchedOn(t *testing.T, e *Embedded, id tenant.ID) *Managed { t.Helper() - d, err := NewEmbedded(t.TempDir()) - require.NoError(t, err) - t.Cleanup(func() { _ = d.Close() }) - return d + m := e.Tenant(id) + require.NoError(t, m.Apply(true)) + t.Cleanup(func() { _ = m.Close() }) + return m } -func TestEmbeddedDeduplicator_FirstSeenThenDuplicate(t *testing.T) { +func TestEmbedded_FirstSeenThenDuplicate(t *testing.T) { t.Parallel() - - d := newEmbedded(t) + m := switchedOn(t, NewEmbedded(t.TempDir()), "acme") ctx := context.Background() - dup, err := d.CheckAndMark(ctx, "event-1") + dup, err := m.CheckAndMark(ctx, "event-1") require.NoError(t, err) assert.False(t, dup, "first occurrence must not be a duplicate") - dup, err = d.CheckAndMark(ctx, "event-1") + dup, err = m.CheckAndMark(ctx, "event-1") require.NoError(t, err) assert.True(t, dup, "second occurrence of the same id must be a duplicate") - dup, err = d.CheckAndMark(ctx, "event-2") + dup, err = m.CheckAndMark(ctx, "event-2") require.NoError(t, err) assert.False(t, dup, "distinct ids are independent") } -func TestEmbeddedDeduplicator_Stats(t *testing.T) { +// Every tenant's seen ids live in one instance and never meet: the tenant +// leads each key, ended by a byte no tenant id holds, so tenant "a" with id +// "bc" and tenant "ab" with id "c" — one key, were the two just joined — are +// two. +func TestEmbedded_TenantsDoNotShareSeenIDs(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + ctx := context.Background() + a, ab := switchedOn(t, e, "a"), switchedOn(t, e, "ab") + + dup, err := a.CheckAndMark(ctx, "bc") + require.NoError(t, err) + assert.False(t, dup) + dup, err = ab.CheckAndMark(ctx, "c") + require.NoError(t, err) + assert.False(t, dup, "another tenant's key, however the two would join") + dup, err = ab.CheckAndMark(ctx, "bc") + require.NoError(t, err) + assert.False(t, dup, "an id tenant a has seen is new to tenant ab") + dup, err = a.CheckAndMark(ctx, "bc") + require.NoError(t, err) + assert.True(t, dup, "and still a duplicate within its own tenant") +} + +// The instance is open exactly while some tenant's store is: nothing is +// opened, nor its directory created, until the first store switches on, and +// it closes with the last one, its files kept for the next open. +func TestEmbedded_OpenWhileAnyTenantStoreIs(t *testing.T) { t.Parallel() + e := NewEmbedded(t.TempDir()) + ctx := context.Background() + acme, globex := e.Tenant("acme"), e.Tenant("globex") + assert.False(t, e.Open()) + assert.NoDirExists(t, e.Dir(), "a store built but never switched on opens nothing") + + require.NoError(t, acme.Apply(true)) + require.NoError(t, globex.Apply(true)) + assert.True(t, e.Open()) + _, err := acme.CheckAndMark(ctx, "e1") + require.NoError(t, err) - d := newEmbedded(t) + require.NoError(t, acme.Apply(false)) + assert.True(t, e.Open(), "globex's store still holds the instance open") + require.NoError(t, globex.Apply(false)) + assert.False(t, e.Open(), "closed with the last store") + entries, err := os.ReadDir(e.Dir()) + require.NoError(t, err) + assert.NotEmpty(t, entries, "its files stay") - stats := d.Stats() - require.NotNil(t, stats) - _, hasWAL := stats["pebble_wal_size"] - _, hasTables := stats["pebble_table_count"] - assert.True(t, hasWAL, "stats must include pebble_wal_size") - assert.True(t, hasTables, "stats must include pebble_table_count") + require.NoError(t, acme.Apply(true)) + t.Cleanup(func() { _ = acme.Close() }) + dup, err := acme.CheckAndMark(ctx, "e1") + require.NoError(t, err) + assert.True(t, dup, "a tenant switched off keeps its seen ids") } -func TestEmbeddedDeduplicator_OpenFailsOnInvalidPath(t *testing.T) { +// The gauges read the one instance: nil while it is closed, and its own +// figures — one set, not one per tenant — while two tenants' stores are open. +func TestEmbedded_StatsAreTheInstances(t *testing.T) { t.Parallel() + e := NewEmbedded(t.TempDir()) + assert.Nil(t, e.Stats(), "closed: the scraper skips the gauges") - // A path that already exists as a regular file is invalid for Pebble — - // it needs a directory. This exercises the error branch of NewEmbedded. - dir := t.TempDir() - path := filepath.Join(dir, "not-a-dir") - f, err := os.Create(path) //nolint:gosec // G304: path is rooted in t.TempDir() + acme := switchedOn(t, e, "acme") + switchedOn(t, e, "globex") + _, err := acme.CheckAndMark(context.Background(), "e1") require.NoError(t, err) - require.NoError(t, f.Close()) + stats := e.Stats() + m := e.db.Metrics() + require.Positive(t, stats["pebble_wal_size"]) + assert.EqualValues(t, m.WAL.Size, stats["pebble_wal_size"], "the instance's, not summed per tenant") + assert.Equal(t, m.Total().NumFiles, stats["pebble_table_count"]) + assert.Len(t, stats, 2) +} + +// An instance that cannot open fails every tenant's store — they share it — +// and stays closed with no store holding it; the next open retries. +func TestEmbedded_OpenFailure(t *testing.T) { + t.Parallel() + e := NewEmbedded(t.TempDir()) + // A regular file where the instance's directory should be is what Pebble + // refuses to open. + require.NoError(t, os.WriteFile(e.Dir(), nil, 0o600)) + acme, globex := e.Tenant("acme"), e.Tenant("globex") + require.Error(t, acme.Apply(true)) + require.Error(t, globex.Apply(true), "one instance: its failure is every tenant's") + assert.False(t, e.Open()) + _, err := acme.CheckAndMark(context.Background(), "e1") + require.ErrorIs(t, err, ErrUnavailable) - _, err = NewEmbedded(path) - require.Error(t, err) + require.NoError(t, os.Remove(e.Dir())) + require.NoError(t, acme.Apply(true), "the next apply retries the open") + t.Cleanup(func() { _ = acme.Close() }) + assert.True(t, e.Open()) } diff --git a/internal/dedupe/managed.go b/internal/dedupe/managed.go index aefe934e9..3337e610d 100644 --- a/internal/dedupe/managed.go +++ b/internal/dedupe/managed.go @@ -22,9 +22,9 @@ var ErrUnavailable = errors.New("dedupe store is not open") // dedupe.enabled setting: Apply(true) opens it through the function // NewManaged was given, Apply(false) closes it, and in-flight CheckAndMark // calls are serialized against that swap so a reload can never close the -// store under a lookup. Which store that is — the embedded Pebble one -// (Embedded), a shared remote backend's per-tenant view later — is the -// opener's business, so every backend gets the same switch semantics. +// store under a lookup. Which store that is — a tenant's share of the +// embedded Pebble instance (Embedded.Tenant), a remote backend's view later — +// is the opener's business, so every backend gets the same switch semantics. type Managed struct { open func() (Deduplicator, error) mu sync.RWMutex @@ -84,17 +84,6 @@ func (m *Managed) CheckAndMark(ctx context.Context, eventID string) (bool, error return m.db.CheckAndMark(ctx, eventID) } -// Stats returns the open store's metrics, or nil while closed (the metrics -// scraper skips a nil map). -func (m *Managed) Stats() map[string]int64 { - m.mu.RLock() - defer m.mu.RUnlock() - if m.db == nil { - return nil - } - return m.db.Stats() -} - // Close releases the store if open. Safe to call when already closed. func (m *Managed) Close() error { return m.Apply(false) diff --git a/internal/dedupe/managed_test.go b/internal/dedupe/managed_test.go index 7e5f81251..74904be69 100644 --- a/internal/dedupe/managed_test.go +++ b/internal/dedupe/managed_test.go @@ -3,8 +3,6 @@ package dedupe import ( "context" "errors" - "os" - "path/filepath" "testing" "github.com/stretchr/testify/assert" @@ -13,19 +11,17 @@ import ( func TestManaged_FollowsEnabled(t *testing.T) { t.Parallel() - m := NewManaged(Embedded(filepath.Join(t.TempDir(), "pebble"))) + m := NewEmbedded(t.TempDir()).Tenant("acme") t.Cleanup(func() { _ = m.Close() }) ctx := context.Background() assert.False(t, m.Open()) - assert.Nil(t, m.Stats(), "closed store reports no stats so the scraper skips it") _, err := m.CheckAndMark(ctx, "e1") require.ErrorIs(t, err, ErrDisabled) require.NoError(t, m.Apply(true)) require.NoError(t, m.Apply(true), "re-applying the same state is a no-op") assert.True(t, m.Open()) - assert.NotNil(t, m.Stats()) dup, err := m.CheckAndMark(ctx, "e1") require.NoError(t, err) assert.False(t, dup) @@ -39,7 +35,7 @@ func TestManaged_FollowsEnabled(t *testing.T) { _, err = m.CheckAndMark(ctx, "e1") require.ErrorIs(t, err, ErrDisabled) - // Re-enabling reopens the same directory: previously seen ids persist. + // Re-enabling reopens the same instance: previously seen ids persist. require.NoError(t, m.Apply(true)) dup, err = m.CheckAndMark(ctx, "e1") require.NoError(t, err) @@ -60,8 +56,7 @@ func (m *memDedup) CheckAndMark(_ context.Context, id string) (bool, error) { m.seen[id] = true return false, nil } -func (m *memDedup) Stats() map[string]int64 { return map[string]int64{"seen": int64(len(m.seen))} } -func (m *memDedup) Close() error { m.closed = true; return nil } +func (m *memDedup) Close() error { m.closed = true; return nil } // The switch semantics belong to Managed, not to Pebble: any Deduplicator // an opener returns gets them, and a failing opener reads as unavailable. @@ -80,7 +75,6 @@ func TestManaged_AnyBackend(t *testing.T) { dup, err = m.CheckAndMark(ctx, "e1") require.NoError(t, err) assert.True(t, dup) - assert.Equal(t, map[string]int64{"seen": 1}, m.Stats()) require.NoError(t, m.Close()) assert.True(t, backend.closed, "closing the switch closes the backend") @@ -93,15 +87,10 @@ func TestManaged_AnyBackend(t *testing.T) { func TestManaged_OpenFailureStaysClosed(t *testing.T) { t.Parallel() - path := filepath.Join(t.TempDir(), "not-a-dir") - f, err := os.Create(path) //nolint:gosec // G304: path is rooted in t.TempDir() - require.NoError(t, err) - require.NoError(t, f.Close()) - - m := NewManaged(Embedded(path)) + m := NewManaged(func() (Deduplicator, error) { return nil, errors.New("disk full") }) require.Error(t, m.Apply(true)) assert.False(t, m.Open()) - _, err = m.CheckAndMark(context.Background(), "e1") + _, err := m.CheckAndMark(context.Background(), "e1") require.ErrorIs(t, err, ErrUnavailable, "switched on but not open must fail closed, not read as disabled") require.NoError(t, m.Close()) _, err = m.CheckAndMark(context.Background(), "e1") diff --git a/internal/dedupe/stores.go b/internal/dedupe/stores.go index 660e9a6d5..614912fec 100644 --- a/internal/dedupe/stores.go +++ b/internal/dedupe/stores.go @@ -11,16 +11,16 @@ import ( ) // Factory builds one tenant's store, closed: Stores calls it the first time -// a tenant is named. It is the seam a shared backend slots into later -// (#583): a factory that puts the tenant in the key of one shared store -// changes nothing that holds the Stores. +// a tenant is named. Whether tenants share a backend is the factory's +// business — Embedded.Tenant keeps them all in one Pebble instance — so +// nothing that holds the Stores changes with it. type Factory func(id tenant.ID) *Managed // Stores is one Managed store per tenant (#583 story 7), each following its // own tenant's dedupe.enabled through Apply. A store is built on first use -// and forgotten by Retain once its tenant is no longer served; its data -// stays on disk either way, so a tenant whose folder comes back finds its -// seen ids where it left them. +// and forgotten by Retain once its tenant is no longer served; its seen ids +// stay either way, so a tenant whose folder comes back finds them where it +// left them. type Stores struct { build Factory mu sync.Mutex @@ -51,11 +51,12 @@ func (s *Stores) For(id tenant.ID) *Managed { // Retain closes and forgets every store whose tenant keep does not name — a // tenant the registry no longer serves — and touches nothing on disk. The // map is edited under the lock and the stores closed outside it, so one -// tenant's close (a Pebble close waits on its flushes and compactions) -// never stalls another tenant's lookup. The close failures are joined; the -// stores are forgotten either way. A store For builds for a dropped tenant -// meanwhile is closed and stays so — only the reconcile that called Retain -// opens one — so no directory is ever open twice. +// tenant's close (the last one closes the Pebble instance, which waits on +// its flushes and compactions) never stalls another tenant's lookup. The +// close failures are joined; the stores are forgotten either way. A store +// For builds for a dropped tenant meanwhile is closed and stays so — only +// the reconcile that called Retain opens one — so a tenant never has two +// stores open at once. func (s *Stores) Retain(keep func(tenant.ID) bool) error { s.mu.Lock() dropped := make(map[tenant.ID]*Managed) @@ -69,31 +70,6 @@ func (s *Stores) Retain(keep func(tenant.ID) bool) error { return closeAll(dropped) } -// Stats sums the open stores' metrics — the process's Pebble footprint, -// which is what the system gauges report — and is nil while no store is -// open, so the scraper skips the gauges as it does for one closed store. The -// stores are read outside the lock: a scrape waiting on one tenant's -// opening store stalls no other tenant's lookup. -func (s *Stores) Stats() map[string]int64 { - s.mu.Lock() - stores := slices.Collect(maps.Values(s.byID)) - s.mu.Unlock() - var sum map[string]int64 - for _, m := range stores { - stats := m.Stats() - if stats == nil { - continue - } - if sum == nil { - sum = make(map[string]int64, len(stats)) - } - for k, v := range stats { - sum[k] += v - } - } - return sum -} - // Close closes every store, in id order, and reports the failures joined. // Safe to call more than once. func (s *Stores) Close() error { diff --git a/internal/dedupe/stores_test.go b/internal/dedupe/stores_test.go index 037951cba..15136343e 100644 --- a/internal/dedupe/stores_test.go +++ b/internal/dedupe/stores_test.go @@ -2,8 +2,6 @@ package dedupe import ( "context" - "os" - "path/filepath" "testing" "time" @@ -13,20 +11,19 @@ import ( "github.com/Wave-RF/WaveHouse/internal/tenant" ) -// pebbleStores is a Stores over a temp root, one directory per tenant, with -// the function naming each tenant's directory. -func pebbleStores(t *testing.T) (*Stores, func(tenant.ID) string) { +// pebbleStores is a Stores over the embedded implementation in a temp +// data_dir, with the implementation itself. +func pebbleStores(t *testing.T) (*Stores, *Embedded) { t.Helper() - root := t.TempDir() - dir := func(id tenant.ID) string { return filepath.Join(root, id.String(), "dedupe") } - s := NewStores(func(id tenant.ID) *Managed { return NewManaged(Embedded(dir(id))) }) + e := NewEmbedded(t.TempDir()) + s := NewStores(e.Tenant) t.Cleanup(func() { _ = s.Close() }) - return s, dir + return s, e } func TestStores_ForBuildsOneClosedStorePerTenant(t *testing.T) { t.Parallel() - s, dir := pebbleStores(t) + s, e := pebbleStores(t) ctx := context.Background() acme := s.For("acme") @@ -35,11 +32,11 @@ func TestStores_ForBuildsOneClosedStorePerTenant(t *testing.T) { assert.False(t, acme.Open(), "built closed: nothing opens until the tenant's switch is applied") _, err := acme.CheckAndMark(ctx, "e1") require.ErrorIs(t, err, ErrDisabled, "a store not yet applied answers as a disabled one, the reload-window case") - assert.NoDirExists(t, dir("acme")) + assert.NoDirExists(t, e.Dir()) require.NoError(t, acme.Apply(true)) - assert.DirExists(t, dir("acme")) - assert.NoDirExists(t, dir("globex"), "the tenant beside it is untouched") + assert.DirExists(t, e.Dir()) + assert.False(t, s.For("globex").Open(), "the tenant beside it is untouched") } func TestStores_TenantsDoNotShareSeenIDs(t *testing.T) { @@ -65,7 +62,7 @@ func TestStores_TenantsDoNotShareSeenIDs(t *testing.T) { func TestStores_RetainClosesTheRestAndKeepsTheirData(t *testing.T) { t.Parallel() - s, dir := pebbleStores(t) + s, _ := pebbleStores(t) ctx := context.Background() acme, globex := s.For("acme"), s.For("globex") require.NoError(t, acme.Apply(true)) @@ -76,11 +73,8 @@ func TestStores_RetainClosesTheRestAndKeepsTheirData(t *testing.T) { require.NoError(t, s.Retain(func(id tenant.ID) bool { return id == "globex" })) assert.False(t, acme.Open(), "the tenant no longer served has its store closed") assert.True(t, globex.Open(), "the one still served is untouched") - entries, err := os.ReadDir(dir("acme")) - require.NoError(t, err, "the directory stays") - assert.NotEmpty(t, entries, "with its files") - // Naming the tenant again builds a fresh store over the same directory: + // Naming the tenant again builds a fresh store over the same instance: // restoring the folder restores the seen ids. restored := s.For("acme") assert.NotSame(t, acme, restored, "the closed store was forgotten") @@ -90,89 +84,42 @@ func TestStores_RetainClosesTheRestAndKeepsTheirData(t *testing.T) { assert.True(t, dup, "an id seen before the tenant was dropped is still seen") } -func TestStores_StatsSumsTheOpenStores(t *testing.T) { - t.Parallel() - s, _ := pebbleStores(t) - assert.Nil(t, s.Stats(), "no store open, no stats: the scraper skips the gauges") - closed := s.For("initech") - assert.Nil(t, s.Stats(), "a closed store adds nothing") - - require.NoError(t, s.For("acme").Apply(true)) - require.NoError(t, s.For("globex").Apply(true)) - want := map[string]int64{} - for _, id := range []tenant.ID{"acme", "globex"} { - for k, v := range s.For(id).Stats() { - want[k] += v - } - } - got := s.Stats() - assert.Equal(t, want, got) - assert.Contains(t, got, "pebble_wal_size") - assert.Contains(t, got, "pebble_table_count") - assert.Nil(t, closed.Stats()) -} - -// gatedDedup is a backend whose Close and Stats block until released: one -// tenant's slow I/O, as a Pebble close waiting on a compaction or an open -// replaying its log. +// gatedDedup is a backend whose Close blocks until released: one tenant's +// slow I/O, as the last Pebble close waiting on a compaction. type gatedDedup struct{ entered, release chan struct{} } func (g *gatedDedup) CheckAndMark(context.Context, string) (bool, error) { return false, nil } -func (g *gatedDedup) Stats() map[string]int64 { g.wait(); return map[string]int64{"seen": 0} } -func (g *gatedDedup) Close() error { g.wait(); return nil } -func (g *gatedDedup) wait() { g.entered <- struct{}{}; <-g.release } +func (g *gatedDedup) Close() error { g.entered <- struct{}{}; <-g.release; return nil } // One tenant's I/O is that tenant's wait alone: Retain edits the map under -// the lock and closes outside it, and Stats reads the stores outside it, so -// a dropped tenant's slow close or a scrape waiting on one store never +// the lock and closes outside it, so a dropped tenant's slow close never // stalls another tenant's lookup, which every dedupe-enabled record makes. func TestStores_IOHappensOutsideTheLock(t *testing.T) { t.Parallel() - setup := func(t *testing.T) (*Stores, *gatedDedup) { - t.Helper() - g := &gatedDedup{entered: make(chan struct{}), release: make(chan struct{})} - s := NewStores(func(tenant.ID) *Managed { - return NewManaged(func() (Deduplicator, error) { return g, nil }) - }) - require.NoError(t, s.For("acme").Apply(true)) - return s, g - } - // forAnswers fails unless For answers while acme's gated call is in - // progress, then releases it. - forAnswers := func(t *testing.T, s *Stores, g *gatedDedup) { - t.Helper() - <-g.entered - defer close(g.release) - got := make(chan *Managed, 1) - go func() { got <- s.For("globex") }() - select { - case m := <-got: - assert.NotNil(t, m) - case <-time.After(2 * time.Second): - t.Fatal("For waited behind another tenant's I/O") - } - } - t.Run("Retain", func(t *testing.T) { - t.Parallel() - s, g := setup(t) - done := make(chan error, 1) - go func() { done <- s.Retain(func(tenant.ID) bool { return false }) }() - forAnswers(t, s, g) - require.NoError(t, <-done) - }) - t.Run("Stats", func(t *testing.T) { - t.Parallel() - s, g := setup(t) - done := make(chan map[string]int64, 1) - go func() { done <- s.Stats() }() - forAnswers(t, s, g) - assert.Equal(t, map[string]int64{"seen": 0}, <-done) + g := &gatedDedup{entered: make(chan struct{}), release: make(chan struct{})} + s := NewStores(func(tenant.ID) *Managed { + return NewManaged(func() (Deduplicator, error) { return g, nil }) }) + require.NoError(t, s.For("acme").Apply(true)) + done := make(chan error, 1) + go func() { done <- s.Retain(func(tenant.ID) bool { return false }) }() + + <-g.entered + got := make(chan *Managed, 1) + go func() { got <- s.For("globex") }() + select { + case m := <-got: + assert.NotNil(t, m) + case <-time.After(2 * time.Second): + t.Error("For waited behind another tenant's I/O") + } + close(g.release) + require.NoError(t, <-done) } func TestStores_CloseClosesEveryStore(t *testing.T) { t.Parallel() - s, _ := pebbleStores(t) + s, e := pebbleStores(t) acme, globex := s.For("acme"), s.For("globex") require.NoError(t, acme.Apply(true)) require.NoError(t, globex.Apply(true)) @@ -180,6 +127,6 @@ func TestStores_CloseClosesEveryStore(t *testing.T) { require.NoError(t, s.Close()) assert.False(t, acme.Open()) assert.False(t, globex.Open()) - assert.Nil(t, s.Stats()) + assert.False(t, e.Open(), "the instance closes with the last store") require.NoError(t, s.Close(), "closing again is a no-op") } diff --git a/internal/ingest/worker.go b/internal/ingest/worker.go index 4c1ff846a..618b2b357 100644 --- a/internal/ingest/worker.go +++ b/internal/ingest/worker.go @@ -65,13 +65,15 @@ type IngestWorker struct { // target resolves a tenant's ClickHouse HTTP wiring per insert // (chconn.Pools.Target in production) so a settings reload that // re-points the tenant applies to the next flush; the zero Target is a - // tenant on no pool, whose insert fails like an unreachable one. + // tenant on no pool, whose batch goes to the dead-letter decision whole + // (parkBatch). target func(tenant.ID) chconn.Target maxBatch int maxWait time.Duration // dlqEnabled reports, per tenant table, whether a row that still fails - // after row-by-row isolation is parked on the DLQ (settings.Store.DLQFor - // in production; nil means always). Resolved at the moment of the failure + // after row-by-row isolation — or every row of a batch with no ClickHouse + // connection — is parked on the DLQ (settings.Store.DLQFor in production; + // nil means always). Resolved at the moment of the failure // under the row's own tenant — the one its topic names — so a settings // reload applies to the next poison row without a restart. dlqEnabled func(id tenant.ID, table string) bool @@ -534,13 +536,19 @@ func (w *IngestWorker) parseMsg(ctx context.Context, m *mq.Message) (parsedMsg, // it falls back to 1-by-1 isolation: each row that re-inserts cleanly is acked, // each that fails again is sent to the DLQ — or, with the DLQ switched off for // the table, left unacked so NATS redelivers it (the row is never dropped, it -// retries until it inserts or the DLQ is switched on). tableLoop guarantees at -// most one concurrent flushTable per tenant table; different tables — two -// tenants' tables of one name included — may flush concurrently. +// retries until it inserts or the DLQ is switched on). A batch whose tenant has +// no ClickHouse connection is not tried at all: it meets its DLQ switch once, +// whole, whatever its column lists (parkBatch). tableLoop guarantees at most +// one concurrent flushTable per tenant table; different tables — two tenants' +// tables of one name included — may flush concurrently. func (w *IngestWorker) flushTable(ctx context.Context, tableName string, msgs []parsedMsg) { if len(msgs) == 0 { return } + if id := msgs[0].tenant; w.target(id).URL == "" { + w.parkBatch(ctx, tableName, msgs, noTargetError(id)) + return + } // One INSERT per distinct column list. The row is positional, so rows // written under different column lists — a schema change mid-stream — @@ -628,7 +636,7 @@ func (w *IngestWorker) insertToClickHouse(ctx context.Context, tableName string, id := msgs[0].tenant t := w.target(id) if t.URL == "" { - return fmt.Errorf("no ClickHouse connection is open for tenant %s", id) + return noTargetError(id) } q := url.Values{} q.Set("database", t.Database) @@ -680,6 +688,31 @@ func (w *IngestWorker) insertToClickHouse(ctx context.Context, tableName string, return nil } +// noTargetError is the failure of tenant id's inserts while it is on no pool: +// no longer served, or refused one, such as by the connection ceiling. No +// request is made. +func noTargetError(id tenant.ID) error { + return fmt.Errorf("no ClickHouse connection is open for tenant %s", id) +} + +// parkBatch disposes of a batch whose tenant has no ClickHouse connection in +// one pass: row-by-row isolation would fail every row the same way, logging +// two lines each. The tenant's DLQ switch is asked once for the batch, which +// is parked under its own topic — a tenant no longer served reads as on +// (dlqFor in internal/app) — or, switched off, left unacked for redelivery +// until the tenant has a pool. +func (w *IngestWorker) parkBatch(ctx context.Context, tableName string, msgs []parsedMsg, cause error) { + id := msgs[0].tenant + if w.dlqEnabled != nil && !w.dlqEnabled(id, tableName) { + slog.ErrorContext(ctx, "no ClickHouse connection for tenant, DLQ disabled for table — batch left unacked, NATS will redeliver it until the tenant has one or dlq is enabled", "tenant", id, "table", tableName, "rows", len(msgs)) + return + } + slog.ErrorContext(ctx, "no ClickHouse connection for tenant, parking the batch on the DLQ", "tenant", id, "table", tableName, "rows", len(msgs)) + for _, pm := range msgs { + w.sendToDLQ(ctx, tableName, pm, cause.Error()) + } +} + func (w *IngestWorker) handleSuccess(ctx context.Context, tableName string, msgs []parsedMsg) { if len(msgs) == 0 { return diff --git a/internal/ingest/worker_test.go b/internal/ingest/worker_test.go index 4c300e459..c3a0988ed 100644 --- a/internal/ingest/worker_test.go +++ b/internal/ingest/worker_test.go @@ -1934,3 +1934,66 @@ func TestInsertToClickHouse_NoTargetIsAnError(t *testing.T) { assert.Equal(t, []tenant.ID{"acme"}, asked, "the target is the batch's own tenant's") assert.Zero(t, rt.Hits()) } + +// A tenant on no pool — no longer served, or refused one by the connection +// ceiling — has no ClickHouse to insert into, so its batch, whatever its +// column lists, skips the row-by-row retry and meets its DLQ switch once: +// parked under its own topic and acked with the switch on — what the wiring's +// dlqFor answers for a tenant it no longer serves — or left unacked for +// redelivery with it off. No request is made for it, and the served tenant +// beside it inserts its own rows into its own ClickHouse alone. +func TestFlushTable_NoTargetParksTheBatchInOnePass(t *testing.T) { + t.Parallel() + for _, dlqOn := range []bool{true, false} { + t.Run(fmt.Sprintf("dlq enabled %v", dlqOn), func(t *testing.T) { + t.Parallel() + rt := &testutil.MockRoundTripper{} + w, pub, _, wait := newTestWorker(rt) + w.target = func(id tenant.ID) chconn.Target { + if id == "acme" { + return chconn.Target{} + } + return chconn.Target{URL: "http://globex-clickhouse:8123", Username: "u", Password: "p", Database: "globex"} + } + var asked []tenant.ID + w.dlqEnabled = func(id tenant.ID, _ string) bool { + asked = append(asked, id) + return dlqOn + } + msg := func(id tenant.ID, data map[string]any) *testutil.MockMessage { + return &testutil.MockMessage{MsgTopic: mq.Topic{Tenant: id, Table: "events"}, MsgData: makeEnvelope(t, "events", "", data)} + } + // Two column lists — a schema change mid-batch — are two INSERTs + // for a tenant with a pool, and still one decision without one. + acme := []*testutil.MockMessage{msg("acme", map[string]any{"id": "acme-1"}), msg("acme", map[string]any{"id": "acme-2", "page": "/"})} + globex := msg("globex", map[string]any{"id": "globex-1"}) + + w.flushTable(context.Background(), "events", parseAll(t, w, acme...)) + w.flushTable(context.Background(), "events", parseAll(t, w, globex)) + wait() + + assert.Equal(t, []tenant.ID{"acme"}, asked, "the switch is asked once for the batch") + captured := rt.Captured() + require.Len(t, captured, 1, "no request is made for acme's rows") + assert.Contains(t, captured[0].URL, "globex-clickhouse") + assert.NotContains(t, string(captured[0].Body), "acme") + assert.True(t, globex.DoubleAcked.Load()) + + published := pub.Published() + if !dlqOn { + assert.Empty(t, published) + for _, m := range acme { + assert.False(t, m.DoubleAcked.Load(), "left unacked for redelivery") + } + return + } + require.Len(t, published, 2) + for i, p := range published { + assert.True(t, p.DeadLetter) + assert.Equal(t, mq.Topic{Tenant: "acme", Table: "events"}, p.Topic, "parked under its own topic") + assert.Equal(t, "no ClickHouse connection is open for tenant acme", p.Headers.Get("X-DLQ-Error")) + assert.True(t, acme[i].DoubleAcked.Load()) + } + }) + } +} diff --git a/internal/observability/metrics.go b/internal/observability/metrics.go index 110238d9b..a227442a6 100644 --- a/internal/observability/metrics.go +++ b/internal/observability/metrics.go @@ -19,8 +19,8 @@ type MQStats struct { // stats from embedded systems (the MQ, Pebble) and push them to OpenTelemetry. // Both are read on every scrape; nil skips the MQ gauges, as a nil // pebbleStats — or a nil map from it, no store being open — skips the Pebble -// ones. The Pebble figures are the process's, summed across the tenants' -// stores (dedupe.Stores.Stats in production). +// ones. The Pebble figures are the one instance's that every tenant's seen +// ids share (dedupe.Embedded.Stats in production). func RegisterSystemMetrics(mqStats func() (MQStats, error), pebbleStats func() map[string]int64) error { meter := otel.Meter("wavehouse-system") diff --git a/internal/settings/registry.go b/internal/settings/registry.go index 8aea7aad0..663e0a232 100644 --- a/internal/settings/registry.go +++ b/internal/settings/registry.go @@ -31,10 +31,10 @@ import ( // registry still knows the tenant, so its requests are refused rather // than unknown — and every other tenant carries on; there is no // previous-snapshot fallback. A reload mirrors the folders: a new one is -// served and a removed one is forgotten. A finding about the directory -// itself (a loose file, an unreadable directory, a changed shape) rejects -// the reload whole and leaves every tenant as it was; at Open it refuses -// boot. +// served and a removed one is forgotten, the last one too. A finding about +// the directory itself (a loose file, an unreadable directory, a changed +// shape) rejects the reload whole and leaves every tenant as it was; at +// Open it refuses boot. type Registry struct { dir string nested bool @@ -186,7 +186,7 @@ func (r *Registry) Reload(trigger string) ([]Finding, bool) { // directory's shape rather than holding it to one. applied reports whether // the registry took the tree at all. func (r *Registry) reload(trigger string, boot bool) (findings []Finding, adopted, applied bool) { - tree, findings := Validate(r.dir) + tree, findings := r.validate(boot) switch { case tree == nil: case boot: @@ -247,6 +247,18 @@ func (r *Registry) reload(trigger string, boot bool) (findings []Finding, adopte return findings, errs == 0, tree != nil } +// validate is Validate, but for an empty root under a registry serving +// folders, which it reads as the nested directory it is with no folder left +// rather than as Validate's flat directory missing its files — so removing +// the last tenant's folder removes the tenant, like any other. At Open an +// empty root still reads as flat: nothing says it was meant to hold folders. +func (r *Registry) validate(boot bool) (*Tree, []Finding) { + if !boot && r.nested && emptyRoot(r.dir) { + return &Tree{Nested: true, Tenants: map[tenant.ID]TenantResult{}}, nil + } + return Validate(r.dir) +} + // ReloadTenant is Reload for one tenant's folder of a nested directory: the // rest of the directory is not read, so it can neither adopt nor drop another // tenant — what the writer of one folder calls when that folder is complete. diff --git a/internal/settings/registry_test.go b/internal/settings/registry_test.go index 05d6500de..0bdff5699 100644 --- a/internal/settings/registry_test.go +++ b/internal/settings/registry_test.go @@ -92,6 +92,12 @@ func TestOpen_RejectsInvalid(t *testing.T) { reg, findings = Open(filepath.Join(t.TempDir(), "nope")) assert.Nil(t, reg) assert.True(t, HasErrors(findings)) + + // Nothing says an empty directory was meant to hold tenant folders, so + // it boots as the four files, missing. + reg, findings = Open(t.TempDir()) + assert.Nil(t, reg) + assert.Contains(t, findingStrings(findings), "error: config.json: missing") } // TestRegistry_SurvivesVanishedDirectory pins the runtime half of the same @@ -384,6 +390,47 @@ func TestRegistry_NestedReloadMirrorsTheFolders(t *testing.T) { assert.True(t, ok, "a badly named folder costs no tenant its settings") } +// Removing the last folder removes the last tenant, like any other: the +// empty root it leaves is no change of shape to a registry serving folders. +// A flat directory emptied the same way is still a rejected reload. +func TestRegistry_NestedReloadRemovesTheLastTenant(t *testing.T) { + t.Parallel() + root := writeTree(t, map[string]map[string]string{"acme": maxRowsFiles(111), "globex": maxRowsFiles(222)}) + reg, _ := Open(root) + require.NotNil(t, reg) + var hooks [][]tenant.ID + reg.AfterAdopt(func(adopted []tenant.ID) { hooks = append(hooks, adopted) }) + + require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) + require.NoError(t, os.RemoveAll(filepath.Join(root, "globex"))) + require.NoError(t, os.WriteFile(filepath.Join(root, ".keep"), nil, 0o600), "dot-prefixed entries count for neither shape") + findings, adopted := reg.Reload("test") + assert.True(t, adopted) + assert.Empty(t, findings) + for _, id := range []tenant.ID{"acme", "globex"} { + _, known := reg.Resolve(id) + assert.False(t, known, "%s is removed, not rejected", id) + } + assert.Equal(t, [][]tenant.ID{nil}, hooks, "the hooks ran, with nothing adopted") + + writeTenant(t, root, "acme", maxRowsFiles(333)) + _, adopted = reg.Reload("test") + require.True(t, adopted) + store, ok := reg.For("acme") + require.True(t, ok, "a folder written back is a tenant again") + assert.Equal(t, 333, store.DefaultMaxRows()) + + flat := newLoadedRegistry(t, map[string]string{FileConfig: configJSON(`{"query": {"default_max_rows": 42}}`)}) + for _, name := range Files() { + require.NoError(t, os.Remove(filepath.Join(flat.Dir(), name))) + } + _, adopted = flat.Reload("test") + assert.False(t, adopted) + store, ok = flat.For(tenant.Default) + require.True(t, ok) + assert.Equal(t, 42, store.DefaultMaxRows(), "the four files' tenant keeps its document") +} + // A tenant whose folder is gone leaves the registry with no finding to show // for it — every request of its just turns into a 404 — so the reload's log // line names it. Captures the default logger, so it is not parallel. @@ -517,10 +564,6 @@ func TestRegistry_RootLevelFailureChangesNothing(t *testing.T) { {name: "the root is gone", want: "does not exist", damage: func(t *testing.T, root string) { require.NoError(t, os.RemoveAll(root)) }}, - {name: "every folder is gone", want: "no longer has the shape this server booted with (one folder per tenant)", damage: func(t *testing.T, root string) { - require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) - require.NoError(t, os.RemoveAll(filepath.Join(root, "globex"))) - }}, {name: "the root turned flat", want: "no longer has the shape this server booted with (one folder per tenant)", damage: func(t *testing.T, root string) { require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) require.NoError(t, os.RemoveAll(filepath.Join(root, "globex"))) diff --git a/internal/settings/settings.go b/internal/settings/settings.go index c73653c5d..d0f836e5a 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -137,9 +137,9 @@ type AuthConfig struct { } // DedupeConfig tunes dedupe behavior, including the switch itself: a reload -// that flips enabled opens or closes the tenant's embedded Pebble store on -// the fly (dedupe.Managed, one per tenant), so the whole block is -// tenant-owned. +// that flips enabled opens or closes the tenant's store on the fly +// (dedupe.Managed, one per tenant, each a share of the one embedded Pebble +// instance), so the whole block is tenant-owned. // // id_field and require_id are required here and optional per table: a table // override inherits whichever field it doesn't name. An empty, diff --git a/internal/settings/tree.go b/internal/settings/tree.go index c0efe8359..5353c9aae 100644 --- a/internal/settings/tree.go +++ b/internal/settings/tree.go @@ -83,10 +83,8 @@ func Validate(root string) (*Tree, []Finding) { // validateFolder is ValidateDir for one tenant folder of a nested root, with // the folder leading each finding's File. // -// This is one of the two places a tenant id becomes a filesystem path; the -// other is the tenant's dedupe store directory under data_dir -// (internal/app's wireDedupe), which takes its ids from the registry, so -// each is tenant 0, the constant, or a folder name checked here first. +// This is the one place a tenant id becomes a filesystem path: the dedupe +// store keys by tenant, not by directory (internal/dedupe's Embedded). // Every caller already hands it an id that passed tenant.Parse (which // forbids '.', '/' and '\'), so the check below is never reached today; it // is here so the guarantee that a name resolves to one folder under root — @@ -106,6 +104,21 @@ func validateFolder(root, folder string) (*Document, []Finding) { return doc, findings } +// emptyRoot reports whether root can be listed and holds nothing but +// dot-prefixed entries, which both shapes skip. +func emptyRoot(root string) bool { + entries, err := os.ReadDir(root) + if err != nil { + return false + } + for _, e := range entries { + if !strings.HasPrefix(e.Name(), ".") { + return false + } + } + return true +} + // looseEntry is a root entry that is not a tenant folder: a file, or // something that could not be stat'ed (err says why). type looseEntry struct { diff --git a/internal/settings/tree_test.go b/internal/settings/tree_test.go index 2282efb70..4c3feeb6a 100644 --- a/internal/settings/tree_test.go +++ b/internal/settings/tree_test.go @@ -146,7 +146,6 @@ func TestValidate_NestedFolderNames(t *testing.T) { {name: "dot", folder: "acme.bak", want: `error: acme.bak: folder name is not a tenant id: tenant id has '.' at byte 4`}, {name: "space", folder: "acme corp", want: `error: acme corp: folder name is not a tenant id: tenant id has ' ' at byte 4`}, {name: "over the length cap", folder: strings.Repeat("a", tenant.MaxLen+1), want: "folder name is not a tenant id: tenant id is 65 bytes, the limit is 64"}, - {name: "reserved", folder: "nats", want: `error: nats: folder name is not a tenant id: tenant id "nats" is reserved: data_dir/nats is the embedded queue's directory`}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { diff --git a/internal/stream/bucket.go b/internal/stream/bucket.go index 637bdd367..966729164 100644 --- a/internal/stream/bucket.go +++ b/internal/stream/bucket.go @@ -7,7 +7,7 @@ import "sync" // that both Hub paths iterate Snapshot, since the schema announcement is per // connection even where the projection is shared per role (Send itself counts // any queue-full drop); Snapshot exposes the members so the event Hub can -// evaluate row visibility per subscriber before sending (and, later, evict). +// evaluate row visibility per subscriber before sending, and evict them. type Bucket interface { Add(sub *Subscriber) Remove(sub *Subscriber) diff --git a/internal/stream/hub.go b/internal/stream/hub.go index 1564feac3..e7ce2d4d4 100644 --- a/internal/stream/hub.go +++ b/internal/stream/hub.go @@ -171,6 +171,27 @@ func (h *Hub) Len(topic mq.Topic) int { return n } +// Prune evicts every subscriber of a tenant served does not vouch for — one a +// reload removed or rejected — so its handler ends the stream, and the +// client's reconnect meets that tenant's 404 or 503 until it is served again. +// Left open, the stream would outlive its tenant: every row withheld under the +// nil policy read for it, the keepalive wheel holding it open, a quiet table +// to the client. The handlers' deferred Remove takes the subscribers out. +func (h *Hub) Prune(served func(tenant.ID) bool) { + h.mu.RLock() + defer h.mu.RUnlock() + for topic, tr := range h.topics { + if served(topic.Tenant) { + continue + } + for _, b := range tr.roles { + for _, sub := range b.Snapshot() { + sub.Evict() + } + } + } +} + // roleBucket pairs a subscribed role with its bucket for the lock-free fan-out. type roleBucket struct { role string diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index 5bbede8e5..4510b475f 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -1572,3 +1572,31 @@ func TestHub_TopicsAreTenantScoped(t *testing.T) { defer mu.Unlock() assert.Equal(t, []tenant.ID{"acme", "globex", "globex", "globex", "initech"}, asked, "each read names the event's or the connection's tenant") } + +// TestHub_PruneEvictsTheTenantsNoLongerServed: Prune evicts every subscriber +// of a tenant no longer served — on each of its topics, under each role — and +// no other tenant's. It only marks them: each stays registered until its +// handler's Remove. +func TestHub_PruneEvictsTheTenantsNoLongerServed(t *testing.T) { + t.Parallel() + hub := NewHub(nil, nil, nil) + evicted := func(sub *Subscriber) bool { + select { + case <-sub.Evicted(): + return true + default: + return false + } + } + acmeClicks := mq.Topic{Tenant: "acme", Table: "clicks"} + viewer, admin, globex := NewSubscriber(nil, nil), NewSubscriber(nil, nil), NewSubscriber(nil, nil) + hub.Add(acmeClicks, "viewer", viewer) + hub.Add(mq.Topic{Tenant: "acme", Table: "views"}, "admin", admin) + hub.Add(mq.Topic{Tenant: "globex", Table: "clicks"}, "viewer", globex) + + hub.Prune(func(id tenant.ID) bool { return id == "globex" }) + assert.True(t, evicted(viewer), "acme is no longer served") + assert.True(t, evicted(admin), "on every topic and under every role of it") + assert.False(t, evicted(globex), "globex is still served") + assert.Equal(t, 1, hub.Len(acmeClicks), "evicted, not removed: the handler removes it") +} diff --git a/internal/stream/subscriber.go b/internal/stream/subscriber.go index 2be41e3a8..548834928 100644 --- a/internal/stream/subscriber.go +++ b/internal/stream/subscriber.go @@ -38,12 +38,13 @@ type Subscriber struct { // row-filter closed. claims map[string]any - // evict is closed (once) to ask the owning handler to disconnect a wedged slow - // consumer; the handler selects on Evicted() and tears the stream down, after - // which the client reconnects and gap-fills via Last-Event-ID. The seam is wired - // here and consumed by the handler; the policy that *closes* it (consecutive-drop - // threshold) lands with the slow-consumer follow-up (#294 / #94). - evict chan struct{} + // evict is closed, once, by Evict to ask the owning handler to end the stream; + // the handler selects on Evicted() and tears the stream down, after which the + // client reconnects. The Hub evicts every subscriber of a tenant no longer + // served (Hub.Prune); disconnecting a wedged slow consumer is the other + // closer it is meant for (#294 / #94). + evict chan struct{} + evictOnce sync.Once // metric counts queue-full drops inside Send itself (by frame kind), so every // producer — the event fan-out, replay, the keepalive wheel — is covered without @@ -155,9 +156,14 @@ func (s *Subscriber) Send(f Frame) bool { } } -// Evicted is closed when the subscriber has been marked for disconnection. The -// handler selects on it to tear the connection down. Inert until the slow-consumer -// follow-up wires the threshold that closes it. +// Evict marks the subscriber for disconnection: its handler ends the stream. +// Safe to call more than once, from any goroutine. +func (s *Subscriber) Evict() { + s.evictOnce.Do(func() { close(s.evict) }) +} + +// Evicted is closed once the subscriber has been marked for disconnection. The +// handler selects on it to tear the connection down. func (s *Subscriber) Evicted() <-chan struct{} { return s.evict } diff --git a/internal/stream/subscriber_test.go b/internal/stream/subscriber_test.go index e1b1a99a6..332d0fac7 100644 --- a/internal/stream/subscriber_test.go +++ b/internal/stream/subscriber_test.go @@ -27,15 +27,23 @@ func TestSubscriber_SendDeliversThenDropsWhenFull(t *testing.T) { assert.True(t, sub.Send(frame), "Send enqueues again once the queue drained") } -func TestSubscriber_EvictedIsOpenUntilClosed(t *testing.T) { +func TestSubscriber_EvictedIsOpenUntilEvicted(t *testing.T) { t.Parallel() sub := NewSubscriber(nil, nil) - // The eviction seam is inert until the slow-consumer follow-up closes it: the - // channel stays open, so a non-blocking read finds nothing. select { case <-sub.Evicted(): t.Fatal("Evicted must not fire until the subscriber is marked for eviction") default: } + + // Evicting twice is safe: the Hub's Prune and the slow-consumer threshold + // may both reach the same subscriber. + sub.Evict() + sub.Evict() + select { + case <-sub.Evicted(): + default: + t.Fatal("Evicted must fire once the subscriber is evicted") + } } diff --git a/internal/tenant/tenant.go b/internal/tenant/tenant.go index bf41b6f89..2a08c97c3 100644 --- a/internal/tenant/tenant.go +++ b/internal/tenant/tenant.go @@ -7,7 +7,6 @@ package tenant import ( "errors" "fmt" - "strings" ) // ID is a validated tenant identifier. It is a string, never a number: ids @@ -26,25 +25,10 @@ const ( MaxLen = 64 ) -// reserved are the names that fit the grammar and are no tenant's, in any -// letter case: the entries data_dir keeps for itself beside the tenants' own -// directories (data_dir/, #583 story 7) — the embedded queue's nats -// and the earlier layout's pebble dedupe store — so no tenant's directory is -// ever one of those. Any letter case, because the path a name becomes is -// only as case-sensitive as the filesystem under data_dir (macOS and Windows -// are not, by default) and the grammar is ASCII, so lowercasing is exact. -// Data-directory conventions, but named here: the grammar is the one check -// every path a tenant is named on goes through. -var reserved = map[string]string{ - "nats": "the embedded queue's directory", - "pebble": "the earlier layout's dedupe store", -} - // Parse validates s against the one grammar an id must satisfy to be safe // both as a folder name and as a message-queue subject token: ASCII letters, -// digits, '_' and '-', at most MaxLen bytes, and not a reserved name in any -// letter case. Dots, slashes, spaces, and wildcards are rejected because -// each means something to one of the two. +// digits, '_' and '-', at most MaxLen bytes. Dots, slashes, spaces, and +// wildcards are rejected because each means something to one of the two. func Parse(s string) (ID, error) { if s == "" { return "", errors.New("tenant id is empty") @@ -60,9 +44,6 @@ func Parse(s string) (ID, error) { return "", fmt.Errorf("tenant id has %q at byte %d: only letters, digits, '_' and '-' are allowed", c, i) } } - if what, ok := reserved[strings.ToLower(s)]; ok { - return "", fmt.Errorf("tenant id %q is reserved: data_dir/%s is %s", s, strings.ToLower(s), what) - } return ID(s), nil } diff --git a/internal/tenant/tenant_test.go b/internal/tenant/tenant_test.go index a0a41e22e..26341da79 100644 --- a/internal/tenant/tenant_test.go +++ b/internal/tenant/tenant_test.go @@ -15,7 +15,6 @@ func TestParse(t *testing.T) { {name: "letters digits underscore dash", in: "Acme_co-42"}, {name: "19 digit id", in: "9223372036854775807"}, {name: "at the length cap", in: strings.Repeat("a", MaxLen)}, - {name: "a reserved name is exact", in: "nats-eu"}, {name: "empty", in: "", wantErr: true}, {name: "over the length cap", in: strings.Repeat("a", MaxLen+1), wantErr: true}, {name: "dot", in: "a.b", wantErr: true}, @@ -27,10 +26,6 @@ func TestParse(t *testing.T) { {name: "subject wildcard tail", in: ">", wantErr: true}, {name: "non-ascii letter", in: "ténant", wantErr: true}, {name: "newline", in: "a\n", wantErr: true}, - {name: "reserved: the queue's directory", in: "nats", wantErr: true}, - {name: "reserved: the earlier dedupe store", in: "pebble", wantErr: true}, - {name: "reserved in any letter case", in: "Pebble", wantErr: true}, - {name: "reserved in upper case", in: "NATS", wantErr: true}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { diff --git a/internal/testutil/mocks.go b/internal/testutil/mocks.go index 6a9dc64af..1b8cf358f 100644 --- a/internal/testutil/mocks.go +++ b/internal/testutil/mocks.go @@ -115,14 +115,6 @@ func NewMockDeduplicator() *MockDeduplicator { return &MockDeduplicator{seen: make(map[string]bool)} } -func (m *MockDeduplicator) Stats() map[string]int64 { - // Return empty stats or mock data for testing - return map[string]int64{ - "pebble_wal_size": 0, - "pebble_table_count": 0, - } -} - func (m *MockDeduplicator) CheckAndMark(_ context.Context, eventID string) (bool, error) { if m.Err != nil { return false, m.Err