From ef93154b6528054466744df0eec2f26bf1575002 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 17:49:08 -0400 Subject: [PATCH 01/15] feat(mq): give every tenant a queue of its own --- AGENTS.md | 8 +- CHANGELOG.md | 8 +- clients/ts/src/dlq.ts | 18 +- clients/ts/src/namespaces.test.ts | 19 + clients/ts/src/types.ts | 4 +- docs/src/content/docs/api.md | 9 +- docs/src/content/docs/architecture.md | 16 +- docs/src/content/docs/configuration.mdx | 2 +- docs/src/content/docs/deployment.md | 18 +- docs/src/content/docs/durability.md | 6 +- docs/src/content/docs/ingest-pipeline.md | 26 +- docs/src/content/docs/sdk/admin.md | 7 + docs/src/content/docs/sdk/reference.md | 4 +- docs/src/content/docs/settings-directory.mdx | 16 +- docs/src/content/docs/why-wavehouse.md | 8 +- internal/api/dlq.go | 29 +- internal/api/dlq_test.go | 138 ++-- internal/api/ingest.go | 2 +- internal/api/router_test.go | 5 +- internal/app/app.go | 9 +- internal/app/app_test.go | 56 +- internal/app/wire.go | 117 +-- internal/ingest/sweeper.go | 41 +- internal/ingest/sweeper_test.go | 31 +- internal/ingest/worker.go | 19 +- internal/ingest/worker_test.go | 28 +- internal/mq/embedded.go | 756 ++++++++++++++----- internal/mq/embedded_test.go | 649 ++++++++++++---- internal/mq/mq.go | 130 ++-- internal/mq/subject.go | 76 +- internal/mq/subject_test.go | 52 +- internal/settings/settings.go | 23 +- internal/settings/store.go | 2 +- internal/stream/subscriber.go | 20 +- internal/testutil/mocks.go | 16 +- internal/testutil/testutil.go | 19 + 36 files changed, 1679 insertions(+), 708 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 59f08ef3..16595721 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,7 +29,7 @@ One binary: Eighteen internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): - **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers -- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive`/`longestGapWindow` for the two settings folded over every tenant served, and `defaultSetting`/`onDefaultAdopt` for the one resource a process still has one of, the MQ, which follows tenant `0`; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it +- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` and the `mq.max_bytes_gb` reconcile handing the MQ each served tenant's own gap window and byte budget, and `defaultSetting`/`onDefaultAdopt` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index). Every key leads with the tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for a result and its singleflight, `...
.` for a namespace — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` advances the tenant version that leads every namespace key of one tenant, orphaning every cached query keyed by its tables in one step (a pipe result names no table and keeps its TTL, [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the table-keyed cache (the structured-query results) of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) - **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config @@ -38,14 +38,14 @@ Eighteen internal packages under `internal/` (plus `internal/testutil/` for shar - **`dedupe/`** — `Deduplicator` interface → `Embedded` (Pebble: every tenant's seen ids in one instance at `data_dir/pebble`, each key led by its tenant, open while any tenant's store is — the layout is the implementation's call, and the wiring hands it `data_dir` once; its `Stats` feed the system gauges), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, `Embedded.Tenant` in production; `Managed` opens its store through a function, so every backend gets the same switch), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its seen ids kept otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) stories 7 and 3) - **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) - **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`), so non-admin callers never reach the proxy. A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) -- **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, a topic without one is refused, and a pre-tenant subject reads as tenant `0`'s) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds the byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload) and `Stats` (the system gauges' source). Subjects, prefixes, wildcards, token encoding, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`); `internal/app` constructs it and hands everything else a `mq.Broker` +- **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, and a topic without one is refused) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds each tenant's byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload, which opens a tenant's queue the first time) and `Stats` (the system gauges' source). Every interface speaks per tenant, never per stream: the embedded implementation gives each tenant a queue of its own (a stream pair, `INGEST_`/`DLQ_`), and nothing outside the package may assume that layout — an external implementation may keep one shared stream. Subjects, prefixes, wildcards, token encoding, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`); `internal/app` constructs it and hands everything else a `mq.Broker` - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). - **`pipes/`** — Named query pipes: `NamedQuery` type + `BindParams` + `Source` (read per request; `settings.Store` in production, `Static(q...)` in tests) - **`policy/`** — Hasura-style access control, **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant splits by operation into `SelectPermissions` (columns, row `filter`, aggregations, the `max_*` limits) and `InsertPermissions` (columns, `check`) — so a field only one side honors does not exist on the other. `Evaluate()` resolves ONE operation and leaves the other side **nil** (`Select *ResolvedSelect` / `Insert *ResolvedInsert`), which every accessor fails closed on — nil is "not resolved", distinct from an empty side, which is "unrestricted" (what the admin return builds). Claim templating (`{{ jwt.claim.path }}`) resolves during that call. Policies come from `Source`, a `func() *Policy` read per call (`settings.Store.Policy` in production, `Static(p)` in tests) - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper folds over the tenants served (`longestGapWindow`); each served tenant has a schema registry of its own (story 6) +- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each served tenant's own gap window (`gapWindows`); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -56,7 +56,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 3. **Schema-driven ingest** — `POST /v1/ingest?table={table}` takes flat JSON, validated against the discovered schema (unknown fields rejected, types/nullability enforced). No envelope. The **declared `Content-Type` chooses the format and the bytes never do** (arity within the JSON family is still the body's): no declaration, one whose **media type** is unsupported or unparseable, a comma-bearing value that, as a whole, does not parse as one media type, or repeated lines that **disagree**, is a `415` decided *before* the body is read. A malformed *parameter* on a comma-free line never costs the request (`; charset=a; charset=b` still reads as its media type), and repeated lines are accepted only when they all resolve to the same **supported** format — two agreeing `text/csv` lines are still a `415`. A body declared NDJSON stays NDJSON whatever its bytes, so a bad line is a per-record error rather than a silent re-framing; the reverse (NDJSON sent as `application/json`) is deliberately **not** caught — record one, `200`, the rest ignored ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). Fail-closed — preserve it when touching `internal/api`. 4. **Async ingestion** — ingest returns 200 after optional dedup + MQ publish; ClickHouse writes happen later via `StartIngestWorker`. NATS full → 503 + Retry-After. 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. -6. **Dead Letter Queue** — failed batch inserts publish to `WAVEHOUSE_DLQ` (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format` — a pre-v2 envelope carries none — or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. +6. **Dead Letter Queue** — failed batch inserts publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format` — a pre-v2 envelope carries none — or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. 7. **Auth: always on, fail-loud, decoupled from authz (security)** — the JWT middleware always runs (no `auth.enabled`/`dev_mode` flag); it verifies with HMAC **or** JWKS (not both), with accepted `alg` pinned to the active verifier and checked before any key is used (rejects `alg:none` and cross-family confusion). No/invalid/expired token → empty role → policy `default_role`, with the bad-token reason stashed so a denying gate returns a loud `401`, not a bare `403`; the one token outcome that never reaches `default_role` is a verifier still fetching its JWKS (`auth.ErrVerifierPending` → `503` + `Retry-After`, `api.refuseUnverifiable`). Elevated access needs a valid granted role. **Sanctioned exception:** a configured non-JWT operator key (`auth.operator_key`; presented via `Authorization: Operator ` or the `X-Operator-Key` alias) deliberately couples authN+authZ — a constant-time match authorizes a full-access platform operator (stamps the admin role plus an operator bit) independent of the verifier (see #11). Detail: architecture.md § `api/` + `internal/auth`; see also #11, §Security Considerations. 8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key, overridable per table. 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. diff --git a/CHANGELOG.md b/CHANGELOG.md index 5bf58019..23c0c715 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,7 +10,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Added -- **A tenant removed or rejected at runtime has its open streams ended** (`internal/stream/{hub,subscriber,bucket}.go` (+ tests), `internal/api/stream.go` (+ tests), `internal/api/router.go`, `internal/settings/{registry,tree}.go` (+ tests), `internal/ingest/worker.go` (+ tests), `internal/app/wire.go` (+ tests), `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/settings-directory.mdx`, `AGENTS.md`): story 3 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). A `GET /v1/stream` used to outlive its tenant: the hub read no policy for it and withheld every row while the keepalive wheel held the connection open, so the client could not tell it from a quiet table. After every reload the hub now evicts the subscribers of each tenant no longer served, removed or rejected alike (`Hub.Prune`, through the close-once `Subscriber.Evict`), and the handler ends the stream: a gap-fill in progress included, and a stream `TenantMW` admitted just before the reload but registered just after it, which the handler checks for as it registers. The client's reconnect gets `404` (removed) or `503` (rejected); the SDK stops on the first and retries the second, resuming from `Last-Event-ID` once the folder is back — except in a browser going cross-origin where tenant `0` is not served or its CORS list does not admit the page, which cannot read either refusal (both are decorated from tenant `0`'s list) and re-dials as after a dropped connection. A flat directory never stops serving tenant `0`, so nothing changes there. Removing a tenant is deleting its folder, then reloading the whole directory, the last folder included: a server started with tenant folders reads the emptied directory as no folder left, where it read as a change of shape and the reload was rejected whole, leaving that tenant served; at boot an empty directory still reads as the four files, missing. Reloading the deleted folder by name leaves its tenant rejected. `GET /v1/health` keeps resolving a tenant, deliberately: its `404` tells a caller with no token no more than every tenant route's does, since they all answer before authenticating, and resolving answers a served tenant's ping from that tenant's own CORS list. A batch queued for a tenant with no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and meets its DLQ switch once, whole, logged once per batch rather than twice per row; a tenant no longer served reads the switch as on, so its queued rows are parked under its own subject rather than dropped, inserted into another tenant's ClickHouse, or left unacked to hold the ack floor that the one shared stream's purge waits on. Over a nested directory `/livez` no longer keeps naming a tenant that stopped being served before any tenant completed a first discovery: the diagnostic goes back to `no tenant has completed a first discovery yet`. +- **A tenant removed or rejected at runtime has its open streams ended** (`internal/stream/{hub,subscriber,bucket}.go` (+ tests), `internal/api/stream.go` (+ tests), `internal/api/router.go`, `internal/settings/{registry,tree}.go` (+ tests), `internal/ingest/worker.go` (+ tests), `internal/app/wire.go` (+ tests), `clients/ts/src/stream/sse.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/settings-directory.mdx`, `AGENTS.md`): story 3 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). A `GET /v1/stream` used to outlive its tenant: the hub read no policy for it and withheld every row while the keepalive wheel held the connection open, so the client could not tell it from a quiet table. After every reload the hub now evicts the subscribers of each tenant no longer served, removed or rejected alike (`Hub.Prune`, through the close-once `Subscriber.Evict`), and the handler ends the stream: a gap-fill in progress included, and a stream `TenantMW` admitted just before the reload but registered just after it, which the handler checks for as it registers. The client's reconnect gets `404` (removed) or `503` (rejected); the SDK stops on the first and retries the second, resuming from `Last-Event-ID` once the folder is back — except in a browser going cross-origin where tenant `0` is not served or its CORS list does not admit the page, which cannot read either refusal (both are decorated from tenant `0`'s list) and re-dials as after a dropped connection. A flat directory never stops serving tenant `0`, so nothing changes there. Removing a tenant is deleting its folder, then reloading the whole directory, the last folder included: a server started with tenant folders reads the emptied directory as no folder left, where it read as a change of shape and the reload was rejected whole, leaving that tenant served; at boot an empty directory still reads as the four files, missing. Reloading the deleted folder by name leaves its tenant rejected. `GET /v1/health` keeps resolving a tenant, deliberately: its `404` tells a caller with no token no more than every tenant route's does, since they all answer before authenticating, and resolving answers a served tenant's ping from that tenant's own CORS list. A batch queued for a tenant with no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the row-by-row retry, which no row of it could pass, and meets its DLQ switch once, whole, logged once per batch rather than twice per row; a tenant no longer served reads the switch as on, so its queued rows are parked under its own subject rather than dropped, inserted into another tenant's ClickHouse, or left unacked, redelivered for as long as the tenant is away and holding its queue's ack floor, which the purge waits on. Over a nested directory `/livez` no longer keeps naming a tenant that stopped being served before any tenant completed a first discovery: the diagnostic goes back to `no tenant has completed a first discovery yet`. - **One ClickHouse pool per tuple and one schema registry per tenant** (`internal/chconn/chconn.go` (+ tests), `internal/discovery/discovery.go` (+ tests), `internal/app/discoveries.go` (new), `internal/app/{app,wire}.go` (+ tests), `internal/api/{schema,ingest,structured_query,pipes,query,health,errors,clickhouse_exec}.go` (+ tests), `internal/stream/hub.go`, `internal/ingest/worker.go`, `internal/cache/{cache,local,version_manager}.go` (+ tests), `internal/testutil/{testutil,mocks}.go`, `tests/integration/{setup,tenants,boot_resilience,query_limits}_test.go`, `clients/ts/src/{schema,table,sql,client,types}.ts` (+ tests), `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/{settings-directory,configuration,access-control,reverse-proxy}.mdx`, `docs/src/content/docs/sdk/{admin,reference,queries}.md`, `AGENTS.md`): the second slice of story 6 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files beyond the three noted at the end. The process opens one native pool per distinct `clickhouse.addr` / `database` / `username` / password / `tls` tuple among the served tenants (`chconn.Identity`, `chconn.Pools`), shared by the tenants naming it and sized to their largest `max_open_conns` and `max_idle_conns`; `http_port`, `http_scheme`, `headers` and `query_timeout` stay each tenant's own. Every reload reconciles the pools: a new tuple opens (never dials), a tenant whose tuple changed is repointed, a tuple no tenant names closes after the longest `query_timeout` among the tenants it had, and a pool whose largest ask changed is resized with the same grace. The boot config's `clickhouse.max_total_conns` now bounds the open pools together: boot is refused naming the sum and the ceiling; at a reload a resize above it is refused with the pool kept at its size, and a tuple that cannot be opened — the ceiling, a certificate file that cannot be read, or options the driver refuses — leaves its tenants on the pool they had (the keep-previous-wiring rule of the connection ceiling) or on none when they had none; both are logged and the next reload retries. A tenant on no pool fails closed: `503` with `Retry-After: 30` on `POST /v1/query`, `GET/POST /v1/pipes/{name}`, `POST /v1/ops/query` and `POST /v1/ops/schema/refresh`, ahead of the cache. Each served tenant gets a `discovery.SchemaRegistry` of its own over its pool, kept fresh by its own loop — the boot retry until the first success, then `schema.refresh_interval` with the first refresh at a random point within the interval so tenants adopted together do not refresh together — created and stopped from the settings reload and stopped under `App.Close`; `wavehouse_schema_refresh_failures_total{tenant}` counts a loop's failed attempts. `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the strict `?tenant=` the pipe reads take (absent is tenant `0`, `400` malformed, `404` unknown, `503` rejected); the SDK sends it as the `tenant` option of `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()` and `wh.sql()`. Over a nested directory `/livez` (and `/v1/health`) is `503` with the latest discovery failure, naming its tenant, while no tenant has completed a first discovery, then `200` for the rest of the process lifetime; `/readyz` pings every open pool at once, is ready at the first answer, and names every pool that did not answer when none does — one tenant's ClickHouse outage is its log line and counter, never a probe failure. The ingest worker inserts each batch into its own tenant's ClickHouse — the tenant the message's topic names, through that tenant's HTTP wiring (`chconn.Pools.Target`); a tenant on no pool takes the failure path an unreachable ClickHouse takes — and its cache invalidation fans out to the tenants on the same address and database as the batch's tenant, whatever their user or `tls` block (they read the same tables), rather than to every known tenant, and a tenant adopted after an absence — rejected or removed, so out of that fan-out — or moved to another address or database has its cached structured-query results orphaned in one step (`Cache.InvalidateTenant`, a tenant generation in every version key), so a repaired folder never serves query rows cached before the inserts it missed (a pipe result names no table, so neither this nor any insert invalidates it: it stays until its TTL expires). Three changes reach the single-tenant directory too: a table lookup before the first discovery is a `503` with `Retry-After: 5` (`schema not loaded yet`) rather than a `404`, on `POST /v1/ingest`, `POST /v1/query` and `GET /v1/ops/schema` (the list included, where `[]` would read as no tables); the first periodic refresh fires at a random point within the interval rather than a full interval after boot; and a reload that moves `clickhouse.addr` or `clickhouse.database` now orphans the structured-query results cached before it, where they were served until their TTL. Until [#529](https://github.com/Wave-RF/WaveHouse/issues/529) every tenant's user authenticates with `WH_CH_PASSWORD`, so the tuple is in effect the address, database, username and `tls` block. - **One token verifier per tenant, built off the boot and reload paths** (`internal/auth/auth.go` (+ tests), `internal/api/router.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `go.mod`, `docs/src/content/docs/sdk/{reference,streaming}.md`, `docs/src/content/docs/{architecture,deployment,api}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `SECURITY.md`): story 9 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). Over a nested settings directory each tenant's folder now wires that tenant's verifier (`auth.jwks_url`, `auth.role_claim`), so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its identity provider's key set — under any other tenant's header it is refused as invalid — where before one verifier, tenant `0`'s, accepted a token under any header. `auth.Config` is now the boot-config half alone (the HMAC secret and the operator key, shared by every tenant) and the new `auth.Wiring` a tenant's half; `NewAuthenticator` builds no verifier, `Reconfigure(id, wiring)` gives a tenant one — swapped atomically when its wiring changed, kept when it did not — `Prune` drops the verifiers of the tenants a reload stopped serving, rejected or removed alike (no work runs for a tenant that is not served; a folder adopted again gets a fresh verifier), and `Close` stops every JWKS refresh as a component of `App.Close`. This holds on every reload because the registry's `AfterAdopt` hooks run after every reload it applied, empty list included, so a per-tenant reload that rejects a folder drops its verifier then rather than at the next adoption. The middleware reads the request tenant's verifier through an injected `auth.TenantSource` — the store `api.TenantMW` resolved names its tenant with `settings.Store.Tenant()`, one read of the context — a tenant-exempt route (the ops tree) verifies as tenant `0`, and a tenant with no verifier fails closed. The operator key stamps the request tenant's `admin_role`, read from that tenant's policy, rather than tenant `0`'s. A JWKS key set is fetched on its own goroutine: the verifier is in place at once and *pending* until a set has been stored — the first fetch retried with backoff from one second to a minute, a stored set then kept fresh by the library hourly and, rate-limited, on an unknown key id — so neither boot nor a reload (which holds the registry's lock) waits on the endpoint, and **an unreachable JWKS no longer refuses boot** — it logs (`jwks refresh failed; no token validates until it succeeds`) and that tenant alone is affected. A token checked against a pending verifier is a new outcome, `auth.ErrVerifierPending`: every `/v1` route answers it `503 {"error": "token verifier not ready: …"}` with `Retry-After: 30` (`api.refuseUnverifiable`) rather than evaluate the request under the `default_role`, which could accept its data under a lesser role while another pod holding the keys would have served it as its own; a request without a token, and the operator key, are unaffected. Every fetch goes through one client that caps the response at 1 MiB, refusing a larger one as unreachable with `jwks response exceeds 1048576 bytes` as the logged cause. Nothing changes for a flat directory beyond that boot rule: its one tenant gets exactly the verifier it had. The boot warning for the secretless posture (no `auth.jwt_secret` and no `jwks_url`, whose token refusal landed in [#607](https://github.com/Wave-RF/WaveHouse/pull/607)) is now per tenant — each served tenant with no `jwks_url` while the boot secret is unset — where one line that any JWKS tenant silenced used to stand for the whole directory. @@ -24,7 +24,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Schema discovery captures each table's DDL, its columns' ordinals and default expressions, and the server version** (`internal/discovery/discovery.go`, `internal/testutil/testutil.go`): `Column` gains `DefaultExpression` and `Position` (both from a widened `system.columns` select), `TableSchema` gains `DDL` from `system.tables.create_table_query`, and `SchemaRegistry` gains `ServerVersion()` from a `SELECT version()` probe next to the existing `SELECT timezone()`. Groundwork for the native type layer, captured on the same refresh as the columns so a stale version cannot outlive the schemas it describes. That is a publication guarantee, not a same-server one: `chconn.Manager` resolves the connection per call, so a reload changing `clickhouse.addr` mid-refresh can still pair a version from one server with schemas from another — narrow, and self-correcting on the next refresh. `DDL` is `json:"-"` and does **not** appear in `/v1/ops/schema`: that endpoint marshals `TableSchema` straight to the client, and an external-engine table (S3, MySQL, PostgreSQL, Kafka) renders its wiring there unconditionally — endpoint, bucket or host, database, username, S3 access key id. ClickHouse masks the password itself as `[HIDDEN]` from ~23.9 (verified on 26.7.3), so the exposure is the topology rather than the secret — except on an older server, or one with `display_secrets_in_show_and_select` enabled. `position` and `default_expression` are additive fields in the response. A table listed in `system.tables` with no `system.columns` rows is skipped rather than published column-less, and both new queries fail the refresh on error exactly as `timezone()` and `system.columns` do — callers keep the prior cache and retry. -- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the `WAVEHOUSE` and `WAVEHOUSE_DLQ` stream limits in place via `EmbeddedNATS.Resize` — shrinking below the buffered size backpressures until the worker drains, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on `WAVEHOUSE_DLQ` and ack; off → leave it unacked for redelivery, never dropped; the DLQ stream and `GET /v1/ops/dlq/stats` now always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** — every `config.json` key is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. +- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the worker drains, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** — every `config.json` key is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. - **"Was this page helpful?" feedback widget on every docs page** (`docs/src/components/PageFeedback.astro` (new), `docs/src/components/Footer.astro`): a thumbs-up / thumbs-down vote below the page content, captured to PostHog as `docs_feedback` with `{ helpful, page }`. It renders from `Footer.astro`'s sidebar branch — the same indirection the Cloud CTA uses — rather than a per-page import or frontmatter flag, so every content page gets it automatically, including ones not written yet; it sits *below* the Cloud CTA on the pages that carry one, and splash pages (the homepage and 404) take the other footer branch and never render it. One vote per page per visitor: the choice is remembered in `localStorage` keyed by pathname, and a revisit renders the thanks message instead of re-prompting (storage is a nicety, not the record — a browser with storage disabled still votes). - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. @@ -32,7 +32,9 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Changed -- **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). The sweeper keeps the longest `stream.gap_window_minutes` among the tenants being served: the ingest queue is one stream and a purge is one bound over it, so purging less is the safe direction until the streams are per tenant. `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. The subject change needs no drain of its own (the v2 envelope's drain, below, still applies): the durable consumers filter `ingest.>`, which the previous two-token subjects match, and a subject with no tenant token reads as tenant `0`'s, so the subject an event in flight arrived on changes nothing about how it is inserted, streamed, or parked, and rows already parked keep counting in `GET /v1/ops/dlq/stats` — which sums a table across tenants until the queue is per tenant; a gap-fill spanning the upgrade omits the pre-upgrade events for one gap window. Per-tenant JetStream streams, the DLQ shrink guard, and `?tenant=` on the DLQ stats route are story 5b, after a research spike. +- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/subscriber.go`, `internal/settings/{settings,store}.go`, `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch is shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes`, keeping no acknowledged history for a tenant no longer served, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. + +- **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. - **The query cache and its singleflight are keyed by tenant** (`internal/cache/{cache,local,version_manager}.go` (+ tests), `internal/settings/{store,registry}.go` (+ tests), `internal/api/{cache_key,pipes,structured_query}.go` (+ tests; `cache_tenant_test.go` new), `internal/ingest/worker.go` (+ tests), `docs/src/content/docs/{architecture,deployment,api,ingest-pipeline}.md`, `AGENTS.md`): story 8 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files — every key simply gains tenant `0`'s prefix. The tenant leads every key the cached read paths build: the key `POST /v1/query?table={table}` and `GET/POST /v1/pipes/{name}` cache a result under is `:query:` and is their singleflight key too, and a version namespace is `.
.
.` (`cache.Namespace` gains `Tenant`; scope stays where it was, and inert). So identical requests from two tenants are two entries and two flights to ClickHouse, a tenant is never served another tenant's cached rows, and a batch the ingest worker inserts bumps the namespaces of the tenant it was inserted for and no other's (`IngestWorker.invalidate` takes the tenant as a parameter — the worker's own until story 5 reads it off the message, so over a nested directory it is still tenant `0`'s namespaces every batch bumps, and another tenant's cached query results expire on their TTL alone until then). The pool stays one Ristretto instance sized by `cache.l1_max_cost`. The handlers read the tenant off the request's store — `settings.Store.Tenant`, stamped by the registry when it creates the store (the commit is shared with story 7) — so nothing new rides the request context. This lands ahead of story 6 on purpose: once each tenant has its own ClickHouse connection, a tenant-blind key would be a silent cross-tenant read. diff --git a/clients/ts/src/dlq.ts b/clients/ts/src/dlq.ts index a783e8eb..c1276225 100644 --- a/clients/ts/src/dlq.ts +++ b/clients/ts/src/dlq.ts @@ -1,7 +1,7 @@ import { err, ok } from "./errors.js"; -import { request } from "./http.js"; +import { request, tenantParam } from "./http.js"; import type { StreamController } from "./stream/controller.js"; -import type { DLQStats, HttpContext, Result, StreamOptions } from "./types.js"; +import type { DLQStats, HttpContext, OpsRequestOptions, Result, StreamOptions } from "./types.js"; type CreateStreamFn = (table: string, opts?: StreamOptions) => StreamController; @@ -15,23 +15,27 @@ export class DLQNamespace { this._createStream = createStream; } - /** Get DLQ statistics (message counts per table). */ - async list(opts?: { signal?: AbortSignal }): Promise> { + /** + * Get DLQ statistics (message counts per table) — of `opts.tenant`, the + * default tenant without it. A tenant with no dead-letter queue is a `404`. + */ + async list(opts?: OpsRequestOptions): Promise> { const { data, error } = await request(this._ctx, { method: "GET", path: "/v1/ops/dlq/stats", + params: tenantParam(opts), signal: opts?.signal, }); if (error) return err(error); return ok(data!); } - /** Get DLQ stats filtered by table name. */ - async table(name: string, opts?: { signal?: AbortSignal }): Promise> { + /** Get DLQ stats filtered by table name — of `opts.tenant`, the default tenant without it. */ + async table(name: string, opts?: OpsRequestOptions): Promise> { const { data, error } = await request(this._ctx, { method: "GET", path: "/v1/ops/dlq/stats", - params: { table: name }, + params: { table: name, ...tenantParam(opts) }, signal: opts?.signal, }); if (error) return err(error); diff --git a/clients/ts/src/namespaces.test.ts b/clients/ts/src/namespaces.test.ts index d1c0db5d..a32a9dfe 100644 --- a/clients/ts/src/namespaces.test.ts +++ b/clients/ts/src/namespaces.test.ts @@ -146,6 +146,25 @@ describe("DLQNamespace", () => { expect(fetchSpy.mock.calls[0][0]).toContain("table=clicks"); }); + it("list() and table() send opts.tenant as ?tenant=, and nothing without it", async () => { + fetchSpy.mockImplementation( + async () => new Response(JSON.stringify({ tables: {}, total: 0 }), { status: 200 }), + ); + const ns = new DLQNamespace(makeCtx(), mockStream); + + await ns.list({ tenant: "acme" }); + await ns.table("clicks", { tenant: "acme" }); + await ns.list(); + await ns.table("clicks"); + + const urls = fetchSpy.mock.calls.map((call) => new URL(call[0])); + expect(urls[0].pathname + urls[0].search).toBe("/v1/ops/dlq/stats?tenant=acme"); + expect(urls[1].searchParams.get("table")).toBe("clicks"); + expect(urls[1].searchParams.get("tenant")).toBe("acme"); + expect(urls[2].search).toBe(""); + expect(urls[3].search).toBe("?table=clicks"); + }); + it("stream() delegates to createStream", () => { const ctrl = {} as any; mockStream.mockReturnValue(ctrl); diff --git a/clients/ts/src/types.ts b/clients/ts/src/types.ts index 5feae56d..158d5c44 100644 --- a/clients/ts/src/types.ts +++ b/clients/ts/src/types.ts @@ -471,8 +471,8 @@ export interface PipeRequestOptions { /** * Options for a call to one of the admin routes that address a tenant: * `wh.pipes.list()`, `wh.pipes.get()`, `wh.settings.reload()`, - * `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()` and - * `wh.sql()`. + * `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()`, + * `wh.dlq.list()`, `wh.dlq.table()` and `wh.sql()`. */ export interface OpsRequestOptions { signal?: AbortSignal; diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 75311090..1634aaab 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -745,14 +745,16 @@ Triggers an immediate re-discovery of the `?tenant=`'s ClickHouse table schemas #### `GET /v1/ops/dlq/stats` — DLQ Statistics -Returns per-table message counts in the Dead Letter Queue — a table's count summed across tenants, since one queue serves every tenant until each has its own. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); the stream and this endpoint always exist. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. +Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The queue is read from the message queue rather than the settings, so a tenant whose folder was rejected or removed is read like one being served, for as long as its queue is kept. The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream exists from the moment the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. **Error responses:** | Status | Body | Cause | | ------ | ---- | ----- | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason) | +| 400 | `{"error":"invalid query string: …"}` / `{"error":"invalid ?tenant: …"}` | The query string does not parse (`?tenant=acme;x=1`, a bad `%` escape), or `tenant` is empty, repeated, or not a tenant id | | 403 | `{"error":"forbidden"}` | Caller's role is not the policy `admin_role` (`"admin"` by default) | +| 404 | `{"error":"no dead-letter queue for tenant: "}` | The tenant has no dead-letter queue: it has never been served on this data directory, or the id names no tenant | | 500 | `{"error":"stream info failed"}` | NATS JetStream stream-info lookup failed | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while tenant `0`'s JWKS has not been fetched yet (the ops tree verifies as tenant `0`); refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | @@ -760,6 +762,7 @@ Returns per-table message counts in the Dead Letter Queue — a table's count su | Param | Type | Default | Description | | ----- | ---- | ------- | ----------- | +| `tenant` | string | `0` | The tenant whose dead-letter queue is read. | | `table` | string | — | Filter stats to a specific table name (e.g., `?table=clicks` returns only the `clicks` count). | **Response:** @@ -873,9 +876,9 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the DLQ NATS stream (`WAVEHOUSE_DLQ`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format` (a pre-v2 message has no `format` field at all, which is how it presents here), or `columns` and `row` that do not pair — is parked without ever reaching a table batch, which is what an operator sees after upgrading across the wire change without draining first. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. +When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format` (a pre-v2 message has no `format` field at all, which is how it presents here), or `columns` and `row` that do not pair — is parked without ever reaching a table batch, which is what an operator sees after upgrading across the wire change without draining first. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. -Use `GET /v1/ops/dlq/stats` to monitor DLQ depth. +Use `GET /v1/ops/dlq/stats` to monitor DLQ depth, per tenant (`?tenant=`). ## Generating a JWT for Testing diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 9d1dc636..6eaf3d54 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -77,20 +77,20 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **router.go** — Route definitions. Public: `/livez`, `/readyz`, and the content-free `/v1/health` SDK ping (plus the permanent `/healthz` alias and the deprecated `/health`, `/ready` aliases). Policy-gated: `/v1/ingest?table={table}`, `/v1/query?table={table}` (structured), `/v1/pipes/{name}` (named pipes), `/v1/stream`. Admin-only (`RequireAdmin` — role == `policy.admin_role`, or a request bearing the operator key's operator bit, which passes even under a nil policy; over a nested settings directory `NewRouter` mounts the gate with no policy at all, whatever `Dependencies.PolicySource` was wired, so the operator key alone passes): `/v1/ops/schema/*`, `/v1/ops/dlq/stats`, `GET /v1/ops/pipes[/{name}]`, `/v1/ops/settings/reload`, `/v1/ops/query` (raw SQL — same gate as the rest of `/v1/ops/*`). - **auth middleware** — the JWT/JWKS authentication middleware is its own package, [`auth/`](#auth--authentication); the router runs it on every `/v1/*` route. -- **tenant.go** — `TenantMW` resolves the request's tenant ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: the [`X-Tenant-ID`](/deployment#multi-tenant-deployments) header (absent means `tenant.Default`), validated by `tenant.Parse` (`400`), looked up in the `settings.Registry` (`resolveStore`: `404` for an id it does not hold, a bare `503` for a tenant whose folder was rejected — the findings stay out of a body answered before authentication), and the resolved `*settings.Store` stored in the request context (`WithStore` / `StoreFromContext` — here rather than in `tenant/`, because `settings` names `tenant.ID`). A handler reads the store once and passes it down as an argument — the per-tenant getters it holds take it as a parameter (`(*settings.Store).Policy`, `.DedupeFor`, `.DefaultMaxRows`, … in production) — and nothing below a handler reads the context; a tenant route reached without a resolved store answers `500` rather than fall back to a tenant. The probes, `/version`, the metrics path, and `/v1/ops/*` are tenant-exempt; an ops route that addresses one tenant — the admin pipe reads, the schema routes, the raw-SQL proxy and the settings reload — names it in `?tenant=` (`opsTenant`, and `opsStore` over it for the routes that need the tenant's store), parsed strictly so that a query `url.ParseQuery` would half-read is a `400` rather than a read of the default tenant, which is what it means when absent on the reads (on the reload, absent is the whole directory). +- **tenant.go** — `TenantMW` resolves the request's tenant ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: the [`X-Tenant-ID`](/deployment#multi-tenant-deployments) header (absent means `tenant.Default`), validated by `tenant.Parse` (`400`), looked up in the `settings.Registry` (`resolveStore`: `404` for an id it does not hold, a bare `503` for a tenant whose folder was rejected — the findings stay out of a body answered before authentication), and the resolved `*settings.Store` stored in the request context (`WithStore` / `StoreFromContext` — here rather than in `tenant/`, because `settings` names `tenant.ID`). A handler reads the store once and passes it down as an argument — the per-tenant getters it holds take it as a parameter (`(*settings.Store).Policy`, `.DedupeFor`, `.DefaultMaxRows`, … in production) — and nothing below a handler reads the context; a tenant route reached without a resolved store answers `500` rather than fall back to a tenant. The probes, `/version`, the metrics path, and `/v1/ops/*` are tenant-exempt; an ops route that addresses one tenant — the admin pipe reads, the schema routes, the raw-SQL proxy, the settings reload and the DLQ stats — names it in `?tenant=` (`opsTenant`, and `opsStore` over it for the routes that need the tenant's store; the DLQ stats need none, since the MQ holds the queue), parsed strictly so that a query `url.ParseQuery` would half-read is a `400` rather than a read of the default tenant, which is what it means when absent on the reads (on the reload, absent is the whole directory). - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. `pipes.json` is the only write path. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ingest.go** — Accepts `POST /v1/ingest?table={table}` in three body shapes: one flat JSON object, a JSON array of them, or NDJSON. The **required** `Content-Type` chooses the format *family* — `application/json` versus the four NDJSON spellings — and within the JSON family the body's first non-whitespace byte picks array versus single object; the bytes never choose the family. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type — a comma inside a *quoted* parameter value is data, so `application/json; a=", application/x-ndjson; b="` is accepted. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer and runs the per-format record readers over those bytes, so the `413` lands before any record is processed and peak memory per request is O(body) rather than O(record). Then it validates each record against the discovered schema, optional dedup, and publishes each row through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row missing the configured `id_field` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). - **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` (the Go-side type conversion lives in the structured-query / pipes path, not here). - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). A stream ends, a gap-fill in progress included, when the server begins shutting down (`Closing`) or its `Subscriber` is evicted because its tenant is no longer served (`Hub.Prune`); one admitted just before the reload that stopped serving its tenant, and registered just after the prune, is ended right after it registers (`Served`). - **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` ahead of the cache, so nothing cached before is served. -- **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for the per-table parked counts (optionally one table) and the total. A dead-letter queue that does not exist (`mq.ErrNoDeadLetterQueue`) reads as empty; any other failure to read it is a 500. The queue itself is `internal/mq`'s. +- **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for one tenant's per-table parked counts (optionally one table) and its total — the tenant `?tenant=` names, read strictly by `opsTenant`, tenant `0` without it. The tenant is looked up in the MQ, not the settings registry, so a rejected or removed tenant's parked rows are read like a served one's; a tenant with no dead-letter queue (`mq.ErrNoDeadLetterQueue`) is a 404, and any other failure to read it a 500. The queue itself is `internal/mq`'s. - **health.go** — Liveness (`/livez`), readiness (`/readyz`), and a content-free `Online` ping (`/v1/health`, the SDK's public liveness check); `/healthz` is a permanent alias of `/livez`, and `/health`/`/ready` are deprecated aliases. All three consult an optional `BootState` so they can return 503 while boot-time schema discovery is still failing in the retry loop (see `internal/app`; over a nested directory, while no tenant's has succeeded); once `BootState.Set(nil)` fires, `/livez` returns 200 and stays there. `/readyz` additionally runs a `Ping` each call — `chconn.Pools.Ping` in production: every open pool at once, ready at the first answer, every pool's error joined when none answers; `/v1/health` deliberately does not. ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one resource a process still has one of, the MQ byte budget, follows the default tenant: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`; the zero value when a nested directory has never served a tenant `0`, warned about once at boot), and `onDefaultAdopt` runs its hook only after a reload that adopted it, so another tenant's reload never moves it and a `0` folder that a reload rejects or removes leaves it as it was — like the one setting read per request that follows tenant `0`, the ops gate's admin role. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. Two settings are shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)); and the sweeper keeps the longest `stream.gap_window_minutes` (`longestGapWindow`, read every sweep), since the ingest queue is one stream and a purge is one bound over it — a stream per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 5b) gives each its own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. The `mq.max_bytes_gb` hook only hands the adopted budget to `mq.Broker.SetMaxBytes` under the App's stop context; how it is split across the streams, the time bounds, and the rollback are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each served tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot and again after every reload, under the App's stop context, which opens that tenant's queue the first time; a queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out @@ -139,16 +139,16 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ - **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line, with `columns` naming its positions (the table's insertable columns — a computed one cannot be named in an `INSERT`); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow`. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy. On a bulk-insert failure the batch is re-inserted row by row — except a batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling), which no row could pass and `parkBatch` takes to the DLQ switch whole, logging once per batch rather than twice per row; rows that succeed are acked, and only the rows that fail again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. - **compact.go** — `EncodeCompactRow`, the positional row encoder every published row goes through, rendering one record over the table's **insertable** columns in declaration order. Serialization only: it validates nothing and judges no value. -- **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: the longest `stream.gap_window_minutes` among the tenants being served — `internal/app`'s `longestGapWindow`). Finding the purge point is `internal/mq`'s (`purge.go`). +- **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: each served tenant's own `stream.gap_window_minutes` — `internal/app`'s `gapWindows` — and none for a tenant no longer served). Finding the purge point is `internal/mq`'s (`purge.go`). ### `mq/` — Message Queue The **only** package that imports NATS/JetStream — a `depguard` rule in `.golangci.yml` fails `make lint` on any `github.com/nats-io` import in every package golangci-lint builds; the `integration`-tagged files under `tests/` sit outside its default build context, so the boundary there rests on convention (AGENTS.md Key Design Decision #20). Every other package talks to the broker through the types below, so a subject, stream, or broker change lands here once. -- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces: `Publisher` (`ErrQueueFull` when the ingest queue is at its byte budget — the API's 503 + `Retry-After`), `Subscriber` (every ingest event, under a named durable consumer — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`; `Consume` delivers on the client goroutine so a blocking handler is backpressure, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer or a closed connection — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic; the caller acks), `DeadLetterStats.DeadLetterCounts` (`ErrNoDeadLetterQueue` when there is none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before a cutoff; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with the byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. -- **subject.go** — The embedded broker's naming, private to the package: the stream names (`WAVEHOUSE`, `WAVEHOUSE_DLQ`), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded. A tail of one token is the form written before the tenant led it and reads as tenant `0`'s table, which is how the events in flight across that upgrade keep inserting, streaming, and counting. -- **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts the sweep. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. -- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream. Creates stream `WAVEHOUSE` with subjects `ingest.>`, capped at the settings directory's `mq.max_bytes_gb`, and stream `WAVEHOUSE_DLQ` (`dlq.>`, `DiscardOld`) at a tenth of it — always present, since an empty stream costs nothing. `SetMaxBytes` applies a reloaded budget to both live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. Its JetStream calls are bounded to ten seconds, plus five more for the rollback (a budget of its own, not the one that just expired), since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. +- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. +- **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. +- **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts the sweep. It runs on each tenant's stream at that tenant's cutoff. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. +- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, or is refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds, plus five more for the rollback (a budget of its own, not the one that just expired), since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. ### `observability/` — OpenTelemetry Pipeline diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index a9c9de1d..8a709426 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -97,7 +97,7 @@ WaveHouse's per-role caps are sent as per-query `SETTINGS` on its connection, so ### Message Queue (NATS) -The stream's disk budget, `mq.max_bytes_gb`, is a hot-reloadable key in the [Settings Directory](/settings-directory#message-queue) — there is no boot-config knob for it. +Each tenant's queue has its own disk budget, `mq.max_bytes_gb`, a hot-reloadable key in the [Settings Directory](/settings-directory#message-queue) — there is no boot-config knob for it. **Durability.** The embedded server runs with JetStream `SyncAlways`, so every event is `fsync`'d to disk before `POST /v1/ingest` returns `200`. This makes your storage's `fsync` latency your ingest latency floor — see [Durability & Storage](/durability) to check whether your substrate can sustain it. There is no knob to relax this today ([#139](https://github.com/Wave-RF/WaveHouse/issues/139) tracks a configurable group-commit interval). diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 020c676c..095b4990 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -379,19 +379,19 @@ settings/ └── roles.json ``` -That is the layout a control plane writes. Each folder's `clickhouse` block is its tenant's own ClickHouse, so a tenant answers queries once its first schema discovery against that ClickHouse succeeds (until then its schema-aware routes answer `503`, `schema not loaded yet`); what tenant `0`'s folder still supplies for the whole process — the message queue's budget, the token verifier of the routes that name no tenant, their CORS list — is listed under "What a tenant's folder decides", below. +That is the layout a control plane writes. Each folder's `clickhouse` block is its tenant's own ClickHouse, so a tenant answers queries once its first schema discovery against that ClickHouse succeeds (until then its schema-aware routes answer `503`, `schema not loaded yet`); what tenant `0`'s folder still supplies for the whole process — the token verifier of the routes that name no tenant, their CORS list — is listed under "What a tenant's folder decides", below. The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe store needs no restructuring: it keys every tenant's seen ids by tenant, and the four files are tenant `0`, as a `0` folder is. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id is a finding of its own — that folder is skipped, and the rest of the directory still loads. -**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. +**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). Its message queue is kept, at the budget it last had, but the history gap-fill replays is purged from it at the next sweep, as a removed tenant's is, so a stream resumed after the fix has a hole where that history was. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. -**Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. That is how a tenant is removed: delete its folder, then reload the whole directory. Its open streams end, its routes answer `404`, and its queued rows are parked on the DLQ under its own subject; nothing it stored is deleted, so restoring the folder restores the tenant, seen ids included. Reloading a deleted folder by name instead leaves its tenant rejected, answering `503`. The last folder can be removed the same way, with two catches, since `wavehouse validate` and boot both read an emptied directory as the four files missing: `validate` exits `1`, so a writer that gates each reload on it has to skip the check for that one reload, and a server restarted before a folder is written back refuses to boot. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. +**Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. That is how a tenant is removed: delete its folder, then reload the whole directory. Its open streams end, its routes answer `404`, and its queued rows are parked on the DLQ under its own subject; nothing it stored is deleted — its message queue is kept at the budget it last had, and only the history gap-fill replays goes from it, at the next sweep — so restoring the folder restores the tenant, seen ids and parked rows included. Reloading a deleted folder by name instead leaves its tenant rejected, answering `503`. The last folder can be removed the same way, with two catches, since `wavehouse validate` and boot both read an emptied directory as the four files missing: `validate` exits `1`, so a writer that gates each reload on it has to skip the check for that one reload, and a server restarted before a folder is written back refuses to boot. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. -**The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` reads the queue the whole process shares and ignores the parameter. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). +**The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` takes it too, and reads a rejected or removed tenant's dead-letter queue like a served one's, since the queue is kept; a tenant that has none is a `404`. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. The process still has one message queue, and its budget, `mq.max_bytes_gb`, follows tenant `0`'s folder. The queue is shared but addressed per tenant: an event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool too, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. Two settings weigh every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`; and the sweeper, which keeps the longest `stream.gap_window_minutes` among them, since every tenant's events share one message-queue stream and a purge is one bound over it. +**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. -**What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. `mq.max_bytes_gb` stays as tenant `0` last adopted it; tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. The sweeper keeps the longest gap window among the tenants still served, so tenant `0`'s history is purged at theirs, and with no tenant left being served the window is zero, which purges the acknowledged history gap-fill replays. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. +**What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. Tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. ### Upgrading behind a proxy that already sends `X-Tenant-ID` @@ -440,13 +440,9 @@ To drain before upgrading: If you skipped the drain, check `wavehouse_ingest_poison_total`, which counts both — `disposition="parked"` is recoverable from `dlq.{table}`, `disposition="dropped"` is gone — see [Dead Letter Queue](#dead-letter-queue-dlq) below. -## Upgrading across the tenant subject token - -Message-queue subjects now lead with the tenant: `ingest.{tenant}.{table}` and `dlq.{tenant}.{table}`, where a settings directory that holds the four files is tenant `0` (`ingest.0.clicks`). The subject change needs no drain of its own; the drain [the envelope upgrade above](#upgrading-across-the-v2-ingest-envelope) asks for still applies. The durable consumers filter `ingest.>`, which the previous subjects match, and a subject with no tenant token reads as tenant `0`'s, so the subject a message arrived on changes nothing about how it is inserted, streamed, or parked, and rows parked under the old `dlq.{table}` keep counting in `GET /v1/ops/dlq/stats`. The one gap is SSE gap-fill, which reads a tenant's own subject: a replay spanning the upgrade omits the events published before it, for one `stream.gap_window_minutes` (15 by default) — the same window as the envelope upgrade's. Clients that need them should backfill over REST. - ## Dead Letter Queue (DLQ) -A failed batch insert is retried row by row; while the tenant's `dlq.enabled` is `true` for the table (the seed default — a hot-reloadable [settings directory](/settings-directory#dead-letter-queue) key, overridable per table), the rows that fail again are published to the `WAVEHOUSE_DLQ` NATS stream under subjects `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) instead of retrying forever. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for, such as by the connection ceiling — skips the row-by-row retry, which no row of it could pass: its tenant's switch is read once for the whole batch, and a tenant no longer served has no switch to read, so its batch is always parked. Monitor DLQ depth via `GET /v1/ops/dlq/stats`. +A failed batch insert is retried row by row; while the tenant's `dlq.enabled` is `true` for the table (the seed default — a hot-reloadable [settings directory](/settings-directory#dead-letter-queue) key, overridable per table), the rows that fail again are published to the tenant's own dead-letter stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) instead of retrying forever. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for, such as by the connection ceiling — skips the row-by-row retry, which no row of it could pass: its tenant's switch is read once for the whole batch, and a tenant no longer served has no switch to read, so its batch is always parked. Monitor DLQ depth via `GET /v1/ops/dlq/stats`, per tenant (`?tenant=`; tenant `0` without it). ## Observability diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 8548ab96..98a8c9c2 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -33,8 +33,8 @@ WaveHouse does not currently expose a knob to relax this — `SyncAlways` is alw Because the publish blocks on `fsync`, **your typical ingest latency is your storage's typical `fsync` latency, and your worst-case publish is your storage's worst-case `fsync`.** When that tail is healthy (sub-millisecond to single-digit milliseconds) the guarantee is essentially free. When it is not, the same code path that handles every production message stalls: - Publishes block for the duration of the `fsync`, so a multi-second `fsync` tail is a multi-second ingest tail. -- The embedded server's stream/consumer setup and every publish run under the JetStream client's request timeout; a slow-enough substrate makes them exceed it. The boot-time symptom is `create stream: ... context deadline exceeded`. -- If the worker cannot drain to ClickHouse faster than producers publish, the stream fills toward [`mq.max_bytes_gb`](/settings-directory#message-queue) and the API returns `503` ([backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs)). +- The embedded server's stream/consumer setup and every publish run under the JetStream client's request timeout; a slow-enough substrate makes them exceed it. The symptom at a first boot, which opens every tenant's queue, is `open dlq stream: ... context deadline exceeded`; a later boot writes nothing, so the first publish is where it shows. +- If the worker cannot drain to ClickHouse faster than producers publish, a tenant's stream fills toward its [`mq.max_bytes_gb`](/settings-directory#message-queue) and the API returns `503` to that tenant ([backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs)). ## Where `SyncAlways` is cheap vs. expensive @@ -100,6 +100,6 @@ If you see any of these, benchmark the `/nats` volume as above: ## See also -- [Settings Directory → Message Queue](/settings-directory#message-queue) — `mq.max_bytes_gb`, the stream's disk budget (hot-reloadable); the SSE gap window inside it is [`stream.gap_window_minutes`](/settings-directory#streaming). +- [Settings Directory → Message Queue](/settings-directory#message-queue) — `mq.max_bytes_gb`, each tenant's queue's disk budget (hot-reloadable); the SSE gap window inside it is [`stream.gap_window_minutes`](/settings-directory#streaming). - [Deployment → Persistent Storage](/deployment#persistent-storage-required-for-containers) — `data_dir` must resolve to a host-backed volume. - [Ingest Pipeline → Backpressure and durability knobs](/ingest-pipeline#backpressure-and-durability-knobs) — the worker-side ack cost and the in-flight backpressure layers. diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index 870133ca..e602e0cf 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -22,15 +22,15 @@ The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the q ## High-level shape -One process consumes a single durable JetStream consumer and fans events out to a goroutine per tenant table — the tenant is the subject's leading token. Each tenant's table batches independently and POSTs to ClickHouse over the HTTP interface (`JSONCompactEachRow`). On a bulk-insert failure the batch is re-inserted row by row, so a single poison row can't sink it: clean rows ack, and only the rows that fail again go to the dead-letter stream. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and meets the dead-letter switch once, whole; a tenant no longer served has no switch to read, so its batch is parked. An envelope the worker cannot *read* — malformed JSON, an unknown row `format` (what a pre-v2 message looks like), or columns and a row that don't pair — never reaches a table loop at all: `parseMsg` parks it on the same dead-letter stream, or, where the DLQ is off for the table, acks and drops it rather than redelivering a message that can never insert. A separate sweeper reclaims stream storage. +Each tenant's events are queued on a JetStream stream of its own. One process holds one durable consumer on each tenant's stream, delivered into one handler, and fans events out to a goroutine per tenant table — the tenant is the subject's leading token. Each tenant's table batches independently and POSTs to ClickHouse over the HTTP interface (`JSONCompactEachRow`). On a bulk-insert failure the batch is re-inserted row by row, so a single poison row can't sink it: clean rows ack, and only the rows that fail again go to the dead-letter stream. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and meets the dead-letter switch once, whole; a tenant no longer served has no switch to read, so its batch is parked. An envelope the worker cannot *read* — malformed JSON, an unknown row `format` (what a pre-v2 message looks like), or columns and a row that don't pair — never reaches a table loop at all: `parseMsg` parks it on the same dead-letter stream, or, where the DLQ is off for the table, acks and drops it rather than redelivering a message that can never insert. A separate sweeper reclaims stream storage. ```mermaid flowchart LR API["POST /v1/ingest"] -->|"publish ingest.TENANT.TABLE"| Stream subgraph NATS["Embedded NATS JetStream (in-process)"] - Stream["WAVEHOUSE stream
all ingest subjects
LimitsPolicy + DiscardNew"] - Cons["buffer-consumer
(durable, pull)"] + Stream["INGEST_TENANT stream, one per tenant
ingest.TENANT.>
LimitsPolicy + DiscardNew"] + Cons["buffer-consumer
(durable, pull, one per tenant stream)"] Stream --> Cons end @@ -46,7 +46,7 @@ flowchart LR TLa -->|"JSONCompactEachRow POST"| CH[("ClickHouse")] TLb --> CH TLc --> CH - TLa -.->|"poison rows"| DLQ["WAVEHOUSE_DLQ
dlq.TENANT.TABLE"] + TLa -.->|"poison rows"| DLQ["DLQ_TENANT stream
dlq.TENANT.TABLE"] D -.->|"unreadable envelope"| DLQ Sweep["Active Sweeper"] -.->|"reads AckFloor, purges"| Stream @@ -204,7 +204,7 @@ Messages still sitting in `msgChan` or the consumer's prefetch buffer at shutdow Delivery can end underneath a running worker: the durable consumer is deleted, or the MQ connection closes. The broker client reports that only through an asynchronous error callback and then stops delivering — no message ever arrives to say so, so a loop that only watches `msgChan` would wait forever while the API kept accepting events nothing writes. `mq.Consumer.Consume` therefore returns a `failed` channel next to `stop` (`mq.ErrDeliveryEnded`, wrapping the broker's reason), and `dispatchLoop` selects on it beside `ctx.Done()` and `msgChan`. On a failure it runs the same bottom-up drain as a shutdown — the rows already in hand are flushed and acked, not abandoned — and then reports the error on the worker's own `failed` channel. A consumer that cannot start at all takes the same path. -The worker does not try to revive the consumer. The app's ingest-worker component returns the error from `app.Run`, which stops every other component and exits non-zero, the same way any failed component does; the supervisor's restart recreates the durable consumer at boot, and everything unacked is redelivered (at-least-once). Passing conditions the client also reports through that callback (a missed heartbeat, a leadership change) are logged at `WARN` and do not end the worker. With the embedded broker (`DontListen`, no external client that could delete the durable) this path is hard to reach today; it matters once a remote broker or per-tenant consumers exist. +The worker does not try to revive the consumer. The app's ingest-worker component returns the error from `app.Run`, which stops every other component and exits non-zero, the same way any failed component does; the supervisor's restart recreates the durable consumer at boot, and everything unacked is redelivered (at-least-once). Passing conditions the client also reports through that callback (a missed heartbeat, a leadership change) are logged at `WARN` and do not end the worker. With the embedded broker (`DontListen`, no external client that could delete a durable) this path is hard to reach; the likeliest way in is a tenant's queue, opened at runtime, that the consumer cannot join. It matters more once a remote broker exists. ## Backpressure and durability knobs @@ -212,23 +212,23 @@ Several layers throttle the pipeline, inner to outer: 1. **`batch`** flushes at `maxBatch` rows or `maxWait`. 2. **`msgChan`** (cap `maxBatch*2`) — when full, the consume callback blocks and delivery pauses. -3. **`pullMaxMessages`** — nats.go's client-side prefetch buffer in front of `msgChan`. -4. **`maxAckPending`** — the server suspends delivery once this many messages are delivered-but-unacked. The outermost in-memory bound. -5. **`MaxBytes` + `DiscardNew`** on the stream (`mq.max_bytes_gb` in the [settings directory](/settings-directory#message-queue), resized in place on reload) — when disk fills (e.g. ClickHouse is down so nothing acks/purges), new publishes are rejected and the API returns 503. +3. **`pullMaxMessages`** — nats.go's client-side prefetch buffer in front of `msgChan`, shared by the tenants' streams (at least one message each). +4. **`maxAckPending`** — the server suspends a tenant's delivery once this many of its messages are delivered-but-unacked; no other tenant's delivery waits on it. The outermost in-memory bound. +5. **`MaxBytes` + `DiscardNew`** on each tenant's stream (its `mq.max_bytes_gb` in the [settings directory](/settings-directory#message-queue), resized in place on reload) — when it fills (e.g. ClickHouse is down so nothing acks/purges), that tenant's new publishes are rejected and the API returns 503. | Knob | Default | Meaning / invariant | | --- | --- | --- | | `maxBatch` | 500 | rows that trigger a flush (soft — coalescing can exceed it) | | `maxWait` | 5s | max time a row waits before its batch flushes | | `ackWait` | 60s | server redelivery timeout; **must exceed `maxWait` + flush time** or in-flight rows get redelivered → duplicate inserts | -| `pullMaxMessages` | 500 | client prefetch; keep `<= maxAckPending` | -| `maxAckPending` | 10,000 | server cap on unacked messages (backpressure) | +| `pullMaxMessages` | 500 | client prefetch, shared by the tenants' streams; keep `<= maxAckPending` | +| `maxAckPending` | 10,000 | server cap on a tenant's unacked messages (backpressure) | `DoubleAck` is used (not fire-and-forget `Ack`) because acking is what records "this data is durably in ClickHouse." With the embedded server's `SyncAlways`, every ack is an fsync and therefore *slow*, which is exactly why acks run in the background (`ackWg`) off the insert path. ## The Active Sweeper -The worker advances the consumer's `AckFloor` by acking; the sweep observes it to decide what is safe to purge. They never call each other — the consumer's `AckFloor` is their only contract. The sweeper (`internal/ingest`) owns the schedule and the window: each tick it calls `mq.Purger.PurgeAcked(buffer-consumer, now − gap window)`, where the window is the longest `stream.gap_window_minutes` among the tenants being served — every tenant's events share one stream and a purge is one bound over it, so purging less is the safe direction until each tenant has its own stream. The steps after the tick below are the embedded broker's implementation of that call. +The worker advances the consumer's `AckFloor` by acking; the sweep observes it to decide what is safe to purge. They never call each other — the consumer's `AckFloor` is their only contract. The sweeper (`internal/ingest`) owns the schedule and the window: each tick it calls `mq.Purger.PurgeAcked(buffer-consumer, cutoffs)` with each served tenant's cutoff at now − its own `stream.gap_window_minutes`; a tenant no longer served — its folder removed or rejected — is given none, and keeps none of the history it has acknowledged. The steps after the tick below are the embedded broker's implementation of that call, run on each tenant's stream at that tenant's cutoff. ```mermaid flowchart TD @@ -236,7 +236,7 @@ flowchart TD Read --> Gap["binary-search the gap-window sequence"] Gap --> Target["target = MIN(ackFloor + 1, gapSeq)"] Target --> Purge["stream.Purge below target"] - Purge -->|"deletes msgs that are BOTH
written to ClickHouse AND past the gap window"| Stream[("WAVEHOUSE stream")] + Purge -->|"deletes msgs that are BOTH
written to ClickHouse AND past the gap window"| Stream[("INGEST_TENANT stream")] ``` `MIN(ackFloor+1, gapSeq)` is the safety argument: never purge past what is in ClickHouse, and never past the SSE replay window. If ClickHouse is down the `AckFloor` stops advancing, purging freezes, and the stream fills toward `MaxBytes` — backpressure by construction. The sweeper is one of `app.Run`'s components (`Sweeper.Start` blocks until the run context is canceled), but an interrupted sweep is harmless and idempotent, so it returns on `ctx.Done()` with no drain of its own — unlike the worker's bounded `stopFunc`. @@ -248,7 +248,7 @@ Today this is a **single-process** design (embedded, in-process NATS — the "co ```mermaid flowchart TD subgraph Cluster["Clustered NATS (Replicas: 3)"] - S["WAVEHOUSE stream"] + S["one shared ingest stream"] end S --> P0["partition 0"] S --> P1["partition 1"] diff --git a/docs/src/content/docs/sdk/admin.md b/docs/src/content/docs/sdk/admin.md index ae1635f7..59db0e5a 100644 --- a/docs/src/content/docs/sdk/admin.md +++ b/docs/src/content/docs/sdk/admin.md @@ -65,6 +65,13 @@ const { data } = await wh.dlq.list(); const { data } = await wh.dlq.table('clicks'); ``` +Each tenant has a dead-letter queue of its own, and the calls read tenant `0`'s without `tenant`. Over [a nested settings directory](/deployment#the-nested-settings-directory), pass `tenant` to read another's — a tenant whose folder was rejected or removed included, since its queue is kept — with the [operator key](/api#authentication), as for the schema reads above. A tenant with no dead-letter queue is a `404`: + +```ts +const { data } = await wh.dlq.list({ tenant: 'acme' }); +const { data: clicks } = await wh.dlq.table('clicks', { tenant: 'acme' }); +``` + `wh.dlq.stream()` exists in the API but is **not yet functional**: there is no server-side DLQ stream today (the SSE bridge only carries `ingest.>` subjects), so it connects and receives no events rather than failing. Live DLQ streaming is tracked in [#197](https://github.com/Wave-RF/WaveHouse/issues/197). --- diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index 56fc09d9..af0cddef 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -104,8 +104,8 @@ createClient(config) → WaveHouseClient ├── .settings (admin) │ └── .reload(opts?) → Promise> ├── .dlq (admin) -│ ├── .list() → Promise> -│ ├── .table(name) → Promise> +│ ├── .list(opts?) → Promise> +│ ├── .table(name, opts?) → Promise> │ └── .stream() → StreamController // not yet functional server-side — #197 └── .sys └── .health() → Promise> diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 9555bf79..c0e7a19b 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -31,7 +31,7 @@ A reload that fails validation is logged (and reported by the endpoint) and the "Previous good settings" is the in-memory snapshot of the running process, nothing more: there is no persisted copy of the files. A restart re-validates the directory from scratch and refuses to start on the same findings the reload rejected, so bad files never survive a restart silently — fix them (or run `wavehouse validate`) before bouncing the server. -A directory that holds one folder per tenant instead of the four files is [a nested settings directory](/deployment#the-nested-settings-directory): each folder is everything this page describes, but it is not watched, a rejected folder stops its tenant being served rather than keeping the previous settings, and the keys the whole process shares are not read from the tenant's own folder: they come from tenant `0`'s, bar the two that weigh every tenant being served: the SSE keepalive, which follows the shortest `stream.keepalive_interval` among them, and the sweeper's gap window, the longest `stream.gap_window_minutes` among them — that section lists which keys. +A directory that holds one folder per tenant instead of the four files is [a nested settings directory](/deployment#the-nested-settings-directory): each folder is everything this page describes, but it is not watched, a rejected folder stops its tenant being served rather than keeping the previous settings, and the keys the whole process shares are not read from the tenant's own folder: they come from tenant `0`'s, bar the one that weighs every tenant being served: the SSE keepalive, which follows the shortest `stream.keepalive_interval` among them — that section lists which keys. Every adoption — boot and every reload — goes through the same `Validate`, so the policy, the roles, and the pipes are checked with the current rules each time they are read; there is no stored copy that can skip validation. All four files are adopted as one snapshot: a request is evaluated against the policy, pipes, and tunables of a single adoption, never a mix. @@ -124,7 +124,7 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `dedupe.id_field` | `event_id` | Dedup key field — see [Deduplication](#deduplication). | | `dedupe.require_id` | `false` | Reject rows missing the id field — see [Deduplication](#deduplication). | | `dedupe.tables.
.{id_field, require_id}` | `{}` | Optional per-table overrides; each entry overrides only the fields it names and inherits the rest. | -| `dlq.enabled` | `true` | Park poison rows — those that still fail after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the `WAVEHOUSE_DLQ` stream (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | +| `dlq.enabled` | `true` | Park poison rows — those that still fail after row-by-row isolation, and every row of a batch whose tenant has no ClickHouse connection — on the tenant's dead-letter stream (`DLQ_{tenant}`) (`false`: leave them unacked for redelivery — except an envelope the worker cannot read, which is dropped and counted) — see [Dead Letter Queue](#dead-letter-queue). | | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | | `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that a structured query's relative time range is truncated to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | | `query.default_max_rows` | `10000` | Fallback result `LIMIT` (`>= 1`) applied to a structured query when the caller and policy specify none. A result-**shaping** default, not a resource limit — server-wide limits (memory, rows scanned, execution time) belong in ClickHouse, see [Server-side resource limits](/configuration#server-side-resource-limits). | @@ -132,7 +132,7 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `stream.keepalive_interval` | `30` | Seconds (`>= 1`) a quiet `GET /v1/stream` connection may go without a write before the server sends a `:` keepalive comment — keep it under your proxy's idle timeout; see [Streaming](#streaming). | | `stream.keepalive_buckets` | `3` | Load-spreading (`>= 1`): connections are spread across N buckets so each tick nudges ~1/N of live streams. Most deployments leave it. | | `stream.gap_window_minutes` | `15` | Minutes (`>= 0`) of written-to-ClickHouse history the Active Sweeper keeps in NATS for `Last-Event-ID` gap-fill; applies from the next sweep. | -| `mq.max_bytes_gb` | `50` | Disk budget (GB, `>= 1`) for the embedded NATS `WAVEHOUSE` ingest stream; the `WAVEHOUSE_DLQ` stream gets a tenth of it. A reload updates the live streams in place. See [Message Queue](#message-queue). | +| `mq.max_bytes_gb` | `50` | Disk budget (GB, `>= 1`) for the tenant's embedded NATS ingest stream (`INGEST_{tenant}`); its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. A reload updates the live streams in place. See [Message Queue](#message-queue). | | `cors.allowed_origins` | `["*"]` | Allowed CORS origins, applied per request. `"*"` allows any browser origin. WaveHouse is a Bearer-token API — `Access-Control-Allow-Credentials` is intentionally never sent, so this allowlist controls *which origins can read responses*, not cookie scope. Tighten to your frontend's exact origin(s) in production (e.g. `["https://dashboard.example.com", "http://localhost:3000"]`). An empty list `[]` denies every browser origin (no `Access-Control-Allow-Origin` is ever sent); `"*"` is the only allow-all spelling. Over [a nested settings directory](/deployment#the-nested-settings-directory) each tenant's list decorates its own responses, the preflight included; which list answers a preflight, the tenant-exempt routes, and a refused request is [spelled out there](/deployment#multi-tenant-deployments). | ```json @@ -212,16 +212,16 @@ The `auth` block is the verifier wiring, minus the secrets. `jwks_url` (absolute A failed batch insert is retried row by row; a row that fails again on its own is a poison row. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the retry, which no row of it could pass, and every row of it is a poison row. `dlq.enabled` (seed default `true`) decides what happens to it, resolved per table (`dlq.tables.
.enabled` → global) at the moment of the failure, so a reload applies to the next poison row: -- `true` — the row is published to the `WAVEHOUSE_DLQ` NATS stream under `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) with the failure in its headers, and its original is acked. Inspect it with `GET /v1/ops/dlq/stats` (admin-only). +- `true` — the row is published to the tenant's dead-letter stream (`DLQ_{tenant}`) under `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) with the failure in its headers, and its original is acked. Inspect it with `GET /v1/ops/dlq/stats` (admin-only; `?tenant=` names the tenant). - `false` — the row is left unacked, so NATS redelivers it and it retries until it inserts or the switch is flipped back. For every row the worker **can read**, nothing is ever dropped either way — the choice is *park it* versus *keep retrying*. **One exception, new in this release:** an envelope the worker cannot read *at all* — malformed JSON, an unknown `format` (what a pre-v2 in-flight message looks like), or `columns` and `row` that do not pair — can never insert, so redelivering it forever would wedge the consumer. With the DLQ off for the table it is acked and **dropped**, logged at `ERROR` and counted by `wavehouse_ingest_poison_total` with `disposition="dropped"` (also labeled by `table` and `reason`; an envelope parked on the DLQ carries `disposition="parked"`). See [Ingest Pipeline](/ingest-pipeline) — and drain the ingest queue before upgrading. -For a tenant no longer served — its folder removed or rejected — there is no switch to read: its rows are always parked, so none of them sits unacked in the shared ingest queue, where it would stop the [Active Sweeper](/ingest-pipeline#the-active-sweeper) purging it. +For a tenant no longer served — its folder removed or rejected — there is no switch to read: its rows are always parked, so none of them sits unacked in its ingest queue, redelivered for as long as the tenant is away and stopping the [Active Sweeper](/ingest-pipeline#the-active-sweeper) purging that queue. -The `WAVEHOUSE_DLQ` stream always exists (an empty stream costs nothing) and the stats endpoint is always registered — the switch is purely behavioral, which is what makes it safe to reload. +A tenant's dead-letter stream exists from the moment the tenant is first served (an empty stream costs nothing) and the stats endpoint is always registered — the switch is purely behavioral, which is what makes it safe to reload. ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the embedded JetStream `WAVEHOUSE` stream that buffers ingested events until the worker writes them to ClickHouse; the `WAVEHOUSE_DLQ` stream gets a tenth of it. The stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the stream refuse new publishes until the worker drains it back under the limit — nothing already accepted is dropped. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Size it from [Durability & Storage](/durability). +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the worker drains it back under the limit — nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to — so a queue fills until its budget or the disk runs out, whichever comes first; size them together from [Durability & Storage](/durability). ## Streaming @@ -229,4 +229,4 @@ The `WAVEHOUSE_DLQ` stream always exists (an empty stream costs nothing) and the - `stream.keepalive_interval` (seed default `30`) — seconds a quiet connection may go without a write before the server sends a `:` keepalive comment. It exists to stay under whatever idle timeout sits between WaveHouse and the client; the default clears the common 55–60s proxy windows with margin, and a tighter edge (Azure Application Gateway 20s, CloudFront 30s) wants a lower value — see [Behind a reverse proxy → Idle timeouts](/reverse-proxy#idle-timeouts-by-provider). A reload rebuilds the keepalive wheel in place: live connections stay open and are redistributed across the new ring, each getting at most one full new period before its next keepalive. - `stream.keepalive_buckets` (seed default `3`) — spreads the keepalive writes across the interval (one bucket fires every `keepalive_interval ÷ keepalive_buckets`) so the server nudges ~1/N of connections per tick instead of all at once. It changes only how the writes are spread in time, never the period. -- `stream.gap_window_minutes` (seed default `15`) — minutes of already-written-to-ClickHouse history the Active Sweeper keeps in NATS so a reconnecting client's `Last-Event-ID` replay can bridge the gap; a drop longer than this resumes with a hole. Bounded by the stream's disk budget, [`mq.max_bytes_gb`](#message-queue). A reload applies from the next sweep (every minute). +- `stream.gap_window_minutes` (seed default `15`) — minutes of already-written-to-ClickHouse history the Active Sweeper keeps in NATS so a reconnecting client's `Last-Event-ID` replay can bridge the gap; a drop longer than this resumes with a hole. Bounded by the tenant's disk budget, [`mq.max_bytes_gb`](#message-queue). A reload applies from the next sweep (every minute). diff --git a/docs/src/content/docs/why-wavehouse.md b/docs/src/content/docs/why-wavehouse.md index a5b63e21..26ac9d70 100644 --- a/docs/src/content/docs/why-wavehouse.md +++ b/docs/src/content/docs/why-wavehouse.md @@ -53,7 +53,7 @@ Even if you remember to batch client-side, a naive ingest path has no safe way t - **No backpressure channel.** If the merger falls behind, ClickHouse raises an error at the *next* insert. The client has already left. - **No DLQ.** Bad events that fail to insert are either lost or logged into ClickHouse's error log. Good luck replaying yesterday's dropped rows. -WaveHouse fixes all three at the gateway: validates every payload against the real `system.columns` schema before accepting, returns `503 Service Unavailable` with a `Retry-After` header when the NATS WAL fills, and routes failed batch inserts to a dedicated `WAVEHOUSE_DLQ` stream you can inspect via `GET /v1/ops/dlq/stats`. +WaveHouse fixes all three at the gateway: validates every payload against the real `system.columns` schema before accepting, returns `503 Service Unavailable` with a `Retry-After` header when the NATS WAL fills, and routes failed batch inserts to a dedicated dead-letter stream, one per tenant, you can inspect via `GET /v1/ops/dlq/stats`. ### No real-time push @@ -156,7 +156,7 @@ flowchart TB | Real-time push | WebSocket service + bridge from Kafka | Built in (`/v1/stream`) | | Schema validation | Custom code in ingest API | Built in (discovers `system.columns`) | | Row/column access control | Custom middleware or a dedicated service | Built in (Hasura-style, JWT-driven) | -| Dead letter queue | Custom retry + dead topic on Kafka | Built in (`WAVEHOUSE_DLQ`) | +| Dead letter queue | Custom retry + dead topic on Kafka | Built in (a dead-letter stream per tenant) | | Client SDK | Each team writes one | `@wavehouse/sdk` (TypeScript, one dependency, codegen) | The DIY path works — big teams run it — but the ops cost is not small. You're paying for a Kafka cluster (or Confluent bill), a second service you wrote from scratch, and all the debugging hours when the batching consumer stalls at 3 a.m. @@ -190,7 +190,7 @@ Tinybird wins on "zero ops to start." WaveHouse wins on "own your data plane and | Self-hosted | ✓ | ✓ | ✗ | ✓ | | Handles N-row inserts safely | ✗ merge blowup | ✓ via Kafka | ✓ | ✓ native | | Schema validation at the edge | ✗ | Custom | ✓ | ✓ (discovers schema) | -| Dead letter queue | ✗ | Custom | Partial | ✓ `WAVEHOUSE_DLQ` | +| Dead letter queue | ✗ | Custom | Partial | ✓ dead-letter stream per tenant | | Backpressure (503 + Retry-After) | ✗ | Custom | ✓ | ✓ | | Idempotent ingest (dedup by ID) | ✗ | Custom | ✓ | ✓ optional | | Real-time push (SSE) | ✗ | Custom service | ✗ | ✓ native, gap-fill | @@ -221,7 +221,7 @@ flowchart TB NATS --> BC["Buffer consumer
5-second batches"]:::wh BC --> CH[("ClickHouse")]:::store - BC -. "on failure" .-> DLQ["WAVEHOUSE_DLQ"]:::fail + BC -. "on failure" .-> DLQ["dead-letter stream"]:::fail ``` **Query path with tiered cache:** diff --git a/internal/api/dlq.go b/internal/api/dlq.go index 9de69ad5..267d2dc4 100644 --- a/internal/api/dlq.go +++ b/internal/api/dlq.go @@ -7,6 +7,7 @@ import ( "net/http" "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // DLQHandler exposes Dead Letter Queue statistics. @@ -18,18 +19,30 @@ func NewDLQHandler(stats mq.DeadLetterStats) *DLQHandler { return &DLQHandler{Counts: stats} } -// Stats returns per-table message counts on the dead-letter queue. -// Supports optional ?table= query parameter to filter by table name. +// Stats returns per-table message counts on one tenant's dead-letter queue: +// the tenant ?tenant= names (read strictly, as every ops read does — +// opsTenant), tenant.Default without it. The queue is the MQ's, not the +// settings', so it is looked up there: a tenant whose folder was rejected or +// removed is read like one being served, for as long as its queue is kept, +// and an id with no queue is a 404. Supports optional ?table= query parameter +// to filter by table name. func (h *DLQHandler) Stats(w http.ResponseWriter, r *http.Request) { - counts, err := h.Counts.DeadLetterCounts(r.Context(), r.URL.Query().Get("table")) + id, named, ok := opsTenant(w, r) + if !ok { + return + } + if !named { + id = tenant.Default + } + counts, err := h.Counts.DeadLetterCounts(r.Context(), id, r.URL.Query().Get("table")) if err != nil { - if !errors.Is(err, mq.ErrNoDeadLetterQueue) { - slog.ErrorContext(r.Context(), "dlq stats failed", "error", err) - writeJSONError(w, http.StatusInternalServerError, "stream info failed") + if errors.Is(err, mq.ErrNoDeadLetterQueue) { + writeJSONError(w, http.StatusNotFound, "no dead-letter queue for tenant: "+id.String()) return } - // No dead-letter queue: nothing can have been parked. - counts = mq.DeadLetterCounts{Tables: map[string]uint64{}} + slog.ErrorContext(r.Context(), "dlq stats failed", "tenant", id, "error", err) + writeJSONError(w, http.StatusInternalServerError, "stream info failed") + return } w.Header().Set("Content-Type", "application/json") diff --git a/internal/api/dlq_test.go b/internal/api/dlq_test.go index 4023e354..d814cd8b 100644 --- a/internal/api/dlq_test.go +++ b/internal/api/dlq_test.go @@ -15,56 +15,53 @@ import ( "github.com/stretchr/testify/require" ) -// parkedMsg is a message as the ingest worker would hand it to the DLQ. -func parkedMsg(table string) *mq.Message { +// parkedMsg is a message as the ingest worker would hand it to the DLQ, +// parked under tenant id's table. +func parkedMsg(id tenant.ID, table string) *mq.Message { return (&testutil.MockMessage{ - MsgTopic: mq.Topic{Tenant: tenant.Default, Table: table}, + MsgTopic: mq.Topic{Tenant: id, Table: table}, MsgData: []byte(`{"table_name":"` + table + `"}`), }).Message() } -func TestDLQStats_EmptyWhenNoStream(t *testing.T) { - // The embedded MQ always has a dead-letter queue, so its absence comes - // from a mock. - handler := NewDLQHandler(&testutil.MockDeadLetterStats{Err: mq.ErrNoDeadLetterQueue}) - - req := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/dlq/stats", nil) +// dlqStats serves GET /v1/ops/dlq/stats with query through handler. +func dlqStats(t *testing.T, handler *DLQHandler, query string) *httptest.ResponseRecorder { + t.Helper() + req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/ops/dlq/stats"+query, nil) rec := httptest.NewRecorder() - handler.Stats(rec, req) + return rec +} - assert.Equal(t, http.StatusOK, rec.Code) - - var resp map[string]any - require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) - - tables, ok := resp["tables"].(map[string]any) - require.True(t, ok) - assert.Empty(t, tables) - assert.Equal(t, float64(0), resp["total"]) +// A tenant with no dead-letter queue — never given one on this data +// directory, or an id nobody uses — is a 404 that names it, not an empty +// count that would read as "nothing parked" for a typo. +func TestDLQStats_NoQueueIs404(t *testing.T) { + // The embedded MQ opens a served tenant's queue at boot, so a queue's + // absence comes from a mock. + stats := &testutil.MockDeadLetterStats{Err: mq.ErrNoDeadLetterQueue} + + rec := dlqStats(t, NewDLQHandler(stats), "?tenant=acmee") + assert.Equal(t, http.StatusNotFound, rec.Code) + assert.Contains(t, rec.Body.String(), "no dead-letter queue for tenant: acmee") + testutil.AssertJSONErrorResponse(t, rec) + assert.Equal(t, tenant.ID("acmee"), stats.Tenant) } func TestDLQStats_ReturnsCorrectCounts(t *testing.T) { - dir := t.TempDir() - emb, err := mq.NewEmbedded(dir, 1024*1024) - require.NoError(t, err) - defer func() { _ = emb.Close() }() + emb := testutil.NewEmbeddedMQ(t, 1024*1024) ctx := context.Background() // Park messages on the dead-letter queue. for i := 0; i < 3; i++ { - require.NoError(t, emb.DeadLetter(ctx, parkedMsg("events"))) + require.NoError(t, emb.DeadLetter(ctx, parkedMsg(tenant.Default, "events"))) } for i := 0; i < 2; i++ { - require.NoError(t, emb.DeadLetter(ctx, parkedMsg("users"))) + require.NoError(t, emb.DeadLetter(ctx, parkedMsg(tenant.Default, "users"))) } - handler := NewDLQHandler(emb) - req := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/dlq/stats", nil) - rec := httptest.NewRecorder() - - handler.Stats(rec, req) + rec := dlqStats(t, NewDLQHandler(emb), "") assert.Equal(t, http.StatusOK, rec.Code) @@ -78,21 +75,20 @@ func TestDLQStats_ReturnsCorrectCounts(t *testing.T) { assert.Equal(t, float64(5), resp["total"]) } +func TestDLQStats_EmptyBeforeAnyFailure(t *testing.T) { + rec := dlqStats(t, NewDLQHandler(testutil.NewEmbeddedMQ(t, 1024*1024)), "") + assert.Equal(t, http.StatusOK, rec.Code) + assert.JSONEq(t, `{"tables":{},"total":0}`, rec.Body.String()) +} + func TestDLQStats_SingleTable(t *testing.T) { - dir := t.TempDir() - emb, err := mq.NewEmbedded(dir, 1024*1024) - require.NoError(t, err) - defer func() { _ = emb.Close() }() + emb := testutil.NewEmbeddedMQ(t, 1024*1024) ctx := context.Background() - require.NoError(t, emb.DeadLetter(ctx, parkedMsg("orders"))) + require.NoError(t, emb.DeadLetter(ctx, parkedMsg(tenant.Default, "orders"))) - handler := NewDLQHandler(emb) - req := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/dlq/stats", nil) - rec := httptest.NewRecorder() - - handler.Stats(rec, req) + rec := dlqStats(t, NewDLQHandler(emb), "") assert.Equal(t, http.StatusOK, rec.Code) @@ -105,33 +101,61 @@ func TestDLQStats_SingleTable(t *testing.T) { } func TestDLQStats_BrokerFailureIsAnError(t *testing.T) { - handler := NewDLQHandler(&testutil.MockDeadLetterStats{Err: errors.New("broker unavailable")}) - - req := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/dlq/stats", nil) - rec := httptest.NewRecorder() - - handler.Stats(rec, req) - + rec := dlqStats(t, NewDLQHandler(&testutil.MockDeadLetterStats{Err: errors.New("broker unavailable")}), "") assert.Equal(t, http.StatusInternalServerError, rec.Code, "a failed read is not an empty queue") } func TestDLQStats_PassesTheTableFilter(t *testing.T) { - emb, err := mq.NewEmbedded(t.TempDir(), 1024*1024) - require.NoError(t, err) - defer func() { _ = emb.Close() }() + emb := testutil.NewEmbeddedMQ(t, 1024*1024) ctx := context.Background() - require.NoError(t, emb.DeadLetter(ctx, parkedMsg("default.orders"))) - require.NoError(t, emb.DeadLetter(ctx, parkedMsg("users"))) + require.NoError(t, emb.DeadLetter(ctx, parkedMsg(tenant.Default, "default.orders"))) + require.NoError(t, emb.DeadLetter(ctx, parkedMsg(tenant.Default, "users"))) - handler := NewDLQHandler(emb) - req := httptest.NewRequestWithContext(ctx, http.MethodGet, "/v1/ops/dlq/stats?table=default.orders", nil) - rec := httptest.NewRecorder() - - handler.Stats(rec, req) + rec := dlqStats(t, NewDLQHandler(emb), "?table=default.orders") var resp map[string]any require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &resp)) assert.Equal(t, map[string]any{"default.orders": float64(1)}, resp["tables"]) assert.Equal(t, float64(2), resp["total"]) } + +// ?tenant= reads that tenant's queue alone, and no parameter reads tenant 0's +// — the ops-read convention. The handler asks the MQ, not the settings, so a +// tenant the settings no longer serve (here, none at all) is read by name +// for as long as its queue is kept. +func TestDLQStats_ReadsTheNamedTenantsQueue(t *testing.T) { + emb := testutil.NewEmbeddedMQ(t, 1024*1024, tenant.Default, "acme") + ctx := context.Background() + require.NoError(t, emb.DeadLetter(ctx, parkedMsg(tenant.Default, "events"))) + for range 2 { + require.NoError(t, emb.DeadLetter(ctx, parkedMsg("acme", "events"))) + } + handler := NewDLQHandler(emb) + + rec := dlqStats(t, handler, "?tenant=acme") + assert.Equal(t, http.StatusOK, rec.Code) + assert.JSONEq(t, `{"tables":{"events":2},"total":2}`, rec.Body.String()) + + rec = dlqStats(t, handler, "?tenant=acme&table=users") + assert.Equal(t, http.StatusOK, rec.Code) + assert.JSONEq(t, `{"tables":{},"total":2}`, rec.Body.String()) + + rec = dlqStats(t, handler, "") + assert.Equal(t, http.StatusOK, rec.Code) + assert.JSONEq(t, `{"tables":{"events":1},"total":1}`, rec.Body.String(), "no parameter reads tenant 0") + + rec = dlqStats(t, handler, "?tenant=globex") + assert.Equal(t, http.StatusNotFound, rec.Code, "a tenant with no queue") +} + +// The parameter is read strictly, like every ops read's (opsTenant): a +// query that misparses must not fall back to tenant 0's counts. +func TestDLQStats_RefusesAMalformedTenant(t *testing.T) { + for _, query := range []string{"?tenant=a.b", "?tenant=", "?tenant=a&tenant=b", "?tenant=acme;x=1"} { + stats := &testutil.MockDeadLetterStats{} + rec := dlqStats(t, NewDLQHandler(stats), query) + assert.Equal(t, http.StatusBadRequest, rec.Code, query) + assert.Empty(t, stats.Tenant, "%s: nothing is read", query) + } +} diff --git a/internal/api/ingest.go b/internal/api/ingest.go index c429aaa4..029799b6 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -707,7 +707,7 @@ func (h *IngestHandler) processRecord( slog.DebugContext(ctx, "publishing event to the ingest queue", "table", table, "scope", scope) if err := h.Publisher.Publish(ctx, mq.Topic{Tenant: store.Tenant(), Table: table, Scope: scope}, payload); err != nil { if errors.Is(err, mq.ErrQueueFull) { - slog.WarnContext(ctx, "ingest queue is full", "table", table, "scope", scope) + slog.WarnContext(ctx, "ingest queue is full", "error", err, "table", table, "scope", scope) return false, nil, &requestAbort{Status: http.StatusServiceUnavailable, Message: "service unavailable", RetryAfter: "30"} } slog.ErrorContext(ctx, "failed to publish to the ingest queue", "error", err, "table", table, "scope", scope) diff --git a/internal/api/router_test.go b/internal/api/router_test.go index e0240ba6..03a39c76 100644 --- a/internal/api/router_test.go +++ b/internal/api/router_test.go @@ -15,7 +15,6 @@ import ( "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/discovery" - "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" @@ -332,9 +331,7 @@ func TestNewRouter_RoutesRegistered(t *testing.T) { pub := &testutil.MockPublisher{} hub := stream.NewHub(nil, nil, nil) - emb, err := mq.NewEmbedded(t.TempDir(), 1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 1024*1024) deps := Dependencies{ Tenants: testTenants(), diff --git a/internal/app/app.go b/internal/app/app.go index f853a76b..dc15bf5d 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -18,8 +18,7 @@ // handed whole to each component's wiring function, which derives the // per-call getters the internal packages take: keyed by the request's store // for the handlers, by tenant id for the async paths (perTenant), and fixed -// to the default tenant for the process-wide resources #583 has not yet made -// per tenant (defaultSetting). +// to the default tenant for the ops gate of a flat directory (defaultSetting). package app import ( @@ -90,9 +89,9 @@ type App struct { listener net.Listener // tenants is the registry every tenant-aware path resolves through, and - // the owner of every reload. The one process-wide resource left, the MQ, - // still follows its default tenant, through defaultStore: tenant 0's - // store as of its last adoption (defaultSetting). + // the owner of every reload. defaultStore is tenant 0's store as of its + // last adoption, which the ops gate of a flat directory reads its admin + // role from (defaultSetting). tenants *settings.Registry defaultStore atomic.Pointer[settings.Store] // policies is the default tenant's policy, for the ops gate of a flat diff --git a/internal/app/app_test.go b/internal/app/app_test.go index fc5d2755..2ecd77d9 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -237,7 +237,7 @@ func TestReload_DrivesTheRegisteredHooks(t *testing.T) { a := newApp(t, cfg, Options{}) dedup := a.dedup.For(tenant.Default) require.False(t, dedup.Open()) - require.Equal(t, int64(1<<30), a.mq.MaxBytes()) + require.Equal(t, int64(1<<30), a.mq.MaxBytes(tenant.Default)) rewriteSettings(t, dir, map[string]any{ "dedupe": map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}}, @@ -246,8 +246,8 @@ func TestReload_DrivesTheRegisteredHooks(t *testing.T) { _, adopted := a.tenants.Reload("test") require.True(t, adopted) assert.True(t, dedup.Open(), "dedupe hook opened the store") - // How the budget is split across the MQ's queues is internal/mq's to test. - assert.Equal(t, int64(2<<30), a.mq.MaxBytes(), "mq hook applied the new byte budget") + // How the budget is split across the tenant's queues is internal/mq's to test. + assert.Equal(t, int64(2<<30), a.mq.MaxBytes(tenant.Default), "mq hook applied the new byte budget") rewriteSettings(t, dir, map[string]any{"mq": map[string]any{"max_bytes_gb": 2}}) _, adopted = a.tenants.Reload("test") @@ -397,12 +397,13 @@ func TestNew_NestedWithoutAnOperatorKeyWarnsTheOpsTreeIsClosed(t *testing.T) { }) } -// The process-wide resources follow tenant 0 alone: another tenant's reload -// never moves them, and a rejected 0 folder leaves them as they were rather -// than reconfiguring them from nothing. The dedupe stores are per tenant -// (story 7), so each follows its own folder instead — the contrast the -// same reloads show. -func TestReload_NestedHooksFollowTheDefaultTenant(t *testing.T) { +// A tenant's queue budget and dedupe store follow its own folder alone: +// another tenant's reload moves neither. A rejected or removed folder keeps +// its tenant's queue at the budget it last had — removing never touches +// data — while its dedupe store closes, its seen ids kept. CORS is read per +// request, so a lost 0 folder is felt at once on the routes that read tenant +// 0's list. +func TestReload_NestedHooksFollowEachTenant(t *testing.T) { dedupeOn := map[string]any{"enabled": true, "id_field": "event_id", "require_id": false, "tables": map[string]any{}} grown := map[string]any{"dedupe": dedupeOn, "mq": map[string]any{"max_bytes_gb": 2}} root := writeNestedSettings(t, map[string]map[string]any{ @@ -413,7 +414,8 @@ func TestReload_NestedHooksFollowTheDefaultTenant(t *testing.T) { dedup0, dedupAcme := a.dedup.For(tenant.Default), a.dedup.For("acme") require.False(t, dedup0.Open()) require.False(t, dedupAcme.Open()) - require.Equal(t, int64(1<<30), a.mq.MaxBytes()) + require.Equal(t, int64(1<<30), a.mq.MaxBytes(tenant.Default)) + require.Equal(t, int64(1<<30), a.mq.MaxBytes("acme"), "each served tenant's queue opens at boot at its own budget") // CORS is per tenant, not a hook's: a tenant route reads its own tenant's // list and the exempt routes tenant 0's (the seed's ["*"] in every folder // here), both through the registry, so a lost 0 folder is felt at once. @@ -434,26 +436,27 @@ func TestReload_NestedHooksFollowTheDefaultTenant(t *testing.T) { _, adopted := a.tenants.Reload("test") require.True(t, adopted) assert.True(t, dedupAcme.Open(), "acme's dedupe switch opens acme's own store") + assert.Equal(t, int64(2<<30), a.mq.MaxBytes("acme"), "acme's budget resizes acme's own queue") assert.False(t, dedup0.Open(), "and moves nothing of tenant 0's") - assert.Equal(t, int64(1<<30), a.mq.MaxBytes()) + assert.Equal(t, int64(1<<30), a.mq.MaxBytes(tenant.Default)) rewriteSettings(t, filepath.Join(root, "0"), grown) _, adopted = a.tenants.Reload("test") require.True(t, adopted) assert.True(t, dedup0.Open()) - assert.Equal(t, int64(2<<30), a.mq.MaxBytes()) + assert.Equal(t, int64(2<<30), a.mq.MaxBytes(tenant.Default)) rewriteSettings(t, filepath.Join(root, "0"), invalidQuery) _, adopted = a.tenants.Reload("test") require.False(t, adopted) assert.False(t, dedup0.Open(), "a rejected 0 folder closes tenant 0's own store, which answers no request now") assert.True(t, dedupAcme.Open(), "and costs acme nothing") - assert.Equal(t, int64(2<<30), a.mq.MaxBytes(), "the process-wide budget stays as tenant 0 last adopted it") + assert.Equal(t, int64(2<<30), a.mq.MaxBytes(tenant.Default), "tenant 0's queue is kept at the budget it last had") assert.Empty(t, allowOrigin("/version"), "the exempt routes read tenant 0 through the registry, which is no longer serving it") assert.Equal(t, "*", allowOrigin("/v1/health", "acme"), "acme's own routes keep acme's list") - // A removed 0 folder is the same: the registry forgets the tenant, the - // process keeps the wiring it last adopted. + // A removed 0 folder is the same: the registry forgets the tenant, and + // its queue stays at the budget it last had. require.NoError(t, os.RemoveAll(filepath.Join(root, "0"))) _, adopted = a.tenants.Reload("test") require.True(t, adopted) @@ -461,7 +464,8 @@ func TestReload_NestedHooksFollowTheDefaultTenant(t *testing.T) { require.False(t, known) assert.False(t, dedup0.Open()) assert.True(t, dedupAcme.Open()) - assert.Equal(t, int64(2<<30), a.mq.MaxBytes()) + assert.Equal(t, int64(2<<30), a.mq.MaxBytes(tenant.Default)) + assert.Equal(t, int64(2<<30), a.mq.MaxBytes("acme")) assert.Empty(t, allowOrigin("/version")) assert.Equal(t, "*", allowOrigin("/v1/health", "acme")) } @@ -682,12 +686,11 @@ func gapWindow(minutes int) map[string]any { return map[string]any{"stream": map[string]any{"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": minutes}} } -// One ingest stream holds every tenant's events and the sweeper purges below -// one sequence, so it keeps the longest gap window among the tenants being -// served — every tenant's gap-fill history is inside it (a stream per tenant -// will honor each tenant's own, #583 story 5b). A flat directory's single +// Each tenant being served keeps its own stream.gap_window_minutes, since +// each has a queue of its own; a rejected tenant is not served, so it is not +// named and keeps no history (mq.Purger.PurgeAcked). A flat directory's single // tenant gets exactly its own window. -func TestLongestGapWindow(t *testing.T) { +func TestGapWindows(t *testing.T) { open := func(t *testing.T, dir string) *settings.Registry { t.Helper() guardGlobals(t) @@ -697,22 +700,21 @@ func TestLongestGapWindow(t *testing.T) { } t.Run("flat directory", func(t *testing.T) { - assert.Equal(t, 45*time.Minute, longestGapWindow(open(t, writeSettings(t, gapWindow(45))))) + assert.Equal(t, map[tenant.ID]time.Duration{tenant.Default: 45 * time.Minute}, gapWindows(open(t, writeSettings(t, gapWindow(45))))) }) t.Run("nested directory", func(t *testing.T) { root := writeNestedSettings(t, map[string]map[string]any{"acme": gapWindow(15), "globex": gapWindow(60), "initech": gapWindow(30)}) tenants := open(t, root) - assert.Equal(t, 60*time.Minute, longestGapWindow(tenants)) + assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute, "globex": 60 * time.Minute, "initech": 30 * time.Minute}, gapWindows(tenants)) - // A rejected tenant is not being served, so its window is not weighed. rewriteSettings(t, filepath.Join(root, "globex"), invalidQuery) tenants.Reload("test") - assert.Equal(t, 30*time.Minute, longestGapWindow(tenants)) + assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute, "initech": 30 * time.Minute}, gapWindows(tenants)) }) - t.Run("no tenant served keeps nothing", func(t *testing.T) { - assert.Zero(t, longestGapWindow(open(t, writeNestedSettings(t, map[string]map[string]any{"acme": invalidQuery})))) + t.Run("no tenant served names none", func(t *testing.T) { + assert.Empty(t, gapWindows(open(t, writeNestedSettings(t, map[string]map[string]any{"acme": invalidQuery})))) }) } diff --git a/internal/app/wire.go b/internal/app/wire.go index 009f9920..60cbdcd8 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -65,18 +65,13 @@ func (a *App) wireSettings() error { return fmt.Errorf("settings directory %s invalid, refusing to start — findings above; `wavehouse validate` reproduces them, `wavehouse bootstrap` writes a starter directory", a.cfg.Settings.Dir) } a.tenants = tenants - // Registered first: hooks run in registration order, and every other one - // reads tenant 0 through the store this one tracks. + // Registered first: hooks run in registration order, so every reload + // updates the tracked store before any other hook runs. a.trackDefaultStore() a.onDefaultAdopt(a.trackDefaultStore) a.policies = func() *policy.Policy { return defaultSetting(a, (*settings.Store).Policy) } - switch _, served := tenants.For(tenant.Default); { - case !tenants.Nested(): - if a.policies() == nil { - slog.Warn("no policy adopted — every token-based request is denied until policies.json defines one (fail closed)") - } - case !served: - slog.Warn("nested settings directory with no tenant 0 being served: the MQ byte budget is still configured from tenant 0's config.json, so it runs unconfigured until a 0 folder is adopted") + if !tenants.Nested() && a.policies() == nil { + slog.Warn("no policy adopted — every token-based request is denied until policies.json defines one (fail closed)") } return nil } @@ -84,20 +79,18 @@ func (a *App) wireSettings() error { // trackDefaultStore remembers tenant 0's store as of its last adoption. The // registry stops handing out a rejected tenant's store and forgets a removed // one, but the store keeps its last adopted document either way — and that is -// what the process-wide resources go on following (defaultSetting). +// what defaultSetting goes on reading. func (a *App) trackDefaultStore() { if store, ok := a.tenants.For(tenant.Default); ok { a.defaultStore.Store(store) } } -// defaultSetting reads one setting of the default tenant, which the one -// process-wide resource left (the MQ) follows until #583 gives each tenant -// its own. It reads tenant 0's last adopted document, so a -// 0 folder a reload rejected or removed leaves every reader as it was — -// the MQ's byte budget a hook reconciles and the one read per request (the ops -// gate's admin role) alike. A nested directory that has never served a tenant -// 0 reads T's zero value, which wireSettings warned about at boot. +// defaultSetting reads one setting of the default tenant: the admin role the +// ops gate of a flat directory reads per request. It reads tenant 0's last +// adopted document, so a 0 folder a reload rejected or removed leaves its +// reader as it was. A nested directory that has never served a tenant 0 reads +// T's zero value, and its ops gate reads no policy at all. func defaultSetting[T any](a *App, get func(*settings.Store) T) T { store := a.defaultStore.Load() if store == nil { @@ -108,8 +101,8 @@ func defaultSetting[T any](a *App, get func(*settings.Store) T) T { } // onDefaultAdopt registers fn to run after each reload that adopts the -// default tenant, so a nested directory's other tenants never move the -// process-wide resources, and a rejected 0 folder leaves them as they were. +// default tenant, so a nested directory's other tenants never move what +// follows it, and a rejected 0 folder leaves that as it was. func (a *App) onDefaultAdopt(fn func()) { a.tenants.AfterAdopt(func(adopted []tenant.ID) { if slices.Contains(adopted, tenant.Default) { @@ -137,20 +130,16 @@ func shortestKeepalive(tenants *settings.Registry) (period time.Duration, bucket return period, buckets } -// longestGapWindow is the shape of the one purge bound every tenant's events -// share: the ingest queue is one stream and the sweeper purges below one -// sequence, so the history kept is the longest stream.gap_window_minutes -// among the tenants being served — purging less, never more, so every -// tenant's gap-fill history survives — at the cost of one tenant holding the -// others' history for longer, which a stream per tenant will end (#583 story -// 5b). A flat directory's one tenant gets exactly its own window; -// with no tenant served the zero window purges everything acknowledged. -func longestGapWindow(tenants *settings.Registry) time.Duration { - var window time.Duration - for _, store := range tenants.All() { - window = max(window, store.GapWindow()) +// gapWindows is the history the sweeper keeps for each tenant being served: +// its own stream.gap_window_minutes, since each tenant's events have a queue +// of their own. A tenant it does not name — removed or rejected — keeps no +// history (mq.Purger.PurgeAcked). +func gapWindows(tenants *settings.Registry) map[tenant.ID]time.Duration { + windows := map[tenant.ID]time.Duration{} + for id, store := range tenants.All() { + windows[id] = store.GapWindow() } - return window + return windows } // served reports whether the registry is serving tenant id: what the @@ -185,10 +174,11 @@ func perTenant[T any](tenants *settings.Registry, get func(*settings.Store) T) f // miss reads as DLQ on, not as the zero value perTenant would give: off lets // the worker drop a message it cannot read, and not knowing the tenant is no // reason to destroy its row. Parked, it survives until the tenant resolves. -// So a removed or rejected tenant's queued rows are parked under its own -// subject rather than left unacked for its return: an unacked row holds the -// ack floor, the sweeper stops purging, and the one shared stream fills -// toward mq.max_bytes_gb until every tenant's ingest answers 503. +// So a removed or rejected tenant's queued rows are parked in its own +// dead-letter queue rather than left unacked for its return: unacked, each +// would be redelivered every ack wait for as long as the tenant is away, and +// would hold the tenant's ack floor, so the sweeper could purge none of its +// queue past it. func dlqFor(tenants *settings.Registry) func(tenant.ID, string) bool { return func(id tenant.ID, table string) bool { store, ok := tenants.For(id) @@ -541,15 +531,23 @@ func (a *App) wireDedupe() error { } // wireMQ starts the MQ — the embedded NATS under data_dir/nats, the one -// place the implementation is chosen; everything after it sees mq.Broker. -// mq.max_bytes_gb is hot-reloadable: after each adoption the new budget is -// handed to the MQ, which owns how it is split across its queues and keeps -// them consistent (see mq.Broker.SetMaxBytes). +// place the implementation is chosen; everything after it sees mq.Broker — +// and hands it each served tenant's mq.max_bytes_gb, which opens that +// tenant's queue the first time. The budget is hot-reloadable: after every +// reload the registry applies, each served tenant's is handed over again, +// and the MQ owns how it is split across the tenant's queues and keeps them +// consistent (see mq.Broker.SetMaxBytes). A tenant no longer served keeps +// its queue at the budget it last had. A queue that cannot be opened or +// resized follows the registry's rule for the shape: a flat directory +// refuses boot, like every other store, and on a reload logs it, keeping the +// previous budget; a nested directory logs it at boot too, so it never costs +// the process — the tenant's ingest answers 503 until a reload opens its +// queue. The hook is registered before the boot apply, as the dedupe one is. func (a *App) wireMQ() error { dir := filepath.Join(a.cfg.DataDir, "nats") config.WarnIfFreshDataDir("nats", dir) var broker mq.Broker - broker, err := mq.NewEmbedded(dir, defaultSetting(a, (*settings.Store).MQMaxBytes)) + broker, err := mq.NewEmbedded(dir) if err != nil { config.LogStorageInitError("mq", dir, err) return fmt.Errorf("mq open: %w", err) @@ -570,17 +568,26 @@ func (a *App) wireMQ() error { // Rooted in the App's stop context, so a reload caught mid-hook by // SIGTERM gives up rather than holding the drain past // server.shutdown_timeout. - a.onDefaultAdopt(func() { - mb := defaultSetting(a, (*settings.Store).MQMaxBytes) - if mb == broker.MaxBytes() { - return - } - if err := broker.SetMaxBytes(a.stopCtx, mb); err != nil { - slog.Error("mq stream resize failed; the next reload retries", "error", err) - return + reconcile := func() error { + var errs []error + for id, store := range a.tenants.All() { + mb := store.MQMaxBytes() + if mb == broker.MaxBytes(id) { + continue + } + if err := broker.SetMaxBytes(a.stopCtx, id, mb); err != nil { + slog.Error("mq queue not reconciled with settings; the next reload retries", "tenant", id, "error", err) + errs = append(errs, fmt.Errorf("tenant %s: %w", id, err)) + continue + } + slog.Info("mq queue reconciled with settings", "tenant", id, "max_bytes_gb", mb>>30) } - slog.Info("mq stream limits reconciled with settings", "max_bytes_gb", mb>>30) - }) + return errors.Join(errs...) + } + a.tenants.AfterAdopt(func([]tenant.ID) { _ = reconcile() }) + if err := reconcile(); err != nil && !a.tenants.Nested() { + return fmt.Errorf("mq open: %w", err) + } return nil } @@ -597,11 +604,11 @@ func (a *App) wireCache() error { } // wireSweeper adds the active sweeper — purges messages that are both -// written to ClickHouse and older than the SSE gap window (the longest -// stream.gap_window_minutes among the tenants served, re-read every sweep — -// see longestGapWindow). Runs every minute. +// written to ClickHouse and older than their tenant's SSE gap window (its own +// stream.gap_window_minutes, re-read every sweep — see gapWindows). Runs +// every minute. func (a *App) wireSweeper() { - sweeper := ingest.NewSweeper(a.mq, func() time.Duration { return longestGapWindow(a.tenants) }) + sweeper := ingest.NewSweeper(a.mq, func() map[tenant.ID]time.Duration { return gapWindows(a.tenants) }) a.add(component{name: "sweeper", run: func(ctx context.Context) error { sweeper.Start(ctx) return nil diff --git a/internal/ingest/sweeper.go b/internal/ingest/sweeper.go index 85383ead..367e22af 100644 --- a/internal/ingest/sweeper.go +++ b/internal/ingest/sweeper.go @@ -7,35 +7,34 @@ import ( "time" "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // Sweeper implements the Active Sweeper pattern. It runs every minute and // asks the MQ to purge the ingest events that satisfy BOTH conditions: // - ACKed by the buffer consumer (written to ClickHouse) -// - Older than the gap window (no longer needed for SSE replay) +// - Older than their tenant's gap window (no longer needed for SSE replay) // -// This guarantees: healthy state keeps exactly gap_window of rolling data; -// ClickHouse down freezes purging; a catastrophic outage fills the queue to -// its byte budget and triggers backpressure (mq.ErrQueueFull). How the MQ -// finds the purge point is its own business (see mq.Purger). +// This guarantees: healthy state keeps exactly each tenant's gap_window of +// rolling data; ClickHouse down freezes purging; a catastrophic outage fills +// a tenant's queue to its byte budget and triggers backpressure +// (mq.ErrQueueFull). How the MQ finds the purge point is its own business +// (see mq.Purger). type Sweeper struct { purger mq.Purger - // gapWindow is the history to keep, read on every sweep so a reload of - // stream.gap_window_minutes applies from the next sweep without a - // restart. The ingest queue is one stream for every tenant and a purge - // is one bound over it, so in production this is the longest window - // among the tenants being served (internal/app's longestGapWindow); a - // tenant's own window follows once the streams are per tenant (#583 - // story 5b). - gapWindow func() time.Duration + // gapWindows is the history to keep for each tenant being served, read on + // every sweep so a reload of stream.gap_window_minutes applies from the + // next sweep without a restart. A tenant it does not name — one removed + // or rejected — keeps no history (mq.Purger.PurgeAcked). + gapWindows func() map[tenant.ID]time.Duration } -// NewSweeper creates the Active Sweeper. gapWindow is resolved per sweep. +// NewSweeper creates the Active Sweeper. gapWindows is resolved per sweep. // TODO: (future) need leader election or shared lock to only run one instance of the sweeper in clustered mode -func NewSweeper(purger mq.Purger, gapWindow func() time.Duration) *Sweeper { +func NewSweeper(purger mq.Purger, gapWindows func() map[tenant.ID]time.Duration) *Sweeper { return &Sweeper{ - purger: purger, - gapWindow: gapWindow, + purger: purger, + gapWindows: gapWindows, } } @@ -54,7 +53,13 @@ func (s *Sweeper) Start(ctx context.Context) { } func (s *Sweeper) sweep(ctx context.Context) { - _, err := s.purger.PurgeAcked(ctx, BufferConsumerName, time.Now().Add(-s.gapWindow())) + now := time.Now() + windows := s.gapWindows() + cutoffs := make(map[tenant.ID]time.Time, len(windows)) + for id, window := range windows { + cutoffs[id] = now.Add(-window) + } + _, err := s.purger.PurgeAcked(ctx, BufferConsumerName, cutoffs) if err != nil { if errors.Is(err, mq.ErrConsumerNotFound) { // Consumer may not exist yet if no messages have been ingested. diff --git a/internal/ingest/sweeper_test.go b/internal/ingest/sweeper_test.go index dfd4e02f..9b1eabab 100644 --- a/internal/ingest/sweeper_test.go +++ b/internal/ingest/sweeper_test.go @@ -7,6 +7,7 @@ import ( "time" "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -15,11 +16,11 @@ import ( // The purge-point arithmetic is the MQ's (internal/mq/purge_test.go); the // sweeper owns only when to ask and what window to ask for. -func TestSweep_AsksForTheBufferConsumerAndTheGapWindow(t *testing.T) { +func TestSweep_AsksForTheBufferConsumerAndEachTenantsGapWindow(t *testing.T) { t.Parallel() - gapWindow := 5 * time.Minute + windows := map[tenant.ID]time.Duration{"acme": 5 * time.Minute, "globex": time.Hour} purger := &testutil.MockPurger{Purged: true} - s := NewSweeper(purger, func() time.Duration { return gapWindow }) + s := NewSweeper(purger, func() map[tenant.ID]time.Duration { return windows }) before := time.Now() s.sweep(context.Background()) @@ -28,29 +29,35 @@ func TestSweep_AsksForTheBufferConsumerAndTheGapWindow(t *testing.T) { require.Len(t, purger.Calls, 1) call := purger.Calls[0] assert.Equal(t, BufferConsumerName, call.Consumer) - assert.False(t, call.OlderThan.Before(before.Add(-gapWindow)), "cutoff is now - gap window") - assert.False(t, call.OlderThan.After(after.Add(-gapWindow)), "cutoff is now - gap window") + require.Len(t, call.OlderThan, 2, "one cutoff per tenant served") + for id, window := range windows { + cutoff := call.OlderThan[id] + assert.False(t, cutoff.Before(before.Add(-window)), "%s: cutoff is now - its own gap window", id) + assert.False(t, cutoff.After(after.Add(-window)), "%s: cutoff is now - its own gap window", id) + } } -func TestSweep_RereadsTheGapWindowEverySweep(t *testing.T) { +func TestSweep_RereadsTheGapWindowsEverySweep(t *testing.T) { t.Parallel() - gapWindow := time.Minute + windows := map[tenant.ID]time.Duration{"acme": time.Minute} purger := &testutil.MockPurger{} - s := NewSweeper(purger, func() time.Duration { return gapWindow }) + s := NewSweeper(purger, func() map[tenant.ID]time.Duration { return windows }) s.sweep(context.Background()) - gapWindow = time.Hour // a settings reload + windows = map[tenant.ID]time.Duration{"acme": time.Hour, "globex": time.Minute} // a settings reload s.sweep(context.Background()) require.Len(t, purger.Calls, 2) - assert.Greater(t, purger.Calls[0].OlderThan.Sub(purger.Calls[1].OlderThan), 50*time.Minute) + assert.Greater(t, purger.Calls[0].OlderThan["acme"].Sub(purger.Calls[1].OlderThan["acme"]), 50*time.Minute) + assert.NotContains(t, purger.Calls[0].OlderThan, tenant.ID("globex")) + assert.Contains(t, purger.Calls[1].OlderThan, tenant.ID("globex"), "a tenant adopted since is named from the next sweep") } func TestSweep_ErrorsDoNotPanic(t *testing.T) { t.Parallel() for _, err := range []error{mq.ErrConsumerNotFound, errors.New("broker unavailable")} { purger := &testutil.MockPurger{Err: err} - s := NewSweeper(purger, func() time.Duration { return time.Minute }) + s := NewSweeper(purger, func() map[tenant.ID]time.Duration { return map[tenant.ID]time.Duration{"acme": time.Minute} }) s.sweep(context.Background()) assert.Len(t, purger.Calls, 1) } @@ -62,7 +69,7 @@ func TestSweep_ErrorsDoNotPanic(t *testing.T) { func TestStart_ContextCancellation(t *testing.T) { t.Parallel() - s := NewSweeper(&testutil.MockPurger{}, func() time.Duration { return 5 * time.Minute }) + s := NewSweeper(&testutil.MockPurger{}, func() map[tenant.ID]time.Duration { return nil }) ctx, cancel := context.WithCancel(context.Background()) cancel() // Cancel immediately. diff --git a/internal/ingest/worker.go b/internal/ingest/worker.go index 618b2b35..9af5b1ef 100644 --- a/internal/ingest/worker.go +++ b/internal/ingest/worker.go @@ -116,10 +116,12 @@ const ( // maxAckPending, and ackWait > defaultMaxWait + CH flush (else in-flight // messages are redelivered mid-processing → duplicate inserts). const ( - // Server-side cap on unacked messages; suspends delivery when hit (backpressure). + // Server-side cap on a tenant's unacked messages; suspends that tenant's + // delivery when hit (backpressure), and no other tenant's. maxAckPending = 10_000 // TODO: raise if NATS delivery becomes the bottleneck - // Client prefetch buffer in front of msgChan (was the implicit jetstream default). + // Client prefetch buffer in front of msgChan (was the implicit jetstream + // default), shared by the tenants' queues (mq.Consumer.Consume). pullMaxMessages = 500 // Redelivery timeout. 60s ≈ 5s batch + ~30s HTTP timeout + margin. @@ -219,8 +221,9 @@ func waitOrDeadline(ctx context.Context, wg *sync.WaitGroup) error { } } -// dispatchLoop owns the single JetStream consumer and fans every message out to -// a tableLoop per tenant table (lazily spawned on first sight of one). It does +// dispatchLoop owns the one consumer — held on every tenant's queue — and fans +// every message out to a tableLoop per tenant table (lazily spawned on first +// sight of one). It does // no batching itself — it parses just enough to route — so a low-volume table // can never strand another table's rows behind a shared timer. It is the ONLY // goroutine that watches ctx; tableLoops stop via channel-close, which gives a @@ -230,11 +233,13 @@ func (w *IngestWorker) dispatchLoop(ctx context.Context, cons mq.Consumer) { msgChan := make(chan *mq.Message, w.maxBatch*2) - // Pull consumer with a push-like callback (the client prefetches pullMaxMessages). - // Hand off to msgChan only, so the consume goroutine never blocks on flush work. + // Pull consumer with a push-like callback (the client prefetches pullMaxMessages, + // shared by the tenants' queues). It runs on one delivery goroutine per tenant, + // so the handoff is a channel send, safe from all of them at once. Hand off to + // msgChan only, so a consume goroutine never blocks on flush work. // The handoff also watches ctx: stop (deferred below) does not wait for a // delivery already in the handler, so once this loop has stopped draining - // msgChan a full channel would otherwise pin the client's delivery goroutine + // msgChan a full channel would otherwise pin a delivery goroutine // forever. A message dropped here is unacked and simply redelivered. stop, deliveryEnded, err := cons.Consume(func(msg *mq.Message) { select { diff --git a/internal/ingest/worker_test.go b/internal/ingest/worker_test.go index c3a0988e..7fe9130e 100644 --- a/internal/ingest/worker_test.go +++ b/internal/ingest/worker_test.go @@ -122,9 +122,7 @@ func TestStartIngestWorker_Validation(t *testing.T) { { name: "nil cache", setup: func(t *testing.T) (Queue, cache.Cache) { - emb, err := mq.NewEmbedded(t.TempDir(), 1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 1024*1024) return emb, nil }, wantErrSub: "cache is nil", @@ -154,9 +152,7 @@ func TestStartIngestWorker_EndToEnd(t *testing.T) { t.Parallel() // ── Embedded MQ ── - emb, err := mq.NewEmbedded(t.TempDir(), 4*1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 4*1024*1024) // ── ClickHouse stub: capture each request body, return 200 ── var ( @@ -247,9 +243,7 @@ func TestStartIngestWorker_EndToEnd(t *testing.T) { func TestStartIngestWorker_StopFunc_RespectsShutdownDeadline(t *testing.T) { t.Parallel() - emb, err := mq.NewEmbedded(t.TempDir(), 1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 1024*1024) // ClickHouse stub that blocks until we say go — keeps the worker's // flush goroutine alive past the stop call. @@ -298,9 +292,7 @@ func TestStartIngestWorker_StopFunc_RespectsShutdownDeadline(t *testing.T) { func TestStartIngestWorker_StopFunc_CleanShutdown(t *testing.T) { t.Parallel() - emb, err := mq.NewEmbedded(t.TempDir(), 1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 1024*1024) // chURL is never dialed: with no messages there is no flush, so a dummy // host/port is fine. @@ -1094,9 +1086,7 @@ func TestDispatchLoop_PerTableBatching_NoCrossTableContamination(t *testing.T) { batchB = maxBatch ) - emb, err := mq.NewEmbedded(t.TempDir(), 8*1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 8*1024*1024) // CH stub: count rows (newlines in the JSONCompactEachRow body) per target table. var ( @@ -1187,9 +1177,7 @@ func TestDispatchLoop_PartialBatchWaitsForOwnTrigger(t *testing.T) { total = 4 // 3 → one full batch on the size trigger; 1 leftover ) - emb, err := mq.NewEmbedded(t.TempDir(), 8*1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 8*1024*1024) // CH stub counts rows and sleeps briefly, so the 4th row is reliably buffered // before the first (3-row) flush completes — that's when the old code would @@ -1791,9 +1779,7 @@ func TestDispatchLoop_BatchesPerTenantTable(t *testing.T) { t.Parallel() const maxBatch = 2 - emb, err := mq.NewEmbedded(t.TempDir(), 8*1024*1024) - require.NoError(t, err) - t.Cleanup(func() { _ = emb.Close() }) + emb := testutil.NewEmbeddedMQ(t, 8*1024*1024, "acme", "globex") // CH stub: record each INSERT's body under the database it named. var ( diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index 4219c34d..dbe35fa5 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -5,12 +5,16 @@ import ( "errors" "fmt" "log/slog" + "maps" + "math" + "slices" "strings" "sync" "sync/atomic" "time" "github.com/Wave-RF/WaveHouse/internal/observability" + "github.com/Wave-RF/WaveHouse/internal/tenant" natsserver "github.com/nats-io/nats-server/v2/server" "github.com/nats-io/nats.go" "github.com/nats-io/nats.go/jetstream" @@ -44,45 +48,73 @@ func (slogNATSLogger) Tracef(format string, v ...any) { slog.Debug(fmt.Sprintf(format, v...), "component", "nats") } -// EmbeddedNATS runs an in-process NATS server with JetStream. +// EmbeddedNATS runs an in-process NATS server with JetStream, and gives each +// tenant a queue of its own: an ingest stream and a dead-letter stream +// (subject.go names them), each with its own byte cap, and the durable +// consumers on the ingest one. Nothing outside this package sees that layout. type EmbeddedNATS struct { server *natsserver.Server conn *nats.Conn js jetstream.JetStream - limitMu sync.Mutex - maxBytes int64 // the ingest stream cap both streams were last reconciled to + // mu guards queues and consumers, and serializes opening or resizing a + // tenant's queue with registering a consumer, so a queue opened while a + // consumer registers is never missed by it. It is held across the + // JetStream calls that open or resize a queue. + mu sync.Mutex + queues map[tenant.ID]*tenantQueue + // consumers are the durable consumers held on every tenant's queue, each + // joined to a queue as it opens. + consumers []*fanIn +} + +// tenantQueue is what the broker knows of one tenant's queue. +type tenantQueue struct { + // ingest and dlq report whether each of the tenant's streams exists. + ingest, dlq bool + // maxBytes is the budget last applied in full (MaxBytes); asked is the + // budget last asked for, which a publish or park that finds a stream + // missing opens it at. Both are read back from the ingest stream at boot, + // so a tenant no longer served keeps the budget it last had. + maxBytes, asked int64 } // EmbeddedNATS is the one implementation of every mq interface. var _ Broker = (*EmbeddedNATS)(nil) const ( - // dlqShare is the DLQ stream's slice of the byte budget: a tenth of the - // ingest stream's cap. + // dlqShare is a tenant's dead-letter stream's slice of its byte budget: a + // tenth of the ingest stream's cap. dlqShare = 10 - // resizeTimeout bounds the JetStream calls SetMaxBytes makes to apply a - // new cap — both streams share it. A settings reload holds the store's - // lock while its hooks run, so an in-process JetStream call that never - // returns would otherwise block every later reload. + // resizeTimeout bounds the JetStream calls SetMaxBytes makes to open a + // tenant's queue or apply a new cap to it — both streams share it. A + // settings reload holds the store's lock while its hooks run, so an + // in-process JetStream call that never returns would otherwise block every + // later reload. resizeTimeout = 10 * time.Second - // rollbackTimeout is the undo's own budget when the DLQ resize fails: - // in-process JetStream fails by stalling rather than erroring, so the - // likely cause is that resizeTimeout has just run out, and an undo on + // rollbackTimeout is the undo's own budget when the dead-letter resize + // fails: in-process JetStream fails by stalling rather than erroring, so + // the likely cause is that resizeTimeout has just run out, and an undo on // that context would fail without touching the stream. SetMaxBytes runs // for at most the sum of the two. rollbackTimeout = 5 * time.Second ) -// NewEmbedded starts an embedded NATS server with JetStream enabled and -// both streams in place: the ingest stream capped at maxBytes and the DLQ -// stream at a tenth of it. The DLQ stream is always present — an empty -// limits-policy stream costs nothing, and whether a poison row lands on it is -// the ingest worker's decision at the moment of the failure. -// The server logs through slog's default logger. The stream names are fixed -// (see subject.go) — the embedded server is private to this process, so -// there's no namespacing to do. -func NewEmbedded(storeDir string, maxBytes int64) (*EmbeddedNATS, error) { +// errNoQueue is why a publish or park finds no queue it can open: no budget +// has been asked for the tenant yet (see SetMaxBytes). Publish reports it as +// ErrQueueFull. +var errNoQueue = errors.New("no queue is open for it yet") + +// NewEmbedded starts an embedded NATS server with JetStream over storeDir and +// takes stock of the tenants' queues already there: a consumer created later +// is held on every one of them, those of tenants no longer served included, +// whose queued rows still have to reach the ingest worker. The pair of streams +// an earlier build kept for every tenant together is deleted, since its +// subjects overlap every tenant's; the events it held are not carried over. A +// tenant's queue is opened by SetMaxBytes, the first time its budget is +// applied, or by a publish or park that finds it missing, at the budget last +// asked for it. The server logs through slog's default logger. +func NewEmbedded(storeDir string) (*EmbeddedNATS, error) { opts := &natsserver.Options{ DontListen: true, JetStream: true, @@ -93,6 +125,18 @@ func NewEmbedded(storeDir string, maxBytes int64) (*EmbeddedNATS, error) { // channel" panic) and os.Exit(0)s past its cleanup. WaveHouse owns // the lifecycle; Close() shuts the server down. See #287. NoSigs: true, + // JetStream counts every stream's byte cap as reserved disk and + // refuses a stream once the caps together pass this limit — by + // default 75% of the free disk at boot. A tenant's mq.max_bytes_gb + // caps that tenant's queue and nothing else; what the tenants' caps + // add up to against the disk is #138's to decide, not a limit the + // server enforces on the side, so its own is set out of reach. Half + // the int64 range, not all of it: the server subtracts its count of + // reserved bytes from this limit, and a stream whose store fails to + // open releases a reservation it never made (nats-server 2.14.6), so + // the count can fall below zero — at the top of the range that + // subtraction overflows, and every stream after it is refused. + JetStreamMaxStore: math.MaxInt64 / 2, } ns, err := natsserver.NewServer(opts) @@ -119,114 +163,285 @@ func NewEmbedded(storeDir string, maxBytes int64) (*EmbeddedNATS, error) { return nil, fmt.Errorf("jetstream new: %w", err) } - if _, err := js.CreateOrUpdateStream(context.Background(), ingestStreamConfig(maxBytes)); err != nil { - nc.Close() - ns.Shutdown() - return nil, fmt.Errorf("create stream: %w", err) + e := &EmbeddedNATS{server: ns, conn: nc, js: js, queues: map[tenant.ID]*tenantQueue{}} + if err := e.takeStock(context.Background()); err != nil { + _ = e.Close() + return nil, err } - if _, err := js.CreateOrUpdateStream(context.Background(), dlqStreamConfig(maxBytes/dlqShare)); err != nil { - nc.Close() - ns.Shutdown() - return nil, fmt.Errorf("create dlq stream: %w", err) + return e, nil +} + +// takeStock deletes the pair of streams an earlier build kept for every +// tenant together, then records every tenant stream on disk, with the budget +// its ingest stream last had. +func (e *EmbeddedNATS) takeStock(ctx context.Context) error { + for _, name := range []string{legacyIngestStream, legacyDLQStream} { + if err := e.deleteLegacy(ctx, name); err != nil { + return err + } + } + streams := e.js.ListStreams(ctx) + for info := range streams.Info() { + name := info.Config.Name + if id, ok := streamTenant(ingestStreamPrefix, name); ok { + q := e.queue(id) + q.ingest = true + q.maxBytes, q.asked = info.Config.MaxBytes, info.Config.MaxBytes + } else if id, ok := streamTenant(dlqStreamPrefix, name); ok { + e.queue(id).dlq = true + } } + if err := streams.Err(); err != nil { + return fmt.Errorf("list streams: %w", err) + } + return nil +} - return &EmbeddedNATS{server: ns, conn: nc, js: js, maxBytes: maxBytes}, nil +// deleteLegacy deletes one stream of the pair an earlier build kept for every +// tenant together, logging what it held; one that is not there is nothing to +// do. +func (e *EmbeddedNATS) deleteLegacy(ctx context.Context, name string) error { + s, err := e.js.Stream(ctx, name) + if errors.Is(err, jetstream.ErrStreamNotFound) { + return nil + } + if err != nil { + return fmt.Errorf("look up stream %s: %w", name, err) + } + held := s.CachedInfo().State.Msgs + if err := e.js.DeleteStream(ctx, name); err != nil { + return fmt.Errorf("delete stream %s: %w", name, err) + } + slog.Warn("mq: deleted the stream an earlier build kept for every tenant together; its messages are not carried over", + "component", "nats", "stream", name, "messages", held) + return nil +} + +// queue returns what the broker knows of tenant id's queue, recording the +// tenant first if it knows nothing. Under e.mu (or before e is shared). +func (e *EmbeddedNATS) queue(id tenant.ID) *tenantQueue { + q := e.queues[id] + if q == nil { + q = &tenantQueue{} + e.queues[id] = q + } + return q } -// ingestStreamConfig is the WAVEHOUSE stream. LimitsPolicy: standard +// ingestTenants lists the tenants whose ingest stream exists, in id order. +// Under e.mu. +func (e *EmbeddedNATS) ingestTenants() []tenant.ID { + var ids []tenant.ID + for id, q := range e.queues { + if q.ingest { + ids = append(ids, id) + } + } + slices.Sort(ids) + return ids +} + +// ingestStreamConfig is tenant id's ingest stream. LimitsPolicy: standard // append-only log; the Active Sweeper handles message purging. MaxBytes caps -// disk usage to protect the shared ClickHouse/NATS disk. DiscardNew rejects -// new messages when full, propagating backpressure to the upstream API. -func ingestStreamConfig(maxBytes int64) jetstream.StreamConfig { +// the tenant's share of the disk. DiscardNew rejects new messages when full, +// propagating backpressure to the upstream API — for this tenant alone. +func ingestStreamConfig(id tenant.ID, maxBytes int64) jetstream.StreamConfig { return jetstream.StreamConfig{ - Name: ingestStream, - Subjects: []string{ingestAll}, + Name: ingestStreamName(id), + Subjects: []string{tenantSubjects(ingestPrefix, id)}, Retention: jetstream.LimitsPolicy, MaxBytes: maxBytes, Discard: jetstream.DiscardNew, } } -// dlqStreamConfig is the WAVEHOUSE_DLQ stream. DiscardOld: a full DLQ drops -// its oldest parked rows rather than refusing new ones — backpressure belongs -// to the ingest stream, not the dead-letter one. -func dlqStreamConfig(maxBytes int64) jetstream.StreamConfig { +// dlqStreamConfig is tenant id's dead-letter stream. DiscardOld: a full one +// drops its oldest parked rows rather than refusing new ones — backpressure +// belongs to the ingest stream, not the dead-letter one. +func dlqStreamConfig(id tenant.ID, maxBytes int64) jetstream.StreamConfig { return jetstream.StreamConfig{ - Name: dlqStream, - Subjects: []string{dlqAll}, + Name: dlqStreamName(id), + Subjects: []string{tenantSubjects(dlqPrefix, id)}, Retention: jetstream.LimitsPolicy, MaxBytes: maxBytes, Discard: jetstream.DiscardOld, } } -// MaxBytes reports the ingest stream cap both streams were last reconciled to -// (by NewEmbedded, then by each successful SetMaxBytes). -func (e *EmbeddedNATS) MaxBytes() int64 { - e.limitMu.Lock() - defer e.limitMu.Unlock() - return e.maxBytes +// MaxBytes reports the budget tenant id's queue was last given in full (by +// SetMaxBytes, or read back from disk at boot), 0 when it has none. +func (e *EmbeddedNATS) MaxBytes(id tenant.ID) int64 { + e.mu.Lock() + defer e.mu.Unlock() + if q := e.queues[id]; q != nil { + return q.maxBytes + } + return 0 } -// SetMaxBytes applies a new byte budget to both streams in place (the -// hot-reloadable mq.max_bytes_gb): the ingest stream takes maxBytes and the -// DLQ stream a tenth of it. JetStream applies a limit change to a live stream -// without touching its messages: growing takes effect immediately; shrinking -// the ingest stream below its current size makes DiscardNew refuse new -// publishes until the worker drains it — nothing buffered is dropped. +// SetMaxBytes applies tenant id's byte budget (its hot-reloadable +// mq.max_bytes_gb) to its queue: the ingest stream takes maxBytes and the +// dead-letter stream a tenth of it. A tenant with no queue yet has one opened, +// its dead-letter stream first, so no row is queued that could not be parked, +// and every registered consumer joins it. No other tenant's queue is touched. +// +// JetStream applies a limit change to a live stream without touching its +// messages: growing takes effect immediately; shrinking the ingest stream +// below its current size makes DiscardNew refuse new publishes until the +// worker drains it — nothing buffered is dropped. The dead-letter stream is +// DiscardOld, which would delete its oldest parked rows to fit a smaller cap, +// so it is never capped below the bytes it holds (#532): it keeps what it +// has, and that is logged. // -// The pair moves together where it can. If the DLQ update fails after the -// ingest one succeeded, the ingest resize is undone so the 10:1 pair stays at +// The pair moves together where it can. If the dead-letter update fails after +// the ingest one succeeded, the ingest resize is undone so the pair stays at // the previous budget, and the next call retries both. Safe in that direction // — the ingest stream is DiscardNew, so shrinking it back drops nothing // stored. The undo is best effort: if it fails too, the ingest stream stays at -// the new limit and the DLQ at the previous, and the error says so. On any -// error MaxBytes keeps reporting the previous budget, so a later call with the -// new budget reapplies both. +// the new limit and the dead-letter one at the previous, and the error says +// so. On any error MaxBytes keeps reporting the previous budget, so a later +// call with the new budget reapplies both. // // The JetStream calls are bounded by resizeTimeout, plus rollbackTimeout for // the undo, both rooted in ctx. That is deliberate: ctx is the process's stop // context, so a reload caught mid-hook by a stop gives up — undo included — // rather than holding the drain past server.shutdown_timeout. A cancellation // between the two updates is therefore the one way to leave the pair split, -// and only for the rest of a process that is exiting: the next boot -// reconciles both streams from the adopted settings. -func (e *EmbeddedNATS) SetMaxBytes(ctx context.Context, maxBytes int64) error { - e.limitMu.Lock() - defer e.limitMu.Unlock() - if maxBytes == e.maxBytes { +// and only for the rest of a process that is exiting: the next boot applies +// the adopted settings to it again. +func (e *EmbeddedNATS) SetMaxBytes(ctx context.Context, id tenant.ID, maxBytes int64) error { + if _, err := tenant.Parse(string(id)); err != nil { + return fmt.Errorf("tenant: %w", err) + } + e.mu.Lock() + defer e.mu.Unlock() + q := e.queue(id) + q.asked = maxBytes + if q.ingest && q.dlq && maxBytes == q.maxBytes { return nil } + return e.apply(ctx, id, q, maxBytes) +} +// apply brings tenant id's queue to maxBytes: opening it when its ingest +// stream is missing, resizing it otherwise (see SetMaxBytes). Under e.mu. +func (e *EmbeddedNATS) apply(ctx context.Context, id tenant.ID, q *tenantQueue, maxBytes int64) error { resizeCtx, cancel := context.WithTimeout(ctx, resizeTimeout) defer cancel() - if _, err := e.js.UpdateStream(resizeCtx, ingestStreamConfig(maxBytes)); err != nil { + if !q.ingest { + if err := e.applyDLQ(resizeCtx, id, q, maxBytes); err != nil { + return err + } + if _, err := e.js.CreateOrUpdateStream(resizeCtx, ingestStreamConfig(id, maxBytes)); err != nil { + return fmt.Errorf("open ingest stream: %w", err) + } + q.ingest, q.maxBytes = true, maxBytes + for _, f := range e.consumers { + if err := f.join(resizeCtx, id); err != nil { + f.fail(fmt.Errorf("tenant %s: %w: join its queue: %w", id, ErrDeliveryEnded, err)) + } + } + return nil + } + if _, err := e.js.UpdateStream(resizeCtx, ingestStreamConfig(id, maxBytes)); err != nil { return fmt.Errorf("resize ingest stream: %w", err) } - if _, err := e.js.CreateOrUpdateStream(resizeCtx, dlqStreamConfig(maxBytes/dlqShare)); err != nil { - // The undo runs on its own budget, not the one the DLQ call has - // likely just exhausted. + if err := e.applyDLQ(resizeCtx, id, q, maxBytes); err != nil { + // The undo runs on its own budget, not the one the dead-letter call + // has likely just exhausted. rollbackCtx, cancelRollback := context.WithTimeout(ctx, rollbackTimeout) defer cancelRollback() - if _, rollbackErr := e.js.UpdateStream(rollbackCtx, ingestStreamConfig(e.maxBytes)); rollbackErr != nil { - return fmt.Errorf("resize dlq stream: %w (ingest stream rollback failed, so it stays at the new limit and the dlq at the previous: %w)", err, rollbackErr) + if _, rollbackErr := e.js.UpdateStream(rollbackCtx, ingestStreamConfig(id, q.maxBytes)); rollbackErr != nil { + return fmt.Errorf("%w (ingest stream rollback failed, so it stays at the new limit and the dlq at the previous: %w)", err, rollbackErr) } - return fmt.Errorf("resize dlq stream: %w (ingest stream restored to the previous limit)", err) + return fmt.Errorf("%w (ingest stream restored to the previous limit)", err) } - e.maxBytes = maxBytes + q.maxBytes = maxBytes return nil } -// Publish stores data on topic's ingest subject. A topic without a valid -// tenant is refused before anything is sent (see subject). A stream at its -// byte budget (DiscardNew) refuses the publish; that is reported as -// ErrQueueFull. +// applyDLQ gives tenant id's dead-letter stream a tenth of maxBytes, creating +// it when it is missing, but never caps it below the bytes it holds: those +// stay, the cap is what they take, and the stream then drops its oldest row +// to make room for each new one, as any full dead-letter stream does. Under +// e.mu. +func (e *EmbeddedNATS) applyDLQ(ctx context.Context, id tenant.ID, q *tenantQueue, maxBytes int64) error { + limit := maxBytes / dlqShare + verb := "resize" + s, err := e.js.Stream(ctx, dlqStreamName(id)) + switch { + case errors.Is(err, jetstream.ErrStreamNotFound): + verb = "open" + case err != nil: + return fmt.Errorf("dlq stream info: %w", err) + default: + // A stream's size fits an int64 as its cap does; the bound is + // checked rather than assumed. + if held := s.CachedInfo().State.Bytes; held <= math.MaxInt64 && int64(held) > limit { + slog.Warn("mq: dead-letter queue kept at what it holds rather than shrunk to its budget, so no parked row is deleted", + "component", "nats", "tenant", id, "held_bytes", held, "budget_bytes", limit) + limit = int64(held) + } + } + if _, err := e.js.CreateOrUpdateStream(ctx, dlqStreamConfig(id, limit)); err != nil { + return fmt.Errorf("%s dlq stream: %w", verb, err) + } + q.dlq = true + return nil +} + +// reopen opens tenant id's queue at the budget last asked for it, for a +// publish or park that found one of its streams missing. errNoQueue when no +// budget has been asked for the tenant yet: a reload can make a tenant +// resolvable an instant before its budget arrives. +func (e *EmbeddedNATS) reopen(ctx context.Context, id tenant.ID) error { + e.mu.Lock() + defer e.mu.Unlock() + q := e.queues[id] + if q == nil || q.asked == 0 { + return fmt.Errorf("tenant %s: %w", id, errNoQueue) + } + // What is missing is asked of JetStream rather than read off the flags, + // which may still say the stream the publish just missed exists — or it + // may be back already, opened by a caller that held mu first. + for _, name := range []string{ingestStreamName(id), dlqStreamName(id)} { + _, err := e.js.Stream(ctx, name) + switch { + case errors.Is(err, jetstream.ErrStreamNotFound): + if name == ingestStreamName(id) { + q.ingest = false + } else { + q.dlq = false + } + case err != nil: + return fmt.Errorf("stream info: %w", err) + } + } + if q.ingest && q.dlq { + return nil + } + return e.apply(ctx, id, q, q.asked) +} + +// Publish stores data on topic's ingest subject, in its tenant's queue. A +// topic without a valid tenant is refused before anything is sent (see +// subject). A tenant with no queue has one opened at the budget last asked +// for it (see SetMaxBytes). A queue that cannot be opened — none asked for +// yet, or JetStream refused it — and a queue at its byte budget (DiscardNew) +// are reported as ErrQueueFull: either way the tenant's queue takes nothing +// now, and a retry is the caller's answer. func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, opts ...PublishOpt) error { subj, err := subject(ingestPrefix, topic) if err != nil { return err } err = e.publish(ctx, subj, data, opts) + if errors.Is(err, jetstream.ErrNoStreamResponse) { + if openErr := e.reopen(ctx, topic.Tenant); openErr != nil { + return fmt.Errorf("%w: %w", ErrQueueFull, openErr) + } + err = e.publish(ctx, subj, data, opts) + } if err != nil && strings.Contains(err.Error(), "maximum bytes exceeded") { // The server reports a full store as a generic store failure whose // text is the only thing that names the cause. @@ -235,12 +450,23 @@ func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, op return err } -// DeadLetter stores msg's data on its topic's DLQ subject — the subject it -// arrived on with the ingest prefix swapped for the DLQ one, nothing decoded -// or re-encoded. The DLQ stream is DiscardOld, so a full DLQ drops its oldest -// parked rows rather than refusing. +// DeadLetter stores msg's data on its topic's dead-letter subject, in its +// tenant's queue — the subject it arrived on with the ingest prefix swapped +// for the dead-letter one, nothing decoded or re-encoded. The dead-letter +// stream is DiscardOld, so a full one drops its oldest parked rows rather than +// refusing. A dead-letter stream found missing is opened again with its +// tenant's queue, as Publish does. func (e *EmbeddedNATS) DeadLetter(ctx context.Context, msg *Message, opts ...PublishOpt) error { - return e.publish(ctx, dlqPrefix+msg.topicKey, msg.Data, opts) + subj := dlqPrefix + msg.topicKey + err := e.publish(ctx, subj, msg.Data, opts) + if errors.Is(err, jetstream.ErrNoStreamResponse) { + if id, ok := keyTenant(msg.topicKey); ok { + if err = e.reopen(ctx, id); err == nil { + err = e.publish(ctx, subj, msg.Data, opts) + } + } + } + return err } func (e *EmbeddedNATS) publish(ctx context.Context, subj string, data []byte, opts []PublishOpt) error { @@ -255,7 +481,10 @@ func (e *EmbeddedNATS) publish(ctx context.Context, subj string, data []byte, op observability.InjectHeaders(ctx, headers) msg.Header = nats.Header(headers) - _, err := e.js.PublishMsg(ctx, msg) + // No retry on "no responders": in-process, that only ever means no + // stream holds the subject — a tenant with no queue, which the callers + // open rather than wait out. + _, err := e.js.PublishMsg(ctx, msg, jetstream.WithRetryAttempts(0)) return err } @@ -275,58 +504,170 @@ func wrapMsg(ctx context.Context, m jetstream.Msg) *Message { ) } +// Subscribe holds a durable explicit-ack consumer named consumerName on every +// tenant's queue, those opened later included, and delivers each message to +// handler with the trace context its headers carry, until ctx is done. A +// tenant's queue that cannot be joined when it opens is logged: its events +// reach handler from the next boot. func (e *EmbeddedNATS) Subscribe(ctx context.Context, consumerName string, handler func(msg *Message) error) error { - cons, err := e.js.CreateOrUpdateConsumer(ctx, ingestStream, jetstream.ConsumerConfig{ - Durable: consumerName, - FilterSubject: ingestAll, - AckPolicy: jetstream.AckExplicitPolicy, - }) - if err != nil { + f := e.newFanIn(ctx, jetstream.ConsumerConfig{Durable: consumerName, AckPolicy: jetstream.AckExplicitPolicy}) + f.fail = func(err error) { + slog.Error("mq: a tenant's events do not reach this consumer until the next boot", "component", "nats", "consumer", consumerName, "error", err) + } + if err := e.register(ctx, f); err != nil { return fmt.Errorf("create consumer: %w", err) } - - cctx, err := cons.Consume(func(m jetstream.Msg) { + stop, err := f.start(func(m jetstream.Msg) { msg := wrapMsg(observability.ExtractHeaders(ctx, m.Headers()), m) if err := handler(msg); err != nil { _ = msg.Nak() } - }) + }, 0, false) if err != nil { return fmt.Errorf("consume: %w", err) } go func() { <-ctx.Done() - cctx.Stop() + stop() }() return nil } // CreateConsumer creates or updates a durable explicit-ack pull consumer on -// the ingest stream. ctx becomes every delivered Message.Ctx (see -// ConsumerManager); it does not stop delivery — Consumer.Consume's stop does. +// every tenant's queue, and joins each queue opened later. ctx becomes every +// delivered Message.Ctx (see ConsumerManager); it does not stop delivery — +// Consumer.Consume's stop does. func (e *EmbeddedNATS) CreateConsumer(ctx context.Context, cfg ConsumerConfig) (Consumer, error) { - cons, err := e.js.CreateOrUpdateConsumer(ctx, ingestStream, jetstream.ConsumerConfig{ - Durable: cfg.Durable, - FilterSubject: ingestAll, - AckPolicy: jetstream.AckExplicitPolicy, - AckWait: cfg.AckWait, - MaxAckPending: cfg.MaxAckPending, - }) - if err != nil { + c := &workerConsumer{ + fanIn: e.newFanIn(ctx, jetstream.ConsumerConfig{ + Durable: cfg.Durable, + AckPolicy: jetstream.AckExplicitPolicy, + AckWait: cfg.AckWait, + MaxAckPending: cfg.MaxAckPending, + }), + failed: make(chan error, 1), + } + c.fail = func(err error) { + // Exactly one error, and nothing once stop has been called. + if c.stopped.Load() { + return + } + select { + case c.failed <- err: + default: + } + } + if err := e.register(ctx, c.fanIn); err != nil { return nil, fmt.Errorf("create consumer: %w", err) } - return &jsConsumer{cons: cons, ctx: ctx}, nil + return c, nil +} + +// newFanIn is a fanIn over cfg, not yet holding any durable; the caller sets +// its fail and registers it. +func (e *EmbeddedNATS) newFanIn(ctx context.Context, cfg jetstream.ConsumerConfig) *fanIn { + return &fanIn{ + e: e, + ctx: ctx, + cfg: cfg, + handles: map[tenant.ID]jetstream.Consumer{}, + running: map[tenant.ID]jetstream.ConsumeContext{}, + } } -// jsConsumer is the Consumer over a JetStream pull consumer. -type jsConsumer struct { - cons jetstream.Consumer - ctx context.Context // each delivered Message.Ctx (see ConsumerManager) +// register holds f's durable on every tenant's queue there is and registers +// f, so every queue opened from here on is joined too. +func (e *EmbeddedNATS) register(ctx context.Context, f *fanIn) error { + e.mu.Lock() + defer e.mu.Unlock() + for _, id := range e.ingestTenants() { + if err := f.join(ctx, id); err != nil { + return fmt.Errorf("tenant %s: %w", id, err) + } + } + e.consumers = append(e.consumers, f) + return nil +} + +// unregister stops joining f to the queues that open from here on. Under +// e.mu. +func (e *EmbeddedNATS) unregister(f *fanIn) { + e.consumers = slices.DeleteFunc(e.consumers, func(c *fanIn) bool { return c == f }) } -func (c *jsConsumer) Consume(handler func(msg *Message), prefetch int) (func(), <-chan error, error) { +// fanIn is one durable consumer held on every tenant's ingest stream — the +// ingest worker's (CreateConsumer) or the hub bridge's (Subscribe) — +// delivering them all into one handler: each tenant's messages on a +// goroutine of their own, so a tenant's arrive in order and different +// tenants' concurrently, and a handler blocked on one tenant holds back that +// tenant alone. Its fields are guarded by e.mu, bar stopped. +type fanIn struct { + e *EmbeddedNATS + ctx context.Context // each delivered Message.Ctx (CreateConsumer), or where Subscribe extracts trace context into + cfg jetstream.ConsumerConfig + + // fail reports a tenant's delivery that ended on its own, or a queue that + // could not be joined when it opened. + fail func(error) + + // handles is the durable on each tenant's ingest stream; running, the + // delivery started on each once deliver is set. + handles map[tenant.ID]jetstream.Consumer + running map[tenant.ID]jetstream.ConsumeContext + deliver func(jetstream.Msg) + // prefetch is the fetch-ahead asked for across the tenants together; 0 + // leaves each tenant the client default. + prefetch int + // watch reports a delivery that ends on its own through fail. + watch bool + stopped atomic.Bool +} + +// join holds f's durable on tenant id's ingest stream — looked up first, and +// created or updated only when missing or configured otherwise, so a boot +// over thousands of queues writes nothing it need not — and starts delivery +// on it when f is delivering. Under e.mu. +func (f *fanIn) join(ctx context.Context, id tenant.ID) error { + stream := ingestStreamName(id) + c, err := f.e.js.Consumer(ctx, stream, f.cfg.Durable) + if err != nil || !sameConsumer(c.CachedInfo().Config, f.cfg) { + if c, err = f.e.js.CreateOrUpdateConsumer(ctx, stream, f.cfg); err != nil { + return err + } + } + f.handles[id] = c + if f.deliver == nil || f.stopped.Load() { + return nil + } + return f.run(id) +} + +// sameConsumer reports whether a durable holds the fields this package sets; +// a zero field in want is the server's default, whatever that resolved to. +func sameConsumer(have, want jetstream.ConsumerConfig) bool { + return have.AckPolicy == want.AckPolicy && + have.FilterSubject == want.FilterSubject && + (want.AckWait == 0 || have.AckWait == want.AckWait) && + (want.MaxAckPending == 0 || have.MaxAckPending == want.MaxAckPending) +} + +// share is one tenant's part of the fetch-ahead: the total spread over the +// tenants' queues, at least one each. 0 leaves the client default. Under +// e.mu. +func (f *fanIn) share() int { + if f.prefetch <= 0 { + return 0 + } + return max(1, f.prefetch/max(1, len(f.handles))) +} + +// run starts delivery from tenant id's durable, once. Under e.mu. +func (f *fanIn) run(id tenant.ID) error { + if _, ok := f.running[id]; ok { + return nil + } // The client reports what goes wrong after Consume returns only through // this handler, never through Consume's own error. It calls it for // passing conditions too (a missed heartbeat, a leadership change) and @@ -339,37 +680,80 @@ func (c *jsConsumer) Consume(handler func(msg *Message), prefetch int) (func(), opts := []jetstream.PullConsumeOpt{ jetstream.ConsumeErrHandler(func(_ jetstream.ConsumeContext, err error) { lastErr.Store(&err) - slog.Warn("mq: consumer reported an error", "component", "nats", "error", err) + slog.Warn("mq: consumer reported an error", "component", "nats", "tenant", id, "error", err) }), } - if prefetch > 0 { - opts = append(opts, jetstream.PullMaxMessages(prefetch)) + if n := f.share(); n > 0 { + opts = append(opts, jetstream.PullMaxMessages(n)) } - cctx, err := c.cons.Consume(func(m jetstream.Msg) { - handler(wrapMsg(c.ctx, m)) - }, opts...) + cctx, err := f.handles[id].Consume(f.deliver, opts...) if err != nil { - return nil, nil, fmt.Errorf("consume: %w", err) + return err + } + f.running[id] = cctx + if !f.watch { + return nil } - - var stopped atomic.Bool - failed := make(chan error, 1) go func() { <-cctx.Closed() - if stopped.Load() { + if f.stopped.Load() { return } - if reason := lastErr.Load(); reason != nil { - failed <- fmt.Errorf("%w: %w", ErrDeliveryEnded, *reason) - return + reason := ErrDeliveryEnded + if r := lastErr.Load(); r != nil { + reason = fmt.Errorf("%w: %w", ErrDeliveryEnded, *r) } - failed <- ErrDeliveryEnded + f.fail(fmt.Errorf("tenant %s: %w", id, reason)) }() - stop := func() { - stopped.Store(true) - cctx.Stop() + return nil +} + +// start begins delivery to deliver from every tenant's durable, and from +// each queue joined later, fetching about prefetch messages ahead across the +// tenants together (see share); watch reports a delivery that ends on its own +// through fail. The returned stop ends every delivery and stops joining new +// queues, without waiting. +func (f *fanIn) start(deliver func(jetstream.Msg), prefetch int, watch bool) (stop func(), err error) { + f.e.mu.Lock() + defer f.e.mu.Unlock() + f.deliver, f.prefetch, f.watch = deliver, prefetch, watch + stop = func() { + f.e.mu.Lock() + defer f.e.mu.Unlock() + f.stopped.Store(true) + for _, cctx := range f.running { + cctx.Stop() + } + f.e.unregister(f) + } + for _, id := range slices.Sorted(maps.Keys(f.handles)) { + if err := f.run(id); err != nil { + f.stopped.Store(true) + for _, cctx := range f.running { + cctx.Stop() + } + f.e.unregister(f) + return nil, fmt.Errorf("tenant %s: %w", id, err) + } + } + return stop, nil +} + +// workerConsumer is the Consumer CreateConsumer returns: a fanIn with the +// failed channel its contract promises. +type workerConsumer struct { + *fanIn + failed chan error +} + +func (c *workerConsumer) Consume(handler func(msg *Message), prefetch int) (func(), <-chan error, error) { + stop, err := c.start(func(m jetstream.Msg) { + handler(wrapMsg(c.ctx, m)) + }, prefetch, true) + if err != nil { + return nil, nil, fmt.Errorf("consume: %w", err) } - return stop, failed, nil + return stop, c.failed, nil } // stream resolves a stream handle by name. @@ -433,40 +817,71 @@ func (s *jsStream) consumerAckFloor(ctx context.Context, consumer string) (uint6 return info.AckFloor.Stream, nil } -// PurgeAcked purges the ingest stream below MIN(consumer's ack floor + 1, -// first sequence stored at or after olderThan) — see purgeAcked. -func (e *EmbeddedNATS) PurgeAcked(ctx context.Context, consumer string, olderThan time.Time) (bool, error) { - s, err := e.stream(ctx, ingestStream) - if err != nil { - return false, fmt.Errorf("get stream: %w", err) +// PurgeAcked purges each tenant's ingest stream below MIN(consumer's ack +// floor + 1, first sequence stored at or after the tenant's cutoff) — see +// purgeAcked. A tenant olderThan does not name is purged up to its ack floor. +// A failure on one tenant's stream is joined into the error and the sweep +// goes on to the next; a done ctx ends it. +func (e *EmbeddedNATS) PurgeAcked(ctx context.Context, consumer string, olderThan map[tenant.ID]time.Time) (bool, error) { + e.mu.Lock() + ids := e.ingestTenants() + e.mu.Unlock() + + now := time.Now() + var ( + errs []error + tenants int + ) + for _, id := range ids { + if err := ctx.Err(); err != nil { + errs = append(errs, err) + break + } + cutoff, ok := olderThan[id] + if !ok { + cutoff = now + } + s, err := e.stream(ctx, ingestStreamName(id)) + if err != nil { + errs = append(errs, fmt.Errorf("tenant %s: get stream: %w", id, err)) + continue + } + report, err := purgeAcked(ctx, s, consumer, cutoff) + if err != nil { + errs = append(errs, fmt.Errorf("tenant %s: %w", id, err)) + continue + } + // The sweep's own log lines: their detail is in sequences, which only + // this package speaks. Per tenant at Debug, since a sweep reaches + // every tenant each minute; the summary below is the Info line. + switch { + case report.purged: + tenants++ + slog.DebugContext(ctx, "sweeper: purged", + "tenant", id, + "purged_below_seq", report.target, + "ack_floor", report.ackFloor, + "gap_seq", report.gapSeq, + ) + case report.gapSeq == 0: + slog.DebugContext(ctx, "sweeper: all messages within gap window, skipping purge", "tenant", id) + } } - report, err := purgeAcked(ctx, s, consumer, olderThan) - if err != nil { - return false, err + if tenants > 0 { + slog.InfoContext(ctx, "sweeper: purged", "tenants", tenants) } - // The sweep's own log lines: their detail is in sequences, which only - // this package speaks. - switch { - case report.purged: - slog.InfoContext(ctx, "sweeper: purged", - "purged_below_seq", report.target, - "ack_floor", report.ackFloor, - "gap_seq", report.gapSeq, - ) - case report.gapSeq == 0: - slog.DebugContext(ctx, "sweeper: all messages within gap window, skipping purge") - } - return report.purged, nil -} - -// DeadLetterCounts reads the DLQ stream's per-subject counts and keys them by -// table across every tenant (see DeadLetterCounts.Tables). The table filter -// matches that table's unscoped subject under any tenant, so it is applied -// to the parsed topic rather than as a subject filter; a scoped topic counts -// under "table.scope". A subject written before the tenant led it counts -// under its table like any other (parseTopicKey). -func (e *EmbeddedNATS) DeadLetterCounts(ctx context.Context, table string) (DeadLetterCounts, error) { - s, err := e.stream(ctx, dlqStream) + return tenants > 0, errors.Join(errs...) +} + +// DeadLetterCounts reads tenant id's dead-letter stream's per-subject counts +// and keys them by table. The table filter matches that table's unscoped +// subject, so it is applied to the parsed topic rather than as a subject +// filter; a scoped topic counts under "table.scope". +func (e *EmbeddedNATS) DeadLetterCounts(ctx context.Context, id tenant.ID, table string) (DeadLetterCounts, error) { + if _, err := tenant.Parse(string(id)); err != nil { + return DeadLetterCounts{}, fmt.Errorf("tenant: %w", err) + } + s, err := e.stream(ctx, dlqStreamName(id)) if err != nil { if errors.Is(err, jetstream.ErrStreamNotFound) { return DeadLetterCounts{}, fmt.Errorf("%w: %w", ErrNoDeadLetterQueue, err) @@ -474,7 +889,7 @@ func (e *EmbeddedNATS) DeadLetterCounts(ctx context.Context, table string) (Dead return DeadLetterCounts{}, fmt.Errorf("get dlq stream: %w", err) } - state, err := s.state(ctx, dlqAll) + state, err := s.state(ctx, tenantSubjects(dlqPrefix, id)) if err != nil { return DeadLetterCounts{}, fmt.Errorf("dlq stream info: %w", err) } @@ -495,21 +910,20 @@ func (e *EmbeddedNATS) DeadLetterCounts(ctx context.Context, table string) (Dead return counts, nil } -// ReplaySince creates an ephemeral consumer on topic's ingest subject starting at -// since (DeliverByStartTime) and drains it to send until caught up. The -// consumer is ack-less and expires on its own once idle. Caught up is the -// client's no-messages or request-timeout answer to a pull; any other pull -// failure (a closed connection, a deleted consumer) is returned so the caller -// knows the replay ended short rather than empty. A done ctx ends the drain -// between pulls and returns ctx's error. A topic without a valid tenant is -// refused like a publish (see subject): the subject it names is exact, so -// events published before the tenant led the subject are not replayed. +// ReplaySince creates an ephemeral consumer on topic's ingest subject, in its +// tenant's queue, starting at since (DeliverByStartTime) and drains it to send +// until caught up. The consumer is ack-less and expires on its own once idle. +// Caught up is the client's no-messages or request-timeout answer to a pull; +// any other pull failure (a closed connection, a deleted consumer) is returned +// so the caller knows the replay ended short rather than empty. A done ctx +// ends the drain between pulls and returns ctx's error. A topic without a +// valid tenant is refused like a publish (see subject). func (e *EmbeddedNATS) ReplaySince(ctx context.Context, topic Topic, since time.Time, send func(data []byte) bool) error { subj, err := subject(ingestPrefix, topic) if err != nil { return err } - cons, err := e.js.CreateOrUpdateConsumer(ctx, ingestStream, jetstream.ConsumerConfig{ + cons, err := e.js.CreateOrUpdateConsumer(ctx, ingestStreamName(topic.Tenant), jetstream.ConsumerConfig{ FilterSubject: subj, DeliverPolicy: jetstream.DeliverByStartTimePolicy, OptStartTime: &since, diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index ab355df7..fe4fd45e 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -2,6 +2,9 @@ package mq import ( "context" + "fmt" + "os" + "path/filepath" "sync" "testing" "time" @@ -13,16 +16,62 @@ import ( "github.com/stretchr/testify/require" ) -// newTestEmbedded spins up an EmbeddedNATS with a temporary store directory -// that is cleaned up by the test framework. -func newTestEmbedded(t *testing.T) *EmbeddedNATS { +// testBudget is the byte budget newTestEmbedded opens each queue at. +const testBudget = 64 << 20 + +// openEmbedded starts an EmbeddedNATS over dir, closed by the test framework. +func openEmbedded(t *testing.T, dir string) *EmbeddedNATS { t.Helper() - e, err := NewEmbedded(t.TempDir(), 64<<20) + e, err := NewEmbedded(dir) require.NoError(t, err) t.Cleanup(func() { _ = e.Close() }) return e } +// newTestEmbedded spins up an EmbeddedNATS over a temporary store directory +// with a queue open for each of tenants — tenant.Default when none is named — +// at testBudget. +func newTestEmbedded(t *testing.T, tenants ...tenant.ID) *EmbeddedNATS { + t.Helper() + e := openEmbedded(t, t.TempDir()) + if len(tenants) == 0 { + tenants = []tenant.ID{tenant.Default} + } + for _, id := range tenants { + require.NoError(t, e.SetMaxBytes(t.Context(), id, testBudget)) + } + return e +} + +// streamConfig is the stored config of the named stream. +func streamConfig(t *testing.T, e *EmbeddedNATS, name string) jetstream.StreamConfig { + t.Helper() + s, err := e.js.Stream(t.Context(), name) + require.NoError(t, err) + return s.CachedInfo().Config +} + +// ackAll consumes every message delivered to consumer on the ingest queue, +// acknowledging each, until n have been acked. +func ackAll(t *testing.T, e *EmbeddedNATS, consumer string, n int) { + t.Helper() + ctx := t.Context() + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: consumer, MaxAckPending: 100}) + require.NoError(t, err) + acked := make(chan error, n) + stop, _, err := cons.Consume(func(msg *Message) { acked <- msg.DoubleAck(ctx) }, 10) + require.NoError(t, err) + t.Cleanup(stop) + for range n { + select { + case err := <-acked: + require.NoError(t, err) + case <-time.After(5 * time.Second): + t.Fatal("timed out waiting for acks") + } + } +} + func TestEmbeddedNATS_PublishSubscribe(t *testing.T) { // No t.Parallel(): each embedded server uses DontListen+InProcessServer, // but starting several in parallel still slows tests unnecessarily. @@ -83,7 +132,7 @@ func TestEmbeddedNATS_PublishHeaders(t *testing.T) { // Read the stored message back raw: the option headers are on the wire // exactly as set, exact-key, with Add appending rather than replacing. - s, err := e.js.Stream(ctx, ingestStream) + s, err := e.js.Stream(ctx, "INGEST_0") require.NoError(t, err) raw, err := s.GetLastMsgForSubject(ctx, "ingest.0.hdr") require.NoError(t, err) @@ -92,24 +141,30 @@ func TestEmbeddedNATS_PublishHeaders(t *testing.T) { assert.Equal(t, []byte("x"), raw.Data) } -func TestNewEmbedded_CreatesBothStreams(t *testing.T) { - e := newTestEmbedded(t) - ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) - defer cancel() - - assert.Equal(t, int64(64<<20), e.MaxBytes()) - - ingest, err := e.js.Stream(ctx, ingestStream) - require.NoError(t, err) - assert.Equal(t, int64(64<<20), ingest.CachedInfo().Config.MaxBytes) - - // The DLQ stream is always present, at a tenth of the budget. - dlq, err := e.js.Stream(ctx, dlqStream) - require.NoError(t, err) - cfg := dlq.CachedInfo().Config - assert.Equal(t, []string{"dlq.>"}, cfg.Subjects) - assert.Equal(t, int64(64<<20)/10, cfg.MaxBytes) - assert.Equal(t, jetstream.DiscardOld, cfg.Discard) +// A tenant's first budget opens its queue: an ingest stream holding its +// subjects alone at the budget, refusing when full, and a dead-letter stream +// at a tenth of it, dropping its oldest when full. No other tenant gets one. +func TestEmbeddedNATS_SetMaxBytes_OpensTheTenantsQueue(t *testing.T) { + e := openEmbedded(t, t.TempDir()) + assert.Zero(t, e.MaxBytes("acme"), "no budget applied yet") + + require.NoError(t, e.SetMaxBytes(t.Context(), "acme", testBudget)) + assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) + + ingest := streamConfig(t, e, "INGEST_acme") + assert.Equal(t, []string{"ingest.acme.>"}, ingest.Subjects) + assert.Equal(t, int64(testBudget), ingest.MaxBytes) + assert.Equal(t, jetstream.DiscardNew, ingest.Discard) + dlq := streamConfig(t, e, "DLQ_acme") + assert.Equal(t, []string{"dlq.acme.>"}, dlq.Subjects) + assert.Equal(t, int64(testBudget)/10, dlq.MaxBytes) + assert.Equal(t, jetstream.DiscardOld, dlq.Discard) + + assert.Zero(t, e.MaxBytes("globex")) + _, err := e.js.Stream(t.Context(), "INGEST_globex") + require.ErrorIs(t, err, jetstream.ErrStreamNotFound, "another tenant's queue opens with its own budget") + + require.Error(t, e.SetMaxBytes(t.Context(), "a.b", testBudget), "a tenant outside the grammar has no queue") } func TestEmbeddedNATS_StreamHandle(t *testing.T) { @@ -120,7 +175,7 @@ func TestEmbeddedNATS_StreamHandle(t *testing.T) { _, err := e.stream(ctx, "NO_SUCH_STREAM") require.Error(t, err, "an unknown stream is an error, not a nil handle") - s, err := e.stream(ctx, ingestStream) + s, err := e.stream(ctx, "INGEST_0") require.NoError(t, err) empty, err := s.state(ctx, "") @@ -196,11 +251,11 @@ func TestEmbeddedNATS_StreamHandle(t *testing.T) { } // TestEmbeddedNATS_CreateConsumer_Config pins the ConsumerConfig → broker -// mapping: AckWait (redelivery timing) and MaxAckPending (ingest backpressure) -// are checkable nowhere else, and a dropped field would compile and pass -// every delivery test. +// mapping on every tenant's queue: AckWait (redelivery timing) and +// MaxAckPending (ingest backpressure, per tenant) are checkable nowhere else, +// and a dropped field would compile and pass every delivery test. func TestEmbeddedNATS_CreateConsumer_Config(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, "acme", "globex") ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() @@ -211,17 +266,16 @@ func TestEmbeddedNATS_CreateConsumer_Config(t *testing.T) { }) require.NoError(t, err) - s, err := e.js.Stream(ctx, ingestStream) - require.NoError(t, err) - cons, err := s.Consumer(ctx, "cfg") - require.NoError(t, err) - info, err := cons.Info(ctx) - require.NoError(t, err) - assert.Equal(t, "cfg", info.Config.Durable) - assert.Equal(t, "ingest.>", info.Config.FilterSubject, "the consumer sees every topic") - assert.Equal(t, jetstream.AckExplicitPolicy, info.Config.AckPolicy) - assert.Equal(t, 42*time.Second, info.Config.AckWait) - assert.Equal(t, 123, info.Config.MaxAckPending) + for _, stream := range []string{"INGEST_acme", "INGEST_globex"} { + cons, err := e.js.Consumer(ctx, stream, "cfg") + require.NoError(t, err, stream) + cfg := cons.CachedInfo().Config + assert.Equal(t, "cfg", cfg.Durable) + assert.Empty(t, cfg.FilterSubject, "%s: the durable sees the whole of its tenant's stream", stream) + assert.Equal(t, jetstream.AckExplicitPolicy, cfg.AckPolicy) + assert.Equal(t, 42*time.Second, cfg.AckWait) + assert.Equal(t, 123, cfg.MaxAckPending) + } } func TestEmbeddedNATS_ReplaySince(t *testing.T) { @@ -262,7 +316,7 @@ func TestEmbeddedNATS_ReplaySince(t *testing.T) { func TestEmbeddedNATS_DefaultLogger(t *testing.T) { // NewEmbedded without a logger should not panic — it falls back to the // default slog logger. - e, err := NewEmbedded(t.TempDir(), 64<<20) + e, err := NewEmbedded(t.TempDir()) require.NoError(t, err) t.Cleanup(func() { _ = e.Close() }) } @@ -301,29 +355,32 @@ func TestSlogNATSLogger_Levels(t *testing.T) { } func TestEmbeddedNATS_SetMaxBytes(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, "acme", "globex") ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - require.NoError(t, e.SetMaxBytes(ctx, 128<<20)) - assert.Equal(t, int64(128<<20), e.MaxBytes()) + require.NoError(t, e.SetMaxBytes(ctx, "acme", 128<<20)) + assert.Equal(t, int64(128<<20), e.MaxBytes("acme")) - ingest, err := e.js.Stream(ctx, ingestStream) - require.NoError(t, err) - assert.Equal(t, int64(128<<20), ingest.CachedInfo().Config.MaxBytes) + ingest := streamConfig(t, e, "INGEST_acme") + assert.Equal(t, int64(128<<20), ingest.MaxBytes) // Everything but the limit is preserved. - assert.Equal(t, []string{"ingest.>"}, ingest.CachedInfo().Config.Subjects) - assert.Equal(t, jetstream.DiscardNew, ingest.CachedInfo().Config.Discard) + assert.Equal(t, []string{"ingest.acme.>"}, ingest.Subjects) + assert.Equal(t, jetstream.DiscardNew, ingest.Discard) - // The DLQ stream follows at a tenth of the budget. - dlq, err := e.js.Stream(ctx, dlqStream) - require.NoError(t, err) - assert.Equal(t, int64(128<<20)/10, dlq.CachedInfo().Config.MaxBytes) - assert.Equal(t, jetstream.DiscardOld, dlq.CachedInfo().Config.Discard) + // The dead-letter stream follows at a tenth of the budget. + dlq := streamConfig(t, e, "DLQ_acme") + assert.Equal(t, int64(128<<20)/10, dlq.MaxBytes) + assert.Equal(t, jetstream.DiscardOld, dlq.Discard) + + // No other tenant's queue moves. + assert.Equal(t, int64(testBudget), e.MaxBytes("globex")) + assert.Equal(t, int64(testBudget), streamConfig(t, e, "INGEST_globex").MaxBytes) + assert.Equal(t, int64(testBudget)/10, streamConfig(t, e, "DLQ_globex").MaxBytes) // The budget already in effect is a no-op, not an error. - require.NoError(t, e.SetMaxBytes(ctx, 128<<20)) - assert.Equal(t, int64(128<<20), e.MaxBytes()) + require.NoError(t, e.SetMaxBytes(ctx, "acme", 128<<20)) + assert.Equal(t, int64(128<<20), e.MaxBytes("acme")) } func TestEmbeddedNATS_SetMaxBytes_DLQFailureRollsBackIngest(t *testing.T) { @@ -331,27 +388,23 @@ func TestEmbeddedNATS_SetMaxBytes_DLQFailureRollsBackIngest(t *testing.T) { ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - // Put the DLQ stream where the update can't follow: JetStream refuses to - // change a live stream's retention policy, so recreating it as a work - // queue makes the DLQ resize fail after the ingest resize has already - // succeeded. - require.NoError(t, e.js.DeleteStream(ctx, dlqStream)) + // Put the dead-letter stream where the update can't follow: JetStream + // refuses to change a live stream's retention policy, so recreating it as + // a work queue makes the dead-letter resize fail after the ingest resize + // has already succeeded. + require.NoError(t, e.js.DeleteStream(ctx, "DLQ_0")) _, err := e.js.CreateStream(ctx, jetstream.StreamConfig{ - Name: dlqStream, Subjects: []string{dlqAll}, Retention: jetstream.WorkQueuePolicy, MaxBytes: (64 << 20) / 10, + Name: "DLQ_0", Subjects: []string{"dlq.0.>"}, Retention: jetstream.WorkQueuePolicy, MaxBytes: testBudget / 10, }) require.NoError(t, err) - err = e.SetMaxBytes(ctx, 128<<20) + err = e.SetMaxBytes(ctx, tenant.Default, 128<<20) require.Error(t, err) assert.Contains(t, err.Error(), "ingest stream restored to the previous limit") - assert.Equal(t, int64(64<<20), e.MaxBytes(), "the budget in effect is unchanged, so the next call retries both") + assert.Equal(t, int64(testBudget), e.MaxBytes(tenant.Default), "the budget in effect is unchanged, so the next call retries both") - ingest, err := e.js.Stream(ctx, ingestStream) - require.NoError(t, err) - assert.Equal(t, int64(64<<20), ingest.CachedInfo().Config.MaxBytes, "the ingest resize is undone so the pair stays at the previous limit") - dlq, err := e.js.Stream(ctx, dlqStream) - require.NoError(t, err) - assert.Equal(t, int64(64<<20)/10, dlq.CachedInfo().Config.MaxBytes) + assert.Equal(t, int64(testBudget), streamConfig(t, e, "INGEST_0").MaxBytes, "the ingest resize is undone so the pair stays at the previous limit") + assert.Equal(t, int64(testBudget)/10, streamConfig(t, e, "DLQ_0").MaxBytes) } func TestEmbeddedNATS_SetMaxBytes_IngestFailureChangesNothing(t *testing.T) { @@ -359,13 +412,86 @@ func TestEmbeddedNATS_SetMaxBytes_IngestFailureChangesNothing(t *testing.T) { ctx, cancel := context.WithCancel(t.Context()) cancel() // a stop caught mid-reload: the first JetStream call gives up - err := e.SetMaxBytes(ctx, 128<<20) + err := e.SetMaxBytes(ctx, tenant.Default, 128<<20) require.ErrorIs(t, err, context.Canceled) - assert.Equal(t, int64(64<<20), e.MaxBytes()) + assert.Equal(t, int64(testBudget), e.MaxBytes(tenant.Default)) - dlq, err := e.js.Stream(t.Context(), dlqStream) - require.NoError(t, err) - assert.Equal(t, int64(64<<20)/10, dlq.CachedInfo().Config.MaxBytes, "the dlq is not touched when the ingest resize fails") + assert.Equal(t, int64(testBudget)/10, streamConfig(t, e, "DLQ_0").MaxBytes, "the dead-letter stream is not touched when the ingest resize fails") +} + +// A tenant whose queue JetStream will not open — here, a file where its +// dead-letter stream's store would go — is refused on its own: SetMaxBytes +// errors and applies no budget, and a publish is refused as a full queue, +// while every other tenant's queue opens after it (which a store limit at +// the very top of the int64 range would refuse: see NewEmbedded). Once the +// cause is gone, a publish opens the queue at the budget last asked for it. +func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { + dir := t.TempDir() + // The dead-letter stream is the first of the pair to open. A failed open + // removes what was in the way, so the obstacle is put back before each + // attempt meant to fail. + block := filepath.Join(dir, "jetstream", "$G", "streams", dlqStreamName("acme")) + obstruct := func() { + t.Helper() + require.NoError(t, os.MkdirAll(filepath.Dir(block), 0o750)) + require.NoError(t, os.WriteFile(block, nil, 0o600)) + } + obstruct() + e := openEmbedded(t, dir) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + require.Error(t, e.SetMaxBytes(ctx, "acme", testBudget)) + assert.Zero(t, e.MaxBytes("acme"), "no budget applied") + + require.NoError(t, e.SetMaxBytes(ctx, "globex", testBudget), "one tenant's failed open costs the next nothing") + require.NoError(t, e.Publish(ctx, Topic{Tenant: "globex", Table: "t"}, []byte("x"))) + + obstruct() + err := e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x")) + require.ErrorIs(t, err, ErrQueueFull, "the tenant's queue takes nothing; a retry is the answer") + + if err := os.Remove(block); err != nil { + require.ErrorIs(t, err, os.ErrNotExist) + } + require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"))) + assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) +} + +// A budget that shrinks a tenant's dead-letter stream below what it holds +// would have DiscardOld delete the oldest parked rows to fit (#532), so the +// stream keeps what it holds, capped at that, and every row survives. +func TestEmbeddedNATS_SetMaxBytes_NeverShrinksTheDeadLetterQueueBelowWhatItHolds(t *testing.T) { + e := openEmbedded(t, t.TempDir()) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + require.NoError(t, e.SetMaxBytes(ctx, "acme", 10<<20)) + + payload := make([]byte, 1<<10) + for range 200 { + msg := NewMessage(ctx, Topic{Tenant: "acme", Table: "t"}, payload, time.Now(), nil, nil, nil) + require.NoError(t, e.DeadLetter(ctx, msg)) + } + dlqState := func() jetstream.StreamState { + t.Helper() + s, err := e.js.Stream(ctx, "DLQ_acme") + require.NoError(t, err) + return s.CachedInfo().State + } + held := dlqState().Bytes + require.Greater(t, held, uint64(100<<10), "the rows take more than a tenth of the budget below") + + // Shrunk to a 1 MB budget: a tenth of it is less than the stream holds. + require.NoError(t, e.SetMaxBytes(ctx, "acme", 1<<20)) + assert.Equal(t, int64(1<<20), e.MaxBytes("acme"), "the budget applies") + assert.Equal(t, int64(1<<20), streamConfig(t, e, "INGEST_acme").MaxBytes) + assert.Equal(t, held, uint64(streamConfig(t, e, "DLQ_acme").MaxBytes), "capped at what it holds, not at a tenth") //nolint:gosec // G115: a stream cap is never negative + assert.Equal(t, uint64(200), dlqState().Msgs, "no parked row is deleted") + + // A budget whose tenth covers what it holds applies as usual. + require.NoError(t, e.SetMaxBytes(ctx, "acme", 4<<20)) + assert.Equal(t, int64(4<<20)/10, streamConfig(t, e, "DLQ_acme").MaxBytes) + assert.Equal(t, uint64(200), dlqState().Msgs) } func TestEmbeddedNATS_ReplaySince_PullFailureIsAnError(t *testing.T) { @@ -410,11 +536,11 @@ func TestEmbeddedNATS_ReplaySince_StopsWhenContextIsDone(t *testing.T) { } func TestEmbeddedNATS_DeadLetter(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, tenant.Default, "acme") ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - empty, err := e.DeadLetterCounts(ctx, "") + empty, err := e.DeadLetterCounts(ctx, tenant.Default, "") require.NoError(t, err) assert.Equal(t, DeadLetterCounts{Tables: map[string]uint64{}}, empty) @@ -426,52 +552,58 @@ func TestEmbeddedNATS_DeadLetter(t *testing.T) { park(Topic{Tenant: tenant.Default, Table: "default.orders"}, "o1") park(Topic{Tenant: tenant.Default, Table: "default.orders"}, "o2") park(Topic{Tenant: tenant.Default, Table: "users"}, "u1") - // Another tenant's table of the same name counts with it: one queue, one - // count, until the queue is per tenant. So does a subject parked before - // the tenant led it — the queue is never drained, so those stay. + // Another tenant's table of the same name is its own queue and its own + // count. park(Topic{Tenant: "acme", Table: "users"}, "acme-u1") - _, err = e.js.Publish(ctx, "dlq.users", []byte("pre-tenant")) - require.NoError(t, err) - // Parked under the same topic on the DLQ stream, headers intact, and - // nothing lands on the ingest stream. - dlq, err := e.js.Stream(ctx, dlqStream) + // Parked under the same topic on the tenant's dead-letter stream, headers + // intact, and nothing lands on the ingest stream. + dlq, err := e.js.Stream(ctx, "DLQ_0") require.NoError(t, err) raw, err := dlq.GetLastMsgForSubject(ctx, "dlq.0.default%2Eorders") require.NoError(t, err) assert.Equal(t, []byte("o2"), raw.Data) assert.Equal(t, "boom", raw.Header.Get("X-DLQ-Error")) - ingest, err := e.stream(ctx, ingestStream) + ingest, err := e.stream(ctx, "INGEST_0") require.NoError(t, err) st, err := ingest.state(ctx, "") require.NoError(t, err) assert.Zero(t, st.Msgs) - all, err := e.DeadLetterCounts(ctx, "") + all, err := e.DeadLetterCounts(ctx, tenant.Default, "") require.NoError(t, err) - assert.Equal(t, DeadLetterCounts{Tables: map[string]uint64{"default.orders": 2, "users": 3}, Total: 5}, all, "table names come back decoded, summed across tenants") + assert.Equal(t, DeadLetterCounts{Tables: map[string]uint64{"default.orders": 2, "users": 1}, Total: 3}, all, "table names come back decoded, the tenant's own alone") - one, err := e.DeadLetterCounts(ctx, "default.orders") + one, err := e.DeadLetterCounts(ctx, tenant.Default, "default.orders") require.NoError(t, err) - assert.Equal(t, DeadLetterCounts{Tables: map[string]uint64{"default.orders": 2}, Total: 5}, one, "Total is every parked message, filter or not") + assert.Equal(t, DeadLetterCounts{Tables: map[string]uint64{"default.orders": 2}, Total: 3}, one, "Total is every parked message of the tenant, filter or not") - users, err := e.DeadLetterCounts(ctx, "users") + acme, err := e.DeadLetterCounts(ctx, "acme", "") require.NoError(t, err) - assert.Equal(t, map[string]uint64{"users": 3}, users.Tables, "the filter is by table under any tenant, the pre-tenant subject included") + assert.Equal(t, DeadLetterCounts{Tables: map[string]uint64{"users": 1}, Total: 1}, acme) - none, err := e.DeadLetterCounts(ctx, "never_failed") + none, err := e.DeadLetterCounts(ctx, tenant.Default, "never_failed") require.NoError(t, err) assert.Empty(t, none.Tables) } +// A tenant with no queue — one never given a budget on this data directory — +// has nothing parked, which is not the same as a failed read. func TestEmbeddedNATS_DeadLetterCounts_NoQueue(t *testing.T) { e := newTestEmbedded(t) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - require.NoError(t, e.js.DeleteStream(ctx, dlqStream)) - _, err := e.DeadLetterCounts(ctx, "") + _, err := e.DeadLetterCounts(ctx, "globex", "") + require.ErrorIs(t, err, ErrNoDeadLetterQueue) + + require.NoError(t, e.js.DeleteStream(ctx, "DLQ_0")) + _, err = e.DeadLetterCounts(ctx, tenant.Default, "") require.ErrorIs(t, err, ErrNoDeadLetterQueue) + + _, err = e.DeadLetterCounts(ctx, "a.b", "") + require.Error(t, err, "an id outside the grammar names no stream") + assert.NotErrorIs(t, err, ErrNoDeadLetterQueue) } func TestEmbeddedNATS_DeadLetterCounts_BrokerFailureIsNotAnEmptyQueue(t *testing.T) { @@ -482,19 +614,19 @@ func TestEmbeddedNATS_DeadLetterCounts_BrokerFailureIsNotAnEmptyQueue(t *testing // A lookup that fails for any reason other than "no such stream" must not // read as an empty queue. e.conn.Close() - _, err := e.DeadLetterCounts(ctx, "") + _, err := e.DeadLetterCounts(ctx, tenant.Default, "") require.Error(t, err) assert.NotErrorIs(t, err, ErrNoDeadLetterQueue) } func TestEmbeddedNATS_DeadLetter_IsAPrefixSwap(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, "a") ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() // A subject this package would never write (four tokens) still parks - // under the very same tail: nothing on the dead-letter path decodes or - // re-encodes it. + // under the very same tail, in the queue of the tenant its first token + // names: nothing on the dead-letter path decodes or re-encodes it. _, err := e.js.Publish(ctx, "ingest.a.b.c.d", []byte("foreign")) require.NoError(t, err) @@ -513,29 +645,72 @@ func TestEmbeddedNATS_DeadLetter_IsAPrefixSwap(t *testing.T) { assert.Equal(t, Topic{Table: "a.b.c.d"}, msg.Topic(), "a foreign tail is the table of no tenant") require.NoError(t, e.DeadLetter(ctx, msg)) - dlq, err := e.js.Stream(ctx, dlqStream) + dlq, err := e.js.Stream(ctx, "DLQ_a") require.NoError(t, err) raw, err := dlq.GetLastMsgForSubject(ctx, "dlq.a.b.c.d") require.NoError(t, err) assert.Equal(t, []byte("foreign"), raw.Data) } -func TestEmbeddedNATS_Publish_QueueFull(t *testing.T) { - e, err := NewEmbedded(t.TempDir(), 4<<10) +// A dead-letter stream that has gone missing is opened again with its +// tenant's queue, at a tenth of the budget last asked for it, rather than +// leaving the row to be redelivered. +func TestEmbeddedNATS_DeadLetter_ReopensAMissingQueue(t *testing.T) { + e := newTestEmbedded(t, "acme") + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + require.NoError(t, e.js.DeleteStream(ctx, "DLQ_acme")) + require.NoError(t, e.DeadLetter(ctx, NewMessage(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"), time.Now(), nil, nil, nil))) + assert.Equal(t, int64(testBudget)/10, streamConfig(t, e, "DLQ_acme").MaxBytes) + counts, err := e.DeadLetterCounts(ctx, "acme", "") require.NoError(t, err) - t.Cleanup(func() { _ = e.Close() }) + assert.Equal(t, uint64(1), counts.Total) +} + +func TestEmbeddedNATS_Publish_QueueFull(t *testing.T) { + e := openEmbedded(t, t.TempDir()) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() + require.NoError(t, e.SetMaxBytes(ctx, "acme", 4<<10)) + require.NoError(t, e.SetMaxBytes(ctx, "globex", 4<<10)) // DiscardNew refuses the publish that would pass the byte budget; that is // the backpressure signal, named so callers need not read broker errors. payload := make([]byte, 1<<10) + var err error for range 8 { - if err = e.Publish(ctx, Topic{Tenant: tenant.Default, Table: "full"}, payload); err != nil { + if err = e.Publish(ctx, Topic{Tenant: "acme", Table: "full"}, payload); err != nil { break } } require.ErrorIs(t, err, ErrQueueFull) + + // Only the tenant at its budget is refused: the next one has a budget of + // its own. + require.NoError(t, e.Publish(ctx, Topic{Tenant: "globex", Table: "full"}, payload)) +} + +// A tenant's queue opens at the budget last asked for it when a publish finds +// it missing, and a tenant never given a budget has no queue to publish to: +// that is refused as a full queue, and nothing is opened for it. +func TestEmbeddedNATS_Publish_OpensTheQueueAtTheLastBudget(t *testing.T) { + e := newTestEmbedded(t, "acme") + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + for _, name := range []string{"INGEST_acme", "DLQ_acme"} { + require.NoError(t, e.js.DeleteStream(ctx, name)) + } + require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"))) + assert.Equal(t, int64(testBudget), streamConfig(t, e, "INGEST_acme").MaxBytes) + assert.Equal(t, int64(testBudget)/10, streamConfig(t, e, "DLQ_acme").MaxBytes) + + err := e.Publish(ctx, Topic{Tenant: "globex", Table: "t"}, []byte("x")) + require.ErrorIs(t, err, ErrQueueFull) + assert.Contains(t, err.Error(), "globex") + _, err = e.js.Stream(ctx, "INGEST_globex") + require.ErrorIs(t, err, jetstream.ErrStreamNotFound) } func TestEmbeddedNATS_PurgeAcked(t *testing.T) { @@ -544,7 +719,7 @@ func TestEmbeddedNATS_PurgeAcked(t *testing.T) { defer cancel() // No consumer yet: the sentinel the sweeper keys its "not yet" warning on. - _, err := e.PurgeAcked(ctx, "buffer", time.Now()) + _, err := e.PurgeAcked(ctx, "buffer", map[tenant.ID]time.Time{tenant.Default: time.Now()}) require.ErrorIs(t, err, ErrConsumerNotFound) for i := range 4 { @@ -570,7 +745,7 @@ func TestEmbeddedNATS_PurgeAcked(t *testing.T) { t.Fatal("timed out waiting for acks") } } - s, err := e.stream(ctx, ingestStream) + s, err := e.stream(ctx, "INGEST_0") require.NoError(t, err) require.Eventually(t, func() bool { floor, err := s.consumerAckFloor(ctx, "buffer") @@ -578,12 +753,12 @@ func TestEmbeddedNATS_PurgeAcked(t *testing.T) { }, 5*time.Second, 20*time.Millisecond) // Everything is acked-or-not but nothing is old enough: keep it all. - purged, err := e.PurgeAcked(ctx, "buffer", time.Now().Add(-time.Hour)) + purged, err := e.PurgeAcked(ctx, "buffer", map[tenant.ID]time.Time{tenant.Default: time.Now().Add(-time.Hour)}) require.NoError(t, err) assert.False(t, purged) // Everything is old enough: only the acked two go. - purged, err = e.PurgeAcked(ctx, "buffer", time.Now().Add(time.Hour)) + purged, err = e.PurgeAcked(ctx, "buffer", map[tenant.ID]time.Time{tenant.Default: time.Now().Add(time.Hour)}) require.NoError(t, err) assert.True(t, purged) st, err := s.state(ctx, "") @@ -592,8 +767,147 @@ func TestEmbeddedNATS_PurgeAcked(t *testing.T) { assert.Equal(t, uint64(2), st.Msgs) } +// Each tenant's queue is purged at its own cutoff and below its own ack +// floor: a tenant keeping an hour of history keeps it while the next one's +// goes, and a tenant the cutoffs do not name — one no longer served — keeps +// no history at all. +func TestEmbeddedNATS_PurgeAcked_EachTenantAtItsOwnCutoff(t *testing.T) { + e := newTestEmbedded(t, "acme", "globex", "initech") + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + for _, id := range []tenant.ID{"acme", "globex", "initech"} { + for i := range 2 { + require.NoError(t, e.Publish(ctx, Topic{Tenant: id, Table: "p"}, []byte{byte(i)})) + } + } + ackAll(t, e, "buffer", 6) + for _, id := range []tenant.ID{"acme", "globex", "initech"} { + s, err := e.stream(ctx, ingestStreamName(id)) + require.NoError(t, err) + require.Eventually(t, func() bool { + floor, err := s.consumerAckFloor(ctx, "buffer") + return err == nil && floor == 2 + }, 5*time.Second, 20*time.Millisecond, id) + } + + purged, err := e.PurgeAcked(ctx, "buffer", map[tenant.ID]time.Time{ + "acme": time.Now().Add(-time.Hour), // an hour of history: all of it inside the window + "globex": time.Now().Add(time.Hour), // everything older than the cutoff + }) + require.NoError(t, err) + assert.True(t, purged) + msgs := func(id tenant.ID) uint64 { + s, err := e.stream(ctx, ingestStreamName(id)) + require.NoError(t, err) + st, err := s.state(ctx, "") + require.NoError(t, err) + return st.Msgs + } + assert.Equal(t, uint64(2), msgs("acme"), "kept for its own window") + assert.Zero(t, msgs("globex"), "past its own window") + assert.Zero(t, msgs("initech"), "a tenant the cutoffs do not name keeps nothing it has acknowledged") +} + +// The isolation per-tenant queues buy: a tenant at MaxAckPending, or one +// whose handler is stuck, holds back its own delivery and no other tenant's — +// each tenant's messages arrive on a delivery of their own, in order. +func TestEmbeddedNATS_Consume_OneTenantsBacklogDoesNotHoldAnother(t *testing.T) { + e := newTestEmbedded(t, "acme", "globex", "initech") + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer", MaxAckPending: 2}) + require.NoError(t, err) + release := make(chan struct{}) + var mu sync.Mutex + delivered := map[tenant.ID][]byte{} + stop, _, err := cons.Consume(func(msg *Message) { + id := msg.Topic().Tenant + mu.Lock() + delivered[id] = append(delivered[id], msg.Data[0]) + mu.Unlock() + if id == "initech" { + <-release // never returns until the test ends + } + if id == "globex" { + _ = msg.Ack() // acme never acks: its delivery stops at MaxAckPending + } + }, 12) + require.NoError(t, err) + t.Cleanup(func() { + close(release) + stop() + }) + + for i := range 5 { + for _, id := range []tenant.ID{"acme", "globex", "initech"} { + require.NoError(t, e.Publish(ctx, Topic{Tenant: id, Table: "t"}, []byte{byte(i)})) + } + } + counts := func() (acme, globex, initech int) { + mu.Lock() + defer mu.Unlock() + return len(delivered["acme"]), len(delivered["globex"]), len(delivered["initech"]) + } + require.Eventually(t, func() bool { + acme, globex, initech := counts() + return acme == 2 && globex == 5 && initech == 1 + }, 5*time.Second, 20*time.Millisecond, "globex is delivered in full while acme waits on its acks and initech on its handler") + time.Sleep(200 * time.Millisecond) + acme, globex, initech := counts() + assert.Equal(t, 2, acme, "no more than MaxAckPending unacked, for acme alone") + assert.Equal(t, 5, globex) + assert.Equal(t, 1, initech, "a stuck handler holds back its own tenant alone") + mu.Lock() + defer mu.Unlock() + assert.Equal(t, []byte{0, 1, 2, 3, 4}, delivered["globex"], "in the order published") +} + +// A tenant's queue opened after the consumer started is joined to it: both +// consumer paths deliver its events as they do the queues that were there +// first, whether those were opened in this process or found on disk. +func TestEmbeddedNATS_ConsumersJoinQueuesOpenedLater(t *testing.T) { + e := newTestEmbedded(t, "acme") + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + worker := make(chan Topic, 4) + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer", MaxAckPending: 10}) + require.NoError(t, err) + stop, _, err := cons.Consume(func(msg *Message) { + _ = msg.Ack() + worker <- msg.Topic() + }, 4) + require.NoError(t, err) + t.Cleanup(stop) + hub := make(chan Topic, 4) + require.NoError(t, e.Subscribe(ctx, "hub-bridge", func(msg *Message) error { + _ = msg.Ack() + hub <- msg.Topic() + return nil + })) + + require.NoError(t, e.SetMaxBytes(ctx, "globex", testBudget)) + for _, id := range []tenant.ID{"acme", "globex"} { + require.NoError(t, e.Publish(ctx, Topic{Tenant: id, Table: "t"}, []byte("x"))) + } + for name, got := range map[string]chan Topic{"worker": worker, "hub": hub} { + var topics []Topic + for range 2 { + select { + case topic := <-got: + topics = append(topics, topic) + case <-time.After(5 * time.Second): + t.Fatalf("%s: timed out; delivered %v", name, topics) + } + } + assert.ElementsMatch(t, []Topic{{Tenant: "acme", Table: "t"}, {Tenant: "globex", Table: "t"}}, topics, name) + } +} + func TestEmbeddedNATS_Consume_ReportsDeliveryEndingOnItsOwn(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, "acme", "globex") ctx, cancel := context.WithTimeout(t.Context(), 30*time.Second) defer cancel() @@ -609,22 +923,23 @@ func TestEmbeddedNATS_Consume_ReportsDeliveryEndingOnItsOwn(t *testing.T) { case <-time.After(200 * time.Millisecond): } - // Deleting the durable underneath a running Consume is terminal: the - // client stops the subscription on its own, and no message will ever say - // so. It must reach the caller. - require.NoError(t, e.js.DeleteConsumer(ctx, ingestStream, "doomed")) + // Deleting one tenant's durable underneath a running Consume is terminal + // for that tenant: the client stops the subscription on its own, and no + // message will ever say so. It must reach the caller. + require.NoError(t, e.js.DeleteConsumer(ctx, "INGEST_globex", "doomed")) select { case err := <-failed: require.ErrorIs(t, err, ErrDeliveryEnded) require.ErrorIs(t, err, jetstream.ErrConsumerDeleted, "the broker's reason is kept") + assert.Contains(t, err.Error(), "globex", "the tenant is named") case <-ctx.Done(): t.Fatal("delivery ended underneath the consumer and nothing was reported") } } func TestEmbeddedNATS_Consume_StopIsNotAFailure(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, "acme", "globex") ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() @@ -639,6 +954,29 @@ func TestEmbeddedNATS_Consume_StopIsNotAFailure(t *testing.T) { t.Fatalf("our own stop was reported as a failure: %v", err) case <-time.After(time.Second): } + // Nor is a queue opened after the stop joined to it. + require.NoError(t, e.SetMaxBytes(ctx, "initech", testBudget)) + _, err = e.js.Consumer(ctx, "INGEST_initech", "stopped") + require.ErrorIs(t, err, jetstream.ErrConsumerNotFound) +} + +// The fetch-ahead asked for is shared by the tenants' queues, at least one +// each, so the rows held client-side stay about what the caller asked for +// however many tenants there are. +func TestFanIn_SharesThePrefetch(t *testing.T) { + t.Parallel() + handles := func(n int) map[tenant.ID]jetstream.Consumer { + m := map[tenant.ID]jetstream.Consumer{} + for i := range n { + m[tenant.ID(fmt.Sprint(i))] = nil + } + return m + } + assert.Equal(t, 500, (&fanIn{prefetch: 500, handles: handles(1)}).share()) + assert.Equal(t, 250, (&fanIn{prefetch: 500, handles: handles(2)}).share()) + assert.Equal(t, 1, (&fanIn{prefetch: 500, handles: handles(1000)}).share(), "at least one per tenant") + assert.Equal(t, 500, (&fanIn{prefetch: 500}).share(), "no tenant yet") + assert.Zero(t, (&fanIn{handles: handles(3)}).share(), "0 leaves the client default") } // Nothing lands on the default tenant by omission (#583): the tenant is a @@ -652,7 +990,7 @@ func TestEmbeddedNATS_Publish_RefusesATopicWithoutATenant(t *testing.T) { require.Error(t, e.Publish(ctx, topic, []byte("x")), "%+v", topic) require.Error(t, e.ReplaySince(ctx, topic, time.Time{}, func([]byte) bool { return true }), "%+v", topic) } - s, err := e.stream(ctx, ingestStream) + s, err := e.stream(ctx, "INGEST_0") require.NoError(t, err) st, err := s.state(ctx, "") require.NoError(t, err) @@ -661,7 +999,7 @@ func TestEmbeddedNATS_Publish_RefusesATopicWithoutATenant(t *testing.T) { // Two tenants, one table name: a replay of one never carries the other's rows. func TestEmbeddedNATS_ReplaySince_IsPerTenant(t *testing.T) { - e := newTestEmbedded(t) + e := newTestEmbedded(t, "acme", "globex") ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() @@ -677,34 +1015,69 @@ func TestEmbeddedNATS_ReplaySince_IsPerTenant(t *testing.T) { assert.Equal(t, []string{"acme1", "acme2"}, got) } -// A message published before the tenant led the subject (#583 story 5) is -// still delivered after the upgrade — the durable consumers filter ingest.> -// — and reads as the default tenant's, so it inserts, streams and parks as -// it did. -func TestEmbeddedNATS_PreTenantSubjectsStillDeliver(t *testing.T) { - e := newTestEmbedded(t) +// A boot over a directory an earlier build wrote deletes the pair of streams +// it kept for every tenant together: their subjects overlap every tenant's, +// so no tenant's queue could open beside them. +func TestNewEmbedded_DeletesTheStreamsAnEarlierBuildShared(t *testing.T) { + dir := t.TempDir() + old, err := NewEmbedded(dir) + require.NoError(t, err) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() - - _, err := e.js.Publish(ctx, "ingest.events", []byte("old")) + for name, subj := range map[string]string{legacyIngestStream: "ingest.>", legacyDLQStream: "dlq.>"} { + _, err := old.js.CreateStream(ctx, jetstream.StreamConfig{Name: name, Subjects: []string{subj}}) + require.NoError(t, err) + } + _, err = old.js.Publish(ctx, "ingest.events", []byte("pre-tenant")) require.NoError(t, err) + require.NoError(t, old.Close()) - got := make(chan *Message, 1) - require.NoError(t, e.Subscribe(ctx, "upgrade", func(msg *Message) error { - got <- msg - return nil - })) - var msg *Message - select { - case msg = <-got: - case <-time.After(5 * time.Second): - t.Fatal("timed out waiting for delivery") + e := openEmbedded(t, dir) + for _, name := range []string{legacyIngestStream, legacyDLQStream} { + _, err := e.js.Stream(ctx, name) + require.ErrorIs(t, err, jetstream.ErrStreamNotFound, name) } - assert.Equal(t, Topic{Tenant: tenant.Default, Table: "events"}, msg.Topic()) - require.NoError(t, e.DeadLetter(ctx, msg)) - dlq, err := e.js.Stream(ctx, dlqStream) + require.NoError(t, e.SetMaxBytes(ctx, tenant.Default, testBudget)) + require.NoError(t, e.Publish(ctx, Topic{Tenant: tenant.Default, Table: "events"}, []byte("x"))) +} + +// A boot takes stock of the queues on disk: each keeps the budget it last +// had, and a consumer created afterwards is held on every one of them — a +// tenant no longer served, which is never given a budget again, included — +// so what such a tenant had queued still reaches the worker. +func TestNewEmbedded_TakesStockOfTheQueuesOnDisk(t *testing.T) { + dir := t.TempDir() + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + first, err := NewEmbedded(dir) + require.NoError(t, err) + require.NoError(t, first.SetMaxBytes(ctx, "acme", 8<<20)) + for i := range 2 { + require.NoError(t, first.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte{byte(i)})) + } + require.NoError(t, first.Close()) + + e := openEmbedded(t, dir) + assert.Equal(t, int64(8<<20), e.MaxBytes("acme"), "the budget is read back") + + got := make(chan byte, 2) + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer", MaxAckPending: 10}) require.NoError(t, err) - raw, err := dlq.GetLastMsgForSubject(ctx, "dlq.events") + stop, _, err := cons.Consume(func(msg *Message) { + _ = msg.Ack() + got <- msg.Data[0] + }, 4) require.NoError(t, err) - assert.Equal(t, []byte("old"), raw.Data, "parked under the tail it arrived on") + t.Cleanup(stop) + for i := range 2 { + select { + case b := <-got: + assert.Equal(t, byte(i), b) + case <-time.After(5 * time.Second): + t.Fatal("the queued rows of a tenant given no budget this boot were not delivered") + } + } + // And a publish to it opens nothing new: the queue is there at its budget. + require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"))) + assert.Equal(t, int64(8<<20), streamConfig(t, e, "INGEST_acme").MaxBytes) } diff --git a/internal/mq/mq.go b/internal/mq/mq.go index 2c5de566..34620f85 100644 --- a/internal/mq/mq.go +++ b/internal/mq/mq.go @@ -141,24 +141,30 @@ func WithHeader(key, value string) PublishOpt { } } -// ErrQueueFull is returned by Publisher.Publish when the ingest queue is at -// its byte budget and refuses new events — the backpressure signal the API -// turns into a 503 with Retry-After. +// ErrQueueFull is returned by Publisher.Publish when the topic's tenant's +// ingest queue refuses new events — it is at its byte budget, or the tenant +// has no queue open yet — the backpressure signal the API turns into a 503 +// with Retry-After. var ErrQueueFull = errors.New("ingest queue is full") // Publisher appends events to the ingest queue. type Publisher interface { - // Publish stores data as one event on topic. ErrQueueFull when the queue - // is at its byte budget. + // Publish stores data as one event on topic, in the ingest queue of the + // topic's tenant. ErrQueueFull when that queue is at its byte budget, or + // the tenant has no queue open yet (see Broker.SetMaxBytes). Publish(ctx context.Context, topic Topic, data []byte, opts ...PublishOpt) error Close() error } // Subscriber delivers every event on the ingest queue, across all tenants -// and topics. +// and topics: each tenant's in the order it was published, and different +// tenants' concurrently. type Subscriber interface { // Subscribe registers a handler for incoming events under a durable - // consumer named consumerName. + // consumer named consumerName, held on every tenant's queue — those + // opened after Subscribe included. The handler runs on one delivery + // goroutine per tenant, one message at a time, so it must be safe to + // call concurrently for different tenants. // // CONTRACT: If the handler intends to return an error to trigger automatic // redelivery, it MUST NOT manually call msg.Ack() or msg.Nak() beforehand. @@ -181,27 +187,32 @@ type ConsumerConfig struct { // AckWait is the redelivery timeout: a message not acked within it is // delivered again. AckWait time.Duration - // MaxAckPending caps unacked messages broker-side; delivery pauses when - // hit (backpressure). + // MaxAckPending caps unacked messages broker-side, per tenant: delivery + // of a tenant's events pauses when that tenant's unacked ones hit it + // (backpressure), and no other tenant's does. MaxAckPending int } // Consumer is a live durable consumer created by ConsumerManager. type Consumer interface { - // Consume delivers each message to handler on the client's delivery - // goroutine, so a handler that blocks holds delivery back — that is the - // backpressure the ingest worker relies on. Up to prefetch messages are - // fetched ahead (0 = the client default). The returned stop asks delivery - // to end and returns without waiting: a handler invocation already in - // flight, or one for a message already queued client-side, may still run - // after stop returns, so a handler must not write to anything the caller - // tears down right after stopping. + // Consume delivers each message to handler on a delivery goroutine of + // its tenant's: one per tenant, so a tenant's messages arrive in order, + // one at a time, while different tenants' arrive concurrently — handler + // must be safe for that. A handler that blocks holds back its tenant's + // delivery — that is the backpressure the ingest worker relies on. About + // prefetch messages are fetched ahead across the tenants together, at + // least one per tenant (0 = the client default, per tenant). The returned + // stop asks delivery to end and returns without waiting: a handler + // invocation already in flight, or one for a message already queued + // client-side, may still run after stop returns, so a handler must not + // write to anything the caller tears down right after stopping. // // Delivery can also end on its own after Consume has returned: the broker // or the client gives up on the consumer (it was deleted, the connection - // closed). That is reported on failed — exactly one error, and nothing - // once stop has been called — because no message will ever arrive to say - // so. A caller that ignores failed waits forever on a dead consumer. + // closed), or a tenant's queue opened later could not be joined. That is + // reported on failed — exactly one error, and nothing once stop has been + // called — because no message will ever arrive to say so. A caller that + // ignores failed waits forever on a dead consumer. Consume(handler func(msg *Message), prefetch int) (stop func(), failed <-chan error, err error) } @@ -209,7 +220,8 @@ type Consumer interface { // broker's reason when it gave one. var ErrDeliveryEnded = errors.New("consumer delivery ended") -// ConsumerManager creates durable consumers on the ingest queue. A delivered +// ConsumerManager creates durable consumers on the ingest queue, held on +// every tenant's queue — those opened later included. A delivered // Message.Ctx is the ctx given to CreateConsumer: unlike Subscriber, the // consumer path does not extract the trace context carried in the message // headers, because its one consumer (the ingest worker) batches across @@ -220,51 +232,53 @@ type ConsumerManager interface { // DeadLetterer parks messages on the dead-letter queue. type DeadLetterer interface { - // DeadLetter stores msg's data on the dead-letter queue under msg's topic, - // with the headers the options set. It does not ack msg: the caller acks - // once the parking is confirmed, so a failure here leaves the original to - // be redelivered. + // DeadLetter stores msg's data on the dead-letter queue of msg's tenant, + // under msg's topic, with the headers the options set. It does not ack + // msg: the caller acks once the parking is confirmed, so a failure here + // leaves the original to be redelivered. DeadLetter(ctx context.Context, msg *Message, opts ...PublishOpt) error } -// DeadLetterCounts is what is parked on the dead-letter queue. +// DeadLetterCounts is what is parked on one tenant's dead-letter queue. type DeadLetterCounts struct { - // Tables maps table name → parked messages, for the tables asked about, - // summed across tenants: one queue serves every tenant until each has its - // own (#583 story 5b), so one count covers them all. Scope is - // not broken out yet (it is inert until #235): a message parked under a - // scoped topic counts under "table.scope", not under its table. + // Tables maps table name → parked messages, for the tables asked about. + // Scope is not broken out yet (it is inert until #235): a message parked + // under a scoped topic counts under "table.scope", not under its table. Tables map[string]uint64 - // Total is every parked message, whatever the filter. + // Total is every parked message of the tenant, whatever the filter. Total uint64 } // ErrNoDeadLetterQueue is returned by DeadLetterStats.DeadLetterCounts when -// the dead-letter queue does not exist (nothing can have been parked). Any -// other failure to read it is a plain error. +// the tenant has no dead-letter queue (nothing can have been parked for it). +// Any other failure to read it is a plain error. var ErrNoDeadLetterQueue = errors.New("dead-letter queue not found") -// DeadLetterStats reports on the dead-letter queue. +// DeadLetterStats reports on the dead-letter queues. type DeadLetterStats interface { - // DeadLetterCounts counts parked messages per table; a non-empty table - // narrows Tables to that one (its unscoped messages, under any tenant — - // see DeadLetterCounts.Tables). - DeadLetterCounts(ctx context.Context, table string) (DeadLetterCounts, error) + // DeadLetterCounts counts tenant id's parked messages per table — a + // tenant served, rejected, or removed alike, for as long as its queue is + // kept. A non-empty table narrows Tables to that one (its unscoped + // messages). + DeadLetterCounts(ctx context.Context, id tenant.ID, table string) (DeadLetterCounts, error) } // ErrConsumerNotFound is returned by Purger.PurgeAcked when the named -// consumer does not exist (yet). +// consumer does not exist (yet) on a tenant's queue. var ErrConsumerNotFound = errors.New("consumer not found") // Purger reclaims ingest-queue storage. type Purger interface { - // PurgeAcked removes the ingest events that are BOTH acknowledged by the - // named durable consumer (everything before its first unacked event) AND - // stored before olderThan. Either bound alone keeps the event: unacked - // events are not yet written, and recent ones are still needed for replay. - // Reports whether anything was removed. ErrConsumerNotFound when the - // consumer has not been created. - PurgeAcked(ctx context.Context, consumer string, olderThan time.Time) (purged bool, err error) + // PurgeAcked removes, from each tenant's ingest queue, the events that + // are BOTH acknowledged by the named durable consumer (everything before + // its first unacked event) AND stored before that tenant's cutoff in + // olderThan. Either bound alone keeps the event: unacked events are not + // yet written, and recent ones are still needed for replay. A tenant + // olderThan does not name — one no longer served — keeps no history: + // everything it has acknowledged goes. Reports whether anything was + // removed. ErrConsumerNotFound when the consumer has not been created on + // some tenant's queue; the other tenants' are purged all the same. + PurgeAcked(ctx context.Context, consumer string, olderThan map[tenant.ID]time.Time) (purged bool, err error) } // Replayer re-delivers stored events for SSE gap-fill. @@ -278,7 +292,7 @@ type Replayer interface { } // Broker is everything the process wiring needs from the MQ: every interface -// above plus the lifecycle and the byte budget. EmbeddedNATS is the one +// above plus the lifecycle and the byte budgets. EmbeddedNATS is the one // implementation; internal/app depends on this, not on it. type Broker interface { Publisher @@ -288,15 +302,17 @@ type Broker interface { DeadLetterStats Purger Replayer - // SetMaxBytes applies a new byte budget (the hot-reloadable - // mq.max_bytes_gb) to the queues as a whole — how it is split between - // them is the implementation's. On an error the implementation restores - // the previous budget where it can (best effort: the error says when it - // could not, and a canceled ctx abandons the restore too), and MaxBytes - // keeps reporting the previous budget so the next call retries. - // MaxBytes reports the budget last applied in full. - SetMaxBytes(ctx context.Context, maxBytes int64) error - MaxBytes() int64 + // SetMaxBytes applies tenant id's byte budget (its hot-reloadable + // mq.max_bytes_gb) to that tenant's queues — how it is split between them + // is the implementation's — opening them if the tenant has none yet. No + // other tenant's queues are touched. On an error the implementation + // restores the previous budget where it can (best effort: the error says + // when it could not, and a canceled ctx abandons the restore too), and + // MaxBytes keeps reporting the previous budget so the next call retries. + // MaxBytes reports the budget last applied in full for id, 0 when none + // has been. + SetMaxBytes(ctx context.Context, id tenant.ID, maxBytes int64) error + MaxBytes(id tenant.ID) int64 // Stats reports the broker counters the system gauges observe. Stats() (observability.MQStats, error) } diff --git a/internal/mq/subject.go b/internal/mq/subject.go index 8d67ace4..489ad241 100644 --- a/internal/mq/subject.go +++ b/internal/mq/subject.go @@ -12,25 +12,53 @@ import ( // The embedded broker's naming. Private to this package: everything else // addresses events by Topic. const ( - // ingestStream / dlqStream are the JetStream stream names. Hardcoded — the - // embedded NATS server is private to the WaveHouse process, so there is - // nothing to namespace against. - ingestStream = "WAVEHOUSE" - dlqStream = "WAVEHOUSE_DLQ" + // Each tenant's queue is a pair of JetStream streams named after it: + // INGEST_ and DLQ_. The prefixes differ in their first + // letter, so no tenant id makes one kind's name the other's, and the + // tenant grammar (tenant.Parse: letters, digits, '_' and '-', at most + // tenant.MaxLen bytes) keeps every name inside JetStream's. No namespacing + // beyond that: the embedded server is private to the WaveHouse process. + ingestStreamPrefix = "INGEST_" + dlqStreamPrefix = "DLQ_" + + // legacyIngestStream / legacyDLQStream are the one pair an earlier build + // kept for every tenant. Their subjects (ingest.> and dlq.>) overlap every + // tenant's, and JetStream refuses a stream whose subjects overlap + // another's, so NewEmbedded deletes them. + legacyIngestStream = "WAVEHOUSE" + legacyDLQStream = "WAVEHOUSE_DLQ" // A topic's subject is .
[.]: the tenant id // verbatim — its grammar (tenant.Parse) admits only letters, digits, '_' // and '-', so it is one token as it is — then the table and scope each // as one encoded token. Tenant first so one wildcard selects a tenant's - // traffic (ingest.acme.>). The same topic has the same tail on both - // streams, so parking a message on the DLQ is a prefix swap. + // traffic (ingest.acme.>), which is what the tenant's streams hold. The + // same topic has the same tail on both kinds, so parking a message on the + // dead-letter queue is a prefix swap. ingestPrefix = "ingest." dlqPrefix = "dlq." - - ingestAll = ingestPrefix + ">" // every topic on the ingest stream - dlqAll = dlqPrefix + ">" // every topic on the DLQ stream ) +// ingestStreamName / dlqStreamName name tenant id's two streams. +func ingestStreamName(id tenant.ID) string { return ingestStreamPrefix + string(id) } +func dlqStreamName(id tenant.ID) string { return dlqStreamPrefix + string(id) } + +// tenantSubjects is every subject of tenant id's under prefix: what its +// stream of that kind holds. +func tenantSubjects(prefix string, id tenant.ID) string { return prefix + string(id) + ".>" } + +// streamTenant recovers the tenant a stream name carries under prefix, false +// for any other name: a stream of the other kind, a legacy one, or a name no +// tenant id could have produced. +func streamTenant(prefix, name string) (tenant.ID, bool) { + rest, ok := strings.CutPrefix(name, prefix) + if !ok { + return "", false + } + id, err := tenant.Parse(rest) + return id, err == nil +} + // encodeToken converts any table or scope name into a safe, single NATS // subject token. It preserves alphanumerics and underscores, but // percent-encodes everything else (so '.', ' ', '*' and '>' can never split @@ -66,28 +94,28 @@ func subject(prefix string, t Topic) (string, error) { } // topicKey is the tail of a subject carrying prefix — the key() of the topic -// it was published on, or a one-token tail written before the tenant led the -// subject (see parseTopicKey). A trim, no decoding. +// it was published on. A trim, no decoding. func topicKey(prefix, subj string) string { return strings.TrimPrefix(subj, prefix) } +// keyTenant is the tenant a topic key leads with — the token that decides +// which tenant's stream its subject lands in — whether or not the rest of +// the key parses. +func keyTenant(key string) (tenant.ID, bool) { + first, _, _ := strings.Cut(key, ".") + id, err := tenant.Parse(first) + return id, err == nil +} + // parseTopicKey recovers the Topic from a subject tail. Three tokens are -// tenant, table and scope; two are tenant and table. One token is the form -// this package wrote before the tenant led the subject (#583 story 5) and -// reads as tenant.Default's table: every event of that era was the default -// tenant's, and the durable consumers still deliver them after the upgrade, -// as the dead-letter queue still holds them. A tail this package could not -// have written — more tokens, a token that does not decode, a tenant outside -// the grammar — cannot be split reliably, so the whole of it becomes the -// table of no tenant rather than being dropped. +// tenant, table and scope; two are tenant and table. A tail this package +// could not have written — one token, more than three, a token that does not +// decode, a tenant outside the grammar — cannot be split reliably, so the +// whole of it becomes the table of no tenant rather than being dropped. func parseTopicKey(tail string) Topic { parts := strings.Split(tail, ".") switch len(parts) { - case 1: - if table, err := decodeToken(parts[0]); err == nil && table != "" { - return Topic{Tenant: tenant.Default, Table: table} - } case 2, 3: id, idErr := tenant.Parse(parts[0]) table, tableErr := decodeToken(parts[1]) diff --git a/internal/mq/subject_test.go b/internal/mq/subject_test.go index e4b8bb82..67536e4e 100644 --- a/internal/mq/subject_test.go +++ b/internal/mq/subject_test.go @@ -1,8 +1,10 @@ package mq import ( + "strings" "testing" + "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -126,15 +128,6 @@ func TestTopicKey_IsInjective(t *testing.T) { assert.Equal(t, Topic{Tenant: "0", Table: "a", Scope: "b"}.key(), Topic{Tenant: "0", Table: "a", Scope: "b"}.key()) } -// The form written before the tenant led the subject (#583 story 5) is the -// default tenant's: it is what the durable consumers deliver across the -// upgrade, and what the dead-letter queue keeps holding after it. -func TestParseTopicKey_PreTenantTailIsTheDefaultTenants(t *testing.T) { - t.Parallel() - assert.Equal(t, Topic{Tenant: "0", Table: "events"}, parseTopicKey("events")) - assert.Equal(t, Topic{Tenant: "0", Table: "default.clicks"}, parseTopicKey("default%2Eclicks")) -} - func TestParseTopicKey_ForeignTailKeepsItself(t *testing.T) { t.Parallel() // Subjects this package did not write still yield one usable topic, of @@ -144,9 +137,50 @@ func TestParseTopicKey_ForeignTailKeepsItself(t *testing.T) { "0.bad%2Gtoken", // a token that does not decode "a%2Eb.events", // a tenant outside the grammar ".events", // a topic whose tenant was never set + "events", // one token: no tenant leads it "bad%2G", // one token that does not decode } { assert.Equal(t, Topic{Table: tail}, parseTopicKey(tail), tail) } assert.Equal(t, Topic{}, parseTopicKey("")) } + +// The tenant a key leads with picks the stream its subject lands in, so it +// is read off the first token whatever the rest of the key holds. +func TestKeyTenant(t *testing.T) { + t.Parallel() + for key, want := range map[string]tenant.ID{"acme.t": "acme", "a.b.c.d": "a", "0.bad%2G": "0"} { + id, ok := keyTenant(key) + assert.True(t, ok, key) + assert.Equal(t, want, id, key) + } + for _, key := range []string{"", ".events", "a%2Eb.events"} { + _, ok := keyTenant(key) + assert.False(t, ok, key) + } +} + +// Every tenant's two streams have names of their own: no id makes one +// kind's name another stream's, none is a stream an earlier build shared, +// and each name gives its tenant back. +func TestStreamNames_NeverCollide(t *testing.T) { + t.Parallel() + ids := []tenant.ID{"0", "acme", "DLQ", "DLQ_acme", "INGEST", "INGEST_acme", "_", "-", "WAVEHOUSE", tenant.ID(strings.Repeat("a", tenant.MaxLen))} + seen := map[string]tenant.ID{legacyIngestStream: "", legacyDLQStream: ""} + for _, id := range ids { + for prefix, name := range map[string]string{ingestStreamPrefix: ingestStreamName(id), dlqStreamPrefix: dlqStreamName(id)} { + other, dup := seen[name] + assert.False(t, dup, "%s names a stream of %q's too", name, other) + seen[name] = id + back, ok := streamTenant(prefix, name) + assert.True(t, ok, name) + assert.Equal(t, id, back, name) + } + } + for _, name := range []string{legacyIngestStream, legacyDLQStream, "INGEST_a.b", "DLQ_"} { + for _, prefix := range []string{ingestStreamPrefix, dlqStreamPrefix} { + _, ok := streamTenant(prefix, name) + assert.False(t, ok, "%s is no tenant's %s stream", name, prefix) + } + } +} diff --git a/internal/settings/settings.go b/internal/settings/settings.go index d0f836e5..55ec089d 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -163,10 +163,10 @@ type TableDedupe struct { } // DLQConfig gates the Dead Letter Queue: whether a row that still fails -// after the row-by-row isolation retry is parked on the WAVEHOUSE_DLQ stream -// (and its original acked) or left unacked to be redelivered indefinitely. -// The stream itself always exists — it is an empty limits-policy stream -// until something lands on it — so the switch is purely behavioral and +// after the row-by-row isolation retry is parked on the tenant's dead-letter +// queue (and its original acked) or left unacked to be redelivered +// indefinitely. The queue exists from the moment the tenant is first served — +// empty until something lands on it — so the switch is purely behavioral and // resolves per table through the same override cascade as dedupe. type DLQConfig struct { Enabled *bool `json:"enabled"` @@ -219,14 +219,15 @@ type StreamConfig struct { GapWindowMinutes *int `json:"gap_window_minutes"` } -// MQConfig sizes the embedded JetStream streams on disk. +// MQConfig sizes the tenant's message queue on disk. type MQConfig struct { - // MaxBytesGB caps the WAVEHOUSE ingest stream (the DLQ stream gets a - // tenth of it). Must be >= 1. A reload updates the live streams in - // place: growing takes effect immediately; shrinking below what is - // currently buffered makes the ingest stream refuse new publishes - // (DiscardNew → 503 backpressure) until the worker drains it — nothing - // already buffered is dropped. + // MaxBytesGB caps the tenant's ingest queue (its dead-letter queue gets a + // tenth of it). Must be >= 1. A reload updates the live queues in place: + // growing takes effect immediately; shrinking below what is currently + // buffered makes the ingest queue refuse new publishes (DiscardNew → 503 + // backpressure) until the worker drains it — nothing already buffered is + // dropped — and a dead-letter queue holding more than a tenth of the new + // budget keeps what it holds rather than dropping its oldest rows. MaxBytesGB *int `json:"max_bytes_gb"` } diff --git a/internal/settings/store.go b/internal/settings/store.go index d1493917..f68a2fbf 100644 --- a/internal/settings/store.go +++ b/internal/settings/store.go @@ -193,7 +193,7 @@ func (s *Store) GapWindow() time.Duration { return time.Duration(*s.doc().Config.Stream.GapWindowMinutes) * time.Minute } -// MQMaxBytes returns the ingest stream's disk budget in bytes. +// MQMaxBytes returns the disk budget of the tenant's ingest queue in bytes. func (s *Store) MQMaxBytes() int64 { return int64(*s.doc().Config.MQ.MaxBytesGB) << 30 } diff --git a/internal/stream/subscriber.go b/internal/stream/subscriber.go index 54883492..5952aa1a 100644 --- a/internal/stream/subscriber.go +++ b/internal/stream/subscriber.go @@ -55,15 +55,17 @@ type Subscriber struct { // which reads nothing here but may run alongside the fan-out. // // It does NOT make Hub.deliver's check→send→record sequence atomic, and - // deliver does not need it to be: Broadcast runs on ONE goroutine — the - // single jetstream Consume callback the hub bridge registers in - // internal/app, invoked inline per message — so no two events race - // to announce the same connection's columns. A future change that fans - // Broadcast out across goroutines must hold a lock across that whole - // sequence, or two events will both send an announcement (harmless) while a - // third slips a row between a check and its record (not harmless: the client - // zips it against the previous list). Replay does not touch this field at - // all — it tracks drift in its own closure; see Hub.ReplayProjector. + // deliver does not need it to be: the hub bridge registered in + // internal/app calls Broadcast inline per message, on one delivery + // goroutine per tenant (mq.Subscriber), and a connection subscribes to one + // tenant's topic — so all of its events come from that one goroutine, and + // no two events race to announce the same connection's columns. A future + // change that fans one tenant's Broadcasts out across goroutines must hold + // a lock across that whole sequence, or two events will both send an + // announcement (harmless) while a third slips a row between a check and its + // record (not harmless: the client zips it against the previous list). + // Replay does not touch this field at all — it tracks drift in its own + // closure; see Hub.ReplayProjector. schemaMu sync.Mutex // lastSchema is the signature of the column list most recently announced to // this connection ("" ⇒ none yet). Rows travel positionally, so a client that diff --git a/internal/testutil/mocks.go b/internal/testutil/mocks.go index 1b8cf358..44f3ebe6 100644 --- a/internal/testutil/mocks.go +++ b/internal/testutil/mocks.go @@ -221,10 +221,10 @@ type MockPurger struct { // PurgeCall records one PurgeAcked call. type PurgeCall struct { Consumer string - OlderThan time.Time + OlderThan map[tenant.ID]time.Time } -func (m *MockPurger) PurgeAcked(_ context.Context, consumer string, olderThan time.Time) (bool, error) { +func (m *MockPurger) PurgeAcked(_ context.Context, consumer string, olderThan map[tenant.ID]time.Time) (bool, error) { m.mu.Lock() defer m.mu.Unlock() m.Calls = append(m.Calls, PurgeCall{Consumer: consumer, OlderThan: olderThan}) @@ -233,13 +233,21 @@ func (m *MockPurger) PurgeAcked(_ context.Context, consumer string, olderThan ti // ── Mock mq.DeadLetterStats ────────────────────────────────────── -// MockDeadLetterStats implements mq.DeadLetterStats with a canned answer. +// MockDeadLetterStats implements mq.DeadLetterStats with a canned answer, +// recording the tenant and table each call asked about. type MockDeadLetterStats struct { Counts mq.DeadLetterCounts Err error + + mu sync.Mutex + Tenant tenant.ID + Table string } -func (m *MockDeadLetterStats) DeadLetterCounts(context.Context, string) (mq.DeadLetterCounts, error) { +func (m *MockDeadLetterStats) DeadLetterCounts(_ context.Context, id tenant.ID, table string) (mq.DeadLetterCounts, error) { + m.mu.Lock() + defer m.mu.Unlock() + m.Tenant, m.Table = id, table return m.Counts, m.Err } diff --git a/internal/testutil/testutil.go b/internal/testutil/testutil.go index 371cdcd4..db119686 100644 --- a/internal/testutil/testutil.go +++ b/internal/testutil/testutil.go @@ -14,6 +14,7 @@ import ( "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" ) @@ -39,6 +40,24 @@ func NewTestSchemaRegistry(t testing.TB, tables []*discovery.TableSchema) *disco // hardcoding the same literal twice. const TestServerVersion = "24.8.1.1" +// NewEmbeddedMQ starts the embedded broker over a temporary directory, closed +// by the test framework, with a queue open for each of tenants — +// tenant.Default when none is named — at maxBytes: a tenant has a queue once +// its budget is applied, as the wiring does for every tenant it serves. +func NewEmbeddedMQ(t testing.TB, maxBytes int64, tenants ...tenant.ID) *mq.EmbeddedNATS { + t.Helper() + emb, err := mq.NewEmbedded(t.TempDir()) + require.NoError(t, err) + t.Cleanup(func() { _ = emb.Close() }) + if len(tenants) == 0 { + tenants = []tenant.ID{tenant.Default} + } + for _, id := range tenants { + require.NoError(t, emb.SetMaxBytes(context.Background(), id, maxBytes)) + } + return emb +} + // schemaConn is a mock driver.Conn serving exactly the queries Refresh issues: // the SELECT timezone() (always "UTC") and SELECT version() probes, the // system.columns scan (rows synthesized from tables), and the system.tables DDL From 8ad98656e412709b5ab879379de50cd49321b493 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 18:25:18 -0400 Subject: [PATCH 02/15] fix(mq): reopen detached from the request, and the upgrade runbook swept --- docs/src/content/docs/api.md | 6 +-- docs/src/content/docs/architecture.md | 9 +++-- docs/src/content/docs/deployment.md | 8 ++-- docs/src/content/docs/durability.md | 4 +- docs/src/content/docs/ingest-pipeline.md | 10 ++--- docs/src/content/docs/sdk/streaming.md | 2 +- docs/src/content/docs/settings-directory.mdx | 4 +- internal/ingest/worker.go | 25 ++++++------ internal/ingest/worker_test.go | 6 +-- internal/mq/embedded.go | 13 +++++- internal/mq/embedded_test.go | 42 ++++++++++++++++++++ internal/settings/settings.go | 7 ++-- 12 files changed, 96 insertions(+), 40 deletions(-) diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 1634aaab..ad712b82 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -274,7 +274,7 @@ The body is a **flat JSON object** whose keys must match column names in the tar | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | | 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | | 500 | `{"error":"publish failed"}` | Message queue error | -| 503 | `{"error":"service unavailable"}` | NATS JetStream stream full (backpressure). Response includes `Retry-After: 30` header. | +| 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure, for that tenant alone) or not open (see [Message Queue](/settings-directory#message-queue)). Response includes `Retry-After: 30` header. | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -385,7 +385,7 @@ A `200` is returned whenever the body was read and the records were processed | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, …"}` (declared variant: `Content-Type "text/plain": ingest requires one of …` — see the note above on how declarations are echoed; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | The request declared no `Content-Type`, one whose media type is unsupported or does not parse, a comma-bearing value that does not parse as a single media type, or repeated header lines that disagree — different formats, or one supported and one not. Checked before the body is parsed | | 500 | `{"error":"publish failed"}` / `{"error":"dedupe failed"}` | Message-queue or dedup-backend failure mid-batch | -| 503 | `{"error":"service unavailable"}` | NATS JetStream full (backpressure) mid-batch; includes `Retry-After: 30` | +| 503 | `{"error":"service unavailable"}` | The tenant's ingest queue is full (backpressure) or not open, mid-batch; includes `Retry-After: 30` | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | :::caution[At-least-once on retry] @@ -876,7 +876,7 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format` (a pre-v2 message has no `format` field at all, which is how it presents here), or `columns` and `row` that do not pair — is parked without ever reaching a table batch, which is what an operator sees after upgrading across the wire change without draining first. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. +When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Use `GET /v1/ops/dlq/stats` to monitor DLQ depth, per tenant (`?tenant=`). diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 6eaf3d54..92508dac 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -190,7 +190,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi ### `tenant/` — Tenant Identifier -- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper folds over the tenants served; each served tenant has a schema registry of its own, built with its id (story 6). +- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper hands the MQ each served tenant's own gap window (`gapWindows`); each served tenant has a schema registry of its own, built with its id (story 6). ### `chconn/` — ClickHouse Connection Pools @@ -228,7 +228,7 @@ Client POST /v1/ingest?table={table} field is published un-deduped + logged/counted, or rejected under require_id) → Publish to NATS JetStream (ingest.{tenant}.{table}) → 200 OK returned immediately - → (If NATS stream is full: 503 + Retry-After header) + → (If the tenant's NATS stream is full, or not open: 503 + Retry-After header) Ingest worker pipeline (StartIngestWorker): ← JetStream pull consumer (buffer-consumer) on ingest.> @@ -251,9 +251,10 @@ Ingest worker pipeline (StartIngestWorker): a no/invalid-token request (resolved to default_role, not admin in a production config) cannot reach the proxy.) -Active Sweeper (async goroutine, every 60s): +Active Sweeper (async goroutine, every 60s), on each tenant's stream: → Read buffer consumer's AckFloor (highest contiguous ACKed seq) - → Binary search for first message within the gap window (the longest among the tenants served) + → Binary search for first message within that tenant's own gap window + (none for a tenant no longer served) → Purge target = MIN(ack_floor + 1, gap_window_seq) → Purge all messages below target from JetStream ``` diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 095b4990..b2bac204 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -421,9 +421,9 @@ WaveHouse discovers this schema on startup and refreshes it every `schema.refres The NATS envelope changed shape in this release: the row now travels positionally, with `format`, `columns` and `row` replacing `data`. **The new worker cannot read a message published by an older version** — it carries no `format`, so there is no way to say which value belongs to which column. -This affects the streaming surface too, and more quietly. SSE gap-fill (`?since=` / `Last-Event-ID`) reads the same stream, and the hub refuses a pre-v2 envelope on the same missing `format` the worker does — it is withheld from every role with **no error and no frame**, and the stream side files no DLQ entry of its own — the worker's copy of the same message is what lands in `dlq.{table}` (next paragraph). Because the stream keeps ACKed messages until the sweeper purges past `stream.gap_window_minutes` (15 by default), this outlives a *correct* drain: for that window, any replay spanning the upgrade silently omits the pre-upgrade events. Clients that need them should backfill over REST. +This affects the streaming surface too, and more quietly. SSE gap-fill (`?since=` / `Last-Event-ID`) replays from the queue, and the upgrade deletes the old one (next paragraph) — the acknowledged history kept for replay included — so this outlives a *correct* drain: any replay spanning the upgrade silently omits the pre-upgrade events, with **no error and no frame**. Clients that need them should backfill over REST. -On the worker side the outcome depends on the DLQ. **With the DLQ enabled for the table**, the message is parked on `dlq.{table}` with `X-DLQ-*` headers and is recoverable by hand — but re-ingest each parked envelope's inner `data` object as a fresh `POST /v1/ingest`; republishing the envelope as-is onto `ingest.{tenant}.{table}` fails the same `format` check and simply re-parks it. **With the DLQ switched off for the table, it is permanently lost**: acked and dropped with an `ERROR` log and a `wavehouse_ingest_poison_total` increment carrying `disposition="dropped"`, unrecoverable from either the ingest stream or the DLQ, because a message that can never insert must not redeliver forever. Draining first is cheaper than a manual replay, and it is the only option at all where the DLQ is off. +**The upgrade does not carry the old queue over at all.** Boot deletes the earlier build's queue and dead-letter queue (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) and everything in them, logging a `WARN` with each one's message count: an event the old build had not yet inserted, and a row it had already parked, do not survive the upgrade. Draining first is the only way to keep them, and anything parked has to be replayed before the upgrade — re-ingest each parked envelope's inner `data` object as a fresh `POST /v1/ingest`. Three audits belong **before** the drain, because none of them announces itself afterwards: @@ -435,10 +435,10 @@ To drain before upgrading: 1. **Stop the producers**, or cut `/v1/ingest` at the reverse proxy. Nothing new should enter the stream. 2. **Wait for the in-flight batches to flush.** A table's batch closes on size or after `maxWait` (5s by default), so a few seconds after the last write is enough; give it longer if ClickHouse is slow or retrying. -3. **Confirm nothing is left unconsumed** before swapping binaries. Not that the stream is empty: it is dual-use, and deliberately retains ACKed messages for the SSE replay window, so a non-zero depth right after a clean drain is expected. Rows landing in ClickHouse is a success signal, **not proof the queue is drained** — when the DLQ is off for a table, a row that fails its retry is skipped without being acked, so NATS keeps redelivering it while its neighbors land. Check that nothing is still failing or redelivering, and note which signal covers which case: [`GET /v1/ops/dlq/stats`](/api#get-v1opsdlqstats--dlq-statistics) is non-zero only where the DLQ is **on**; `wavehouse_ingest_poison_total` counts unreadable envelopes on **either** setting, told apart by its `disposition` label (`parked` / `dropped`); and for a twice-failed row with the DLQ off — the case just described — the **only** signal is the `ERROR` log (`isolated bad row, DLQ disabled for table`). A clean `dlq/stats` with the DLQ off proves nothing. There is no queue-depth gauge today ([#544](https://github.com/Wave-RF/WaveHouse/issues/544) tracks the related in-flight accounting), and `wavehouse_nats_in_msgs_total` going flat is a supporting signal rather than a guarantee. Enabling the DLQ is not itself a drain — replay from `dlq.{table}` is manual. +3. **Confirm nothing is left unconsumed** before swapping binaries. Not that the stream is empty: it is dual-use, and deliberately retains ACKed messages for the SSE replay window, so a non-zero depth right after a clean drain is expected. Rows landing in ClickHouse is a success signal, **not proof the queue is drained** — when the DLQ is off for a table, a row that fails its retry is skipped without being acked, so NATS keeps redelivering it while its neighbors land. Check that nothing is still failing or redelivering, and note which signal covers which case: [`GET /v1/ops/dlq/stats`](/api#get-v1opsdlqstats--dlq-statistics) is non-zero only where the DLQ is **on**; `wavehouse_ingest_poison_total` counts unreadable envelopes on **either** setting, told apart by its `disposition` label (`parked` / `dropped`); and for a twice-failed row with the DLQ off — the case just described — the **only** signal is the `ERROR` log (`isolated bad row, DLQ disabled for table`). A clean `dlq/stats` with the DLQ off proves nothing. There is no queue-depth gauge today ([#544](https://github.com/Wave-RF/WaveHouse/issues/544) tracks the related in-flight accounting), and `wavehouse_nats_in_msgs_total` going flat is a supporting signal rather than a guarantee. Enabling the DLQ is not itself a drain — replay from `dlq.{table}` is manual, and has to happen before the upgrade deletes it. 4. **Upgrade**, then re-enable ingest. -If you skipped the drain, check `wavehouse_ingest_poison_total`, which counts both — `disposition="parked"` is recoverable from `dlq.{table}`, `disposition="dropped"` is gone — see [Dead Letter Queue](#dead-letter-queue-dlq) below. +If you skipped the drain, the boot's `WARN` line for each deleted stream (`deleted the stream an earlier build kept for every tenant together`) says how many messages went with it. ## Dead Letter Queue (DLQ) diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 98a8c9c2..353608f2 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -81,7 +81,7 @@ Read the measured p99 against these bands, which track WaveHouse's `SyncAlways` | 1–5 ms | **Good** | | 5–50 ms | **Workable** — watch bursty load | | 50 ms – 1 s | **Marginal** — relax durability once `mq.sync_interval` ([#139](https://github.com/Wave-RF/WaveHouse/issues/139)) lands, or move to faster storage | -| > 1 s | **Broken** — `create stream` will time out under load; fix the storage substrate | +| > 1 s | **Broken** — opening a tenant's queue (`open dlq stream`) will time out under load; fix the storage substrate | :::caution[macOS `fsync` lies by default] A plain `fsync()` on macOS returns once data is in the drive's volatile cache — it does **not** force a flush to NAND; only `fcntl(fd, F_FULLFSYNC)` does (NATS, Postgres, and SQLite all use it). On a Mac, any per-flush number under ~1 ms is almost certainly not a real flush — the gap between plain `fsync()` and `F_FULLFSYNC` can be ~180× on the same consumer NVMe. `fio` on macOS calls plain `fsync()`, so don't trust Mac `fio` numbers for tail-latency planning. This mostly matters when benchmarking a dev machine; production WaveHouse runs on Linux, where `fio` is honest. @@ -93,7 +93,7 @@ A self-contained `wavehouse storage-check` preflight subcommand that bakes this If you see any of these, benchmark the `/nats` volume as above: -- `create stream: ... context deadline exceeded` at startup. +- `open dlq stream: ... context deadline exceeded` when a tenant's queue first opens, at the first boot or at the reload that adopts the tenant. - Ingest p99 latency in the seconds, or occasional `200`s that take multiple seconds to return. - Intermittent `503 Service Unavailable` from `/v1/ingest` when ClickHouse is healthy (the worker can't drain fast enough because acking is `fsync`-bound). - Flaky CI or load tests that pass on fast storage and fail on a shared/virtualized host. diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index e602e0cf..b504fef4 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -61,7 +61,7 @@ Inserts also pin `input_format_null_as_default=1`. A positional row has one valu ::: :::note[ClickHouse timestamp parsing] -Inserts pin `date_time_input_format=best_effort` — the server default since ClickHouse 26.5, but on older servers the `basic` default rejects the canonical RFC 3339 form's `Z` suffix ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The ordinary spellings (zone-less date-times, 9–10-digit Unix-seconds strings) parse identically under both settings. (This is moot for anything still buffered from an older build: a message published before the v2 envelope cannot be read at all — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope).) Bare digit-strings of other lengths are the exception: `best_effort` reads them as ClickHouse's calendar/epoch shapes, where `basic` read a plain `DateTime` column's digit string of five or more digits as Unix seconds (shorter runs it rejected outright, where `best_effort` reads `"2026"` as a year): under `best_effort` `"20260711"` stores 2026-07-11, where `basic` stored 1970-08-23. `DateTime64` columns diverge the same way on calendar-shaped runs, and additionally whenever an epoch run's unit doesn't match the column scale (under `basic`, runs longer than 10 digits are ticks at the column's own scale; `best_effort` unit-detects 13/16/19-digit runs as ms/µs/ns). A producer relying on the old `basic` reading changes meaning as soon as this WaveHouse version is deployed — the pin, not a ClickHouse upgrade, is what flips the parse. +Inserts pin `date_time_input_format=best_effort` — the server default since ClickHouse 26.5, but on older servers the `basic` default rejects the canonical RFC 3339 form's `Z` suffix ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). The ordinary spellings (zone-less date-times, 9–10-digit Unix-seconds strings) parse identically under both settings. (This is moot for anything an older build buffered: the upgrade deletes it — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope).) Bare digit-strings of other lengths are the exception: `best_effort` reads them as ClickHouse's calendar/epoch shapes, where `basic` read a plain `DateTime` column's digit string of five or more digits as Unix seconds (shorter runs it rejected outright, where `best_effort` reads `"2026"` as a year): under `best_effort` `"20260711"` stores 2026-07-11, where `basic` stored 1970-08-23. `DateTime64` columns diverge the same way on calendar-shaped runs, and additionally whenever an epoch run's unit doesn't match the column scale (under `basic`, runs longer than 10 digits are ticks at the column's own scale; `best_effort` unit-detects 13/16/19-digit runs as ms/µs/ns). A producer relying on the old `basic` reading changes meaning as soon as this WaveHouse version is deployed — the pin, not a ClickHouse upgrade, is what flips the parse. ::: ## The journey of one event @@ -70,7 +70,7 @@ Inserts pin `date_time_input_format=best_effort` — the server default since Cl sequenceDiagram participant P as POST /v1/ingest participant JS as JetStream - participant CB as Consume callback + participant CB as Consume callback (the tenant's) participant D as dispatchLoop participant TL as tableLoop participant CH as ClickHouse @@ -89,11 +89,11 @@ sequenceDiagram ## Goroutine topology -The design rule is **single-owner state, lock-free**: each piece of mutable state is touched by exactly one goroutine. There are no mutexes in the hot path. +The design rule is **single-owner state, lock-free**: each piece of mutable state is touched by exactly one goroutine. There are no mutexes in the hot path. The one fan-in is at the top: each tenant's stream is delivered on a nats.go goroutine of its own, and they all send into the one `msgChan`, which is safe from all of them at once; everything from `dispatchLoop` down stays single-owner, and a full `msgChan` pauses every tenant's delivery (layer 2 below). ```mermaid flowchart TD - CB["Consume callback
(nats.go goroutine)"] -->|"msgChan (cap maxBatch*2)"| D + CB["Consume callbacks
(one nats.go goroutine per tenant stream)"] -->|"msgChan (cap maxBatch*2)"| D D["dispatchLoop
1 goroutine — owns the routing map
the ONLY ctx watcher — tracked by wg"] D -->|"per-tenant-table chan (cap maxBatch)"| T1["tableLoop: clicks
owns its batch + timer
tracked by tableWg"] D --> T2["tableLoop: events
tracked by tableWg"] @@ -213,7 +213,7 @@ Several layers throttle the pipeline, inner to outer: 1. **`batch`** flushes at `maxBatch` rows or `maxWait`. 2. **`msgChan`** (cap `maxBatch*2`) — when full, the consume callback blocks and delivery pauses. 3. **`pullMaxMessages`** — nats.go's client-side prefetch buffer in front of `msgChan`, shared by the tenants' streams (at least one message each). -4. **`maxAckPending`** — the server suspends a tenant's delivery once this many of its messages are delivered-but-unacked; no other tenant's delivery waits on it. The outermost in-memory bound. +4. **`maxAckPending`** — the server suspends a tenant's delivery once this many of its messages are delivered-but-unacked; no other tenant's delivery waits on it. The outermost in-memory bound, and a per-tenant one: while ClickHouse stalls, the worker can hold up to `maxAckPending` rows for every tenant served. 5. **`MaxBytes` + `DiscardNew`** on each tenant's stream (its `mq.max_bytes_gb` in the [settings directory](/settings-directory#message-queue), resized in place on reload) — when it fills (e.g. ClickHouse is down so nothing acks/purges), that tenant's new publishes are rejected and the API returns 503. | Knob | Default | Meaning / invariant | diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index aab6dc6e..94a7e334 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -116,7 +116,7 @@ A dropped stream reconnects on a jittered exponential backoff, capped at 30s, an :::caution[Resumption is at-least-once, and time-bounded] Delivery across a reconnect is **at-least-once**. The `Last-Event-ID` the client sends is the last event's `received_timestamp`, and the server replays from that instant *inclusively* — so the last event you already saw, and anything sharing its timestamp, arrives again. The SDK does not deduplicate live frames — `liveQuery()` makes one pass at the backfill seam, and only under an ascending order ([#449](https://github.com/Wave-RF/WaveHouse/issues/449)) — so key on `timestamp` plus your own row identity if duplicates matter. -Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. The same silence applies for one gap window after a server upgrade across the v2 ingest envelope: the hub refuses an envelope whose `format` it does not recognize and a pre-v2 message carries none, so a replay spanning that boundary omits them without an error — backfill over REST if you need them. +Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. The same silence applies across a server upgrade to this release: the server deletes the previous release's queue at boot, so a replay spanning the upgrade omits the events published before it, without an error — backfill over REST if you need them. **A column-set change across a gap-fill is a known limitation.** If the table's columns change while you are connected *and* your client replays across that change, live rows arriving after the replay may not be preceded by a fresh `event: schema` frame until the columns next change or you reconnect. The SDK drops a row whose **length** disagrees with the list it was last told, rather than zipping it under the wrong names — so an added or removed column costs you rows, not wrong ones. A **same-length** change is the residual case the arity check cannot see: a `RENAME COLUMN`, or a drop paired with an add, zips values under the wrong names until the next announcement. Reconnecting resynchronizes either way. Full schema-change handling is deferred to the schema-versioning work ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). ::: diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index c0e7a19b..41e04187 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -213,7 +213,7 @@ The `auth` block is the verifier wiring, minus the secrets. `jwks_url` (absolute A failed batch insert is retried row by row; a row that fails again on its own is a poison row. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the retry, which no row of it could pass, and every row of it is a poison row. `dlq.enabled` (seed default `true`) decides what happens to it, resolved per table (`dlq.tables.
.enabled` → global) at the moment of the failure, so a reload applies to the next poison row: - `true` — the row is published to the tenant's dead-letter stream (`DLQ_{tenant}`) under `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) with the failure in its headers, and its original is acked. Inspect it with `GET /v1/ops/dlq/stats` (admin-only; `?tenant=` names the tenant). -- `false` — the row is left unacked, so NATS redelivers it and it retries until it inserts or the switch is flipped back. For every row the worker **can read**, nothing is ever dropped either way — the choice is *park it* versus *keep retrying*. **One exception, new in this release:** an envelope the worker cannot read *at all* — malformed JSON, an unknown `format` (what a pre-v2 in-flight message looks like), or `columns` and `row` that do not pair — can never insert, so redelivering it forever would wedge the consumer. With the DLQ off for the table it is acked and **dropped**, logged at `ERROR` and counted by `wavehouse_ingest_poison_total` with `disposition="dropped"` (also labeled by `table` and `reason`; an envelope parked on the DLQ carries `disposition="parked"`). See [Ingest Pipeline](/ingest-pipeline) — and drain the ingest queue before upgrading. +- `false` — the row is left unacked, so NATS redelivers it and it retries until it inserts or the switch is flipped back. For every row the worker **can read**, nothing is ever dropped either way — the choice is *park it* versus *keep retrying*. **One exception, new in this release:** an envelope the worker cannot read *at all* — malformed JSON, an unknown `format`, or `columns` and `row` that do not pair — can never insert, so redelivering it forever would wedge the consumer. With the DLQ off for the table it is acked and **dropped**, logged at `ERROR` and counted by `wavehouse_ingest_poison_total` with `disposition="dropped"` (also labeled by `table` and `reason`; an envelope parked on the DLQ carries `disposition="parked"`). See [Ingest Pipeline](/ingest-pipeline) — and drain the ingest queue before upgrading. For a tenant no longer served — its folder removed or rejected — there is no switch to read: its rows are always parked, so none of them sits unacked in its ingest queue, redelivered for as long as the tenant is away and stopping the [Active Sweeper](/ingest-pipeline#the-active-sweeper) purging that queue. @@ -221,7 +221,7 @@ A tenant's dead-letter stream exists from the moment the tenant is first served ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the worker drains it back under the limit — nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to — so a queue fills until its budget or the disk runs out, whichever comes first; size them together from [Durability & Storage](/durability). +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume. A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. ## Streaming diff --git a/internal/ingest/worker.go b/internal/ingest/worker.go index 9af5b1ef..b4defa36 100644 --- a/internal/ingest/worker.go +++ b/internal/ingest/worker.go @@ -117,7 +117,10 @@ const ( // messages are redelivered mid-processing → duplicate inserts). const ( // Server-side cap on a tenant's unacked messages; suspends that tenant's - // delivery when hit (backpressure), and no other tenant's. + // delivery when hit (backpressure), and no other tenant's. The worker holds + // every delivered row until its batch is acked, so while ClickHouse stalls + // it can hold up to maxAckPending rows per tenant: the in-memory bound + // grows with the tenants served. maxAckPending = 10_000 // TODO: raise if NATS delivery becomes the bottleneck // Client prefetch buffer in front of msgChan (was the implicit jetstream @@ -471,13 +474,12 @@ func firstDuplicate(cols []string) (string, bool) { } // parseMsg unmarshals one envelope into a parsedMsg. An envelope the worker can -// never insert is poison — malformed JSON, a row format it doesn't know (which -// is what a pre-v2 envelope looks like: it carries no `format` at all), or +// never insert is poison — malformed JSON, a row format it doesn't know, or // columns and a row it can't pair. Poison is parked on the DLQ rather than -// dropped, so an operator who skipped the documented pre-deploy drain finds -// those rows waiting instead of gone; when the DLQ is off for the table it is -// acked-and-dropped with a counted error, because a message that can never -// insert must not redeliver forever. ok is false either way so the caller skips it. +// dropped, so an operator finds those rows waiting instead of gone; when the +// DLQ is off for the table it is acked-and-dropped with a counted error, +// because a message that can never insert must not redeliver forever. ok is +// false either way so the caller skips it. func (w *IngestWorker) parseMsg(ctx context.Context, m *mq.Message) (parsedMsg, bool) { var envelope EventMessage @@ -493,7 +495,7 @@ func (w *IngestWorker) parseMsg(ctx context.Context, m *mq.Message) (parsedMsg, slog.ErrorContext(ctx, "event envelope declares an unknown row format", "format", envelope.Format, "tenant", id, "table", envelope.TableName) w.rejectPoison(ctx, m, id, envelope.TableName, "unknown_format", - fmt.Sprintf("unknown row format %q (a pre-v2 envelope carries none); drain the ingest queue before upgrading", envelope.Format)) + fmt.Sprintf("unknown row format %q", envelope.Format)) return parsedMsg{}, false } if len(envelope.Columns) == 0 || len(envelope.Row) == 0 { @@ -794,10 +796,9 @@ func (w *IngestWorker) rejectPoison(ctx context.Context, m *mq.Message, id tenan if w.dlqEnabled == nil || w.dlqEnabled(id, tableName) { // Backgrounded on ackWg for the same reason handleSuccess backgrounds its // acks: parkOnDLQ does a DLQ publish AND an fsync-bound DoubleAck, - // and parseMsg runs on the dispatchLoop goroutine. The scenario this whole - // change targets is an operator who skipped the drain, where EVERY backlog - // message is poison — done inline that is one publish plus one fsync per - // message in series, with intake stalled behind it. dispatchLoop adds and + // and parseMsg runs on the dispatchLoop goroutine. When a whole backlog + // is poison, done inline that is one publish plus one fsync per message + // in series, with intake stalled behind it. dispatchLoop adds and // waits on the same goroutine, so each Add still happens-before the Wait. w.ackWg.Go(func() { if w.parkOnDLQ(ctx, m, tableName, detail) { diff --git a/internal/ingest/worker_test.go b/internal/ingest/worker_test.go index 7fe9130e..935eae77 100644 --- a/internal/ingest/worker_test.go +++ b/internal/ingest/worker_test.go @@ -1265,10 +1265,10 @@ func v1Envelope(t *testing.T, table string, data map[string]any) []byte { } // TestParseMsg_PoisonEnvelope_ParkedOnDLQ: an envelope the worker can never -// insert — a pre-v2 message left in the queue across an upgrade, malformed +// insert — one of an unknown format (the pre-v2 shape carries none), malformed // JSON, or columns and a row that can't be paired — is preserved on the DLQ -// rather than dropped, so a missed pre-deploy drain costs an operator a replay -// rather than the rows themselves. +// rather than dropped, so it costs an operator a replay rather than the rows +// themselves. func TestParseMsg_PoisonEnvelope_ParkedOnDLQ(t *testing.T) { t.Parallel() tests := []struct { diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index dbe35fa5..e082915c 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -336,8 +336,13 @@ func (e *EmbeddedNATS) apply(ctx context.Context, id tenant.ID, q *tenantQueue, return fmt.Errorf("open ingest stream: %w", err) } q.ingest, q.maxBytes = true, maxBytes + // The joins run on a budget of their own: a queue that opened but no + // consumer holds fails every consumer (fail), so a slow open must not + // leave them no time. + joinCtx, cancelJoin := context.WithTimeout(ctx, resizeTimeout) + defer cancelJoin() for _, f := range e.consumers { - if err := f.join(resizeCtx, id); err != nil { + if err := f.join(joinCtx, id); err != nil { f.fail(fmt.Errorf("tenant %s: %w: join its queue: %w", id, ErrDeliveryEnded, err)) } } @@ -394,7 +399,13 @@ func (e *EmbeddedNATS) applyDLQ(ctx context.Context, id tenant.ID, q *tenantQueu // publish or park that found one of its streams missing. errNoQueue when no // budget has been asked for the tenant yet: a reload can make a tenant // resolvable an instant before its budget arrives. +// +// It runs detached from ctx's cancellation, bounded by its own timeouts: +// ctx is one caller's — an ingest request — while the queue is every +// consumer's, and a client that goes away between the open and the joins +// would leave a queue no consumer holds, which fails the ingest worker. func (e *EmbeddedNATS) reopen(ctx context.Context, id tenant.ID) error { + ctx = context.WithoutCancel(ctx) e.mu.Lock() defer e.mu.Unlock() q := e.queues[id] diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index fe4fd45e..06120cc5 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -713,6 +713,48 @@ func TestEmbeddedNATS_Publish_OpensTheQueueAtTheLastBudget(t *testing.T) { require.ErrorIs(t, err, jetstream.ErrStreamNotFound) } +// The context a publish reopens a queue under is one client's request, but +// the queue is every consumer's: a client gone before the consumers join must +// not leave a queue that no consumer holds, which the ingest worker would +// report as its delivery ending. So the reopen — joins included — outlives +// the caller's cancellation. +func TestEmbeddedNATS_ReopenOutlivesTheCallersCancellation(t *testing.T) { + e := newTestEmbedded(t, "acme") + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer", MaxAckPending: 10}) + require.NoError(t, err) + for _, name := range []string{"INGEST_acme", "DLQ_acme"} { + require.NoError(t, e.js.DeleteStream(ctx, name)) + } + + gone, stop := context.WithCancel(ctx) + stop() + require.NoError(t, e.reopen(gone, "acme")) + + _, err = e.js.Consumer(ctx, "INGEST_acme", "buffer") + require.NoError(t, err, "the consumer joined the reopened queue") + select { + case err := <-cons.(*workerConsumer).failed: + t.Fatalf("the reopen was reported as the consumer's failure: %v", err) + default: + } + got := make(chan byte, 1) + stopConsume, _, err := cons.Consume(func(msg *Message) { + _ = msg.Ack() + got <- msg.Data[0] + }, 4) + require.NoError(t, err) + t.Cleanup(stopConsume) + require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte{7})) + select { + case b := <-got: + assert.Equal(t, byte(7), b) + case <-time.After(5 * time.Second): + t.Fatal("the reopened queue is not delivered") + } +} + func TestEmbeddedNATS_PurgeAcked(t *testing.T) { e := newTestEmbedded(t) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) diff --git a/internal/settings/settings.go b/internal/settings/settings.go index 55ec089d..2ce9b120 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -225,9 +225,10 @@ type MQConfig struct { // tenth of it). Must be >= 1. A reload updates the live queues in place: // growing takes effect immediately; shrinking below what is currently // buffered makes the ingest queue refuse new publishes (DiscardNew → 503 - // backpressure) until the worker drains it — nothing already buffered is - // dropped — and a dead-letter queue holding more than a tenth of the new - // budget keeps what it holds rather than dropping its oldest rows. + // backpressure) until the sweeper purges it back under the limit — + // nothing already buffered is dropped — and a dead-letter queue holding + // more than a tenth of the new budget keeps what it holds rather than + // dropping its oldest rows. MaxBytesGB *int `json:"max_bytes_gb"` } From 650a28e20a4f3ab9879e021341cbfa4c2b9b7bd3 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 18:57:43 -0400 Subject: [PATCH 03/15] fix(app): boot opens queues under New's context; docs review fixes --- AGENTS.md | 2 +- docs/src/content/docs/api.md | 6 +-- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/deployment.md | 10 ++--- docs/src/content/docs/ingest-pipeline.md | 2 +- docs/src/content/docs/sdk/streaming.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- internal/app/app.go | 5 ++- internal/app/app_test.go | 44 ++++++++++++++++++++ internal/app/wire.go | 26 +++++++----- internal/mq/embedded.go | 7 +++- 11 files changed, 81 insertions(+), 27 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 16595721..dbf19e84 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -56,7 +56,7 @@ The invariant index — what must stay true. Full narrative and rationale live i 3. **Schema-driven ingest** — `POST /v1/ingest?table={table}` takes flat JSON, validated against the discovered schema (unknown fields rejected, types/nullability enforced). No envelope. The **declared `Content-Type` chooses the format and the bytes never do** (arity within the JSON family is still the body's): no declaration, one whose **media type** is unsupported or unparseable, a comma-bearing value that, as a whole, does not parse as one media type, or repeated lines that **disagree**, is a `415` decided *before* the body is read. A malformed *parameter* on a comma-free line never costs the request (`; charset=a; charset=b` still reads as its media type), and repeated lines are accepted only when they all resolve to the same **supported** format — two agreeing `text/csv` lines are still a `415`. A body declared NDJSON stays NDJSON whatever its bytes, so a bad line is a per-record error rather than a silent re-framing; the reverse (NDJSON sent as `application/json`) is deliberately **not** caught — record one, `200`, the rest ignored ([#561](https://github.com/Wave-RF/WaveHouse/issues/561)). Fail-closed — preserve it when touching `internal/api`. 4. **Async ingestion** — ingest returns 200 after optional dedup + MQ publish; ClickHouse writes happen later via `StartIngestWorker`. NATS full → 503 + Retry-After. 5. **Per-tenant-table batching** — the worker groups events by tenant table (the tenant read off each message's `mq.Topic`), so one INSERT never mixes tenants and a batch invalidates its own tenant's cache namespaces; then it splits each batch by column list (`groupByColumns`), emitting one `INSERT INTO … (cols) FORMAT JSONCompactEachRow` per distinct list so a schema change mid-stream can't corrupt a statement. Each tenant table's batch is independent. -6. **Dead Letter Queue** — failed batch inserts publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format` — a pre-v2 envelope carries none — or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. +6. **Dead Letter Queue** — failed batch inserts publish to the tenant's own dead-letter queue (`dlq..
`), gated per table by the tenant's `dlq.enabled` in the settings directory's `config.json` (hot-reloadable; off = leave the row unacked for redelivery). No silent data loss on the insert path. The one drop is an envelope the worker cannot READ (malformed JSON, an unknown **or absent** `format`, or columns and row that don't pair): it is poison by construction, so with the DLQ off it is acked-and-dropped rather than redelivered forever — logged at `ERROR` and counted by `wavehouse_ingest_poison_total` under `disposition="dropped"`. With the DLQ on it is parked like any other failure, and counted under `disposition="parked"`. 7. **Auth: always on, fail-loud, decoupled from authz (security)** — the JWT middleware always runs (no `auth.enabled`/`dev_mode` flag); it verifies with HMAC **or** JWKS (not both), with accepted `alg` pinned to the active verifier and checked before any key is used (rejects `alg:none` and cross-family confusion). No/invalid/expired token → empty role → policy `default_role`, with the bad-token reason stashed so a denying gate returns a loud `401`, not a bare `403`; the one token outcome that never reaches `default_role` is a verifier still fetching its JWKS (`auth.ErrVerifierPending` → `503` + `Retry-After`, `api.refuseUnverifiable`). Elevated access needs a valid granted role. **Sanctioned exception:** a configured non-JWT operator key (`auth.operator_key`; presented via `Authorization: Operator ` or the `X-Operator-Key` alias) deliberately couples authN+authZ — a constant-time match authorizes a full-access platform operator (stamps the admin role plus an operator bit) independent of the verifier (see #11). Detail: architecture.md § `api/` + `internal/auth`; see also #11, §Security Considerations. 8. **Optional dedup, per tenant** — opt-in via `dedupe.enabled` in the settings directory's `config.json` (hot-reloadable: a reload opens or closes that tenant's store via `dedupe.Managed`, one per tenant in `dedupe.Stores`, each a share of the one Pebble instance whose keys lead with the tenant; the ingest handler picks the store off the request's `settings.Store`, so one tenant's seen ids are never another's); `dedupe.id_field` there selects the JSON key, overridable per table. 9. **Singleflight** — the cached read handlers coalesce concurrent misses (`x/sync/singleflight`) under the tenant-led cache key to prevent cache stampede, per tenant. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index ad712b82..9b675b4c 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -612,7 +612,7 @@ Opens a persistent SSE connection for real-time event streaming. Supports histor | ------ | ----------- | | `Last-Event-ID` | RFC 3339 timestamp of the last received event. If present, overrides the `since` query parameter for automatic reconnection (standard `EventSource` behavior). | -**Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back, while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. +**Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back — with a hole where the tenant's history was, which the sweeper purges within a minute of the tenant no longer being served — while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. **Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event, the row slot is that column as published (`null` where the record omitted it, which ClickHouse replaces with the column's default on insert). @@ -858,7 +858,7 @@ The message format used on NATS JetStream between ingest and the batch consumer: | `columns` | string[] | The table's **insertable** column names, in declaration order — what each position in `row` means. A `MATERIALIZED` or `ALIAS` column is computed by ClickHouse and cannot be named in an `INSERT`, so it never appears here. | | `row` | array | One `JSONCompactEachRow` line: one value per entry in `columns`, in that order. A column the request body omitted is `null` here; for a **non-nullable** column the insert turns that back into the column's default (`input_format_null_as_default`), but a `Nullable(T) DEFAULT …` column stores `NULL` — only an *absent* key ever took the default, and a positional row has one slot per column and no way to express absence. Parseable `DateTime`/`DateTime64` values are rewritten to canonical RFC 3339 UTC (see [timestamp canonicalization](#timestamp-canonicalization)); other values as originally sent. | -`columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize, which is what a pre-v2 message looks like. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). +`columns` and `row` are only meaningful together: a reader that cannot pair them — a length mismatch, an undecodable row, a `columns` list naming one column twice — has no way to map a value to a column. Both readers also refuse an envelope whose `format` they do not recognize. Either way the SSE fan-out withholds such an envelope rather than guess, and the batch consumer parks it on the DLQ with `X-DLQ-*` headers — acking and dropping it only where the DLQ is switched off for that table, since it can never insert on retry. Both outcomes increment `wavehouse_ingest_poison_total`, separated by its `disposition` label (`parked` / `dropped`). ### Client-Facing Format (SSE) @@ -876,7 +876,7 @@ Three values, where the envelope above has four: this is the frame a role restri ## Dead Letter Queue (DLQ) -When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: a pre-v2 `data` object, malformed JSON, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. +When a batch insert to ClickHouse fails (e.g., type errors, connection issues), the worker re-inserts the batch row by row: rows that succeed are acked, and only the rows that fail again are published to the tenant's own DLQ NATS stream (`DLQ_{tenant}`) under subjects `dlq.{tenant}.{table}` (the tenant the row was ingested under; `0` for a settings directory that holds the four files). This prevents infinite retry loops — those messages are ACKed from the main stream and moved to the DLQ for inspection. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and is parked whole; only a served tenant whose DLQ is off for the table leaves it for redelivery, since a tenant no longer served has no switch to read. A second class lands here too: an envelope the worker cannot *read* at all — malformed JSON, an unknown **or absent** `format`, or `columns` and `row` that do not pair — is parked without ever reaching a table batch. **Two different body shapes land here, and a consumer must not assume one decoder.** A row that failed its INSERT is parked as the `EventMessage` envelope above. An envelope the worker could not *read* is parked as **its original bytes, verbatim** — `parkOnDLQ` republishes what arrived — so it is whatever the producer sent: malformed JSON, an envelope of an unknown `format`, or a v2 envelope whose `columns` and `row` do not pair. Being undecodable as an `EventMessage` is precisely why it was parked, so decode defensively and fall back on the `X-DLQ-Error` header, which names the reason. For the first shape the body is the published `EventMessage` envelope (`{"table_name":…,"scope":"","received_timestamp":…,"format":…,"columns":[…],"row":[…]}` — the failed row is the `row` array, read against `columns`, its `DateTime`/`DateTime64` values as published: canonicalized where WaveHouse could parse them, otherwise the producer's original spelling — see [timestamp canonicalization](#timestamp-canonicalization)); the failure reason, table, and time travel in the `X-DLQ-Table` / `X-DLQ-Error` / `X-DLQ-Timestamp` message headers. Use `GET /v1/ops/dlq/stats` to monitor DLQ depth, per tenant (`?tenant=`). diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 92508dac..511a2311 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -148,7 +148,7 @@ The **only** package that imports NATS/JetStream — a `depguard` rule in `.gola - **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. - **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts the sweep. It runs on each tenant's stream at that tenant's cutoff. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. -- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, or is refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds, plus five more for the rollback (a budget of its own, not the one that just expired), since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. +- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, or is refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. ### `observability/` — OpenTelemetry Pipeline diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index b2bac204..996d165f 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -379,7 +379,7 @@ settings/ └── roles.json ``` -That is the layout a control plane writes. Each folder's `clickhouse` block is its tenant's own ClickHouse, so a tenant answers queries once its first schema discovery against that ClickHouse succeeds (until then its schema-aware routes answer `503`, `schema not loaded yet`); what tenant `0`'s folder still supplies for the whole process — the token verifier of the routes that name no tenant, their CORS list — is listed under "What a tenant's folder decides", below. +That is the layout a control plane writes. Each folder's `clickhouse` block is its tenant's own ClickHouse, so a tenant answers queries once its first schema discovery against that ClickHouse succeeds (until then its schema-aware routes answer `503`, `schema not loaded yet`); what tenant `0`'s folder still supplies for the whole process — the token verifier of the routes that name no tenant, their CORS list — is listed under "What a lost tenant `0` costs", below. The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe store needs no restructuring: it keys every tenant's seen ids by tenant, and the four files are tenant `0`, as a `0` folder is. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id is a finding of its own — that folder is skipped, and the rest of the directory still loads. @@ -389,7 +389,7 @@ The folder name is the tenant id, and each folder is a complete settings directo **The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` takes it too, and reads a rejected or removed tenant's dead-letter queue like a served one's, since the queue is kept; a tenant that has none is a `404`. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. +**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. **What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. Tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. @@ -423,7 +423,7 @@ The NATS envelope changed shape in this release: the row now travels positionall This affects the streaming surface too, and more quietly. SSE gap-fill (`?since=` / `Last-Event-ID`) replays from the queue, and the upgrade deletes the old one (next paragraph) — the acknowledged history kept for replay included — so this outlives a *correct* drain: any replay spanning the upgrade silently omits the pre-upgrade events, with **no error and no frame**. Clients that need them should backfill over REST. -**The upgrade does not carry the old queue over at all.** Boot deletes the earlier build's queue and dead-letter queue (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) and everything in them, logging a `WARN` with each one's message count: an event the old build had not yet inserted, and a row it had already parked, do not survive the upgrade. Draining first is the only way to keep them, and anything parked has to be replayed before the upgrade — re-ingest each parked envelope's inner `data` object as a fresh `POST /v1/ingest`. +**The upgrade does not carry the old queue over at all.** Boot deletes the earlier build's queue and dead-letter queue (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) and everything in them, logging a `WARN` with each one's message count: an event the old build had not yet inserted, and a row it had already parked, do not survive the upgrade. Draining first keeps the events not yet inserted; a row already parked is lost with the queue, since the earlier build offers no way to read one back (`GET /v1/ops/dlq/stats` returns counts only). Three audits belong **before** the drain, because none of them announces itself afterwards: @@ -435,10 +435,10 @@ To drain before upgrading: 1. **Stop the producers**, or cut `/v1/ingest` at the reverse proxy. Nothing new should enter the stream. 2. **Wait for the in-flight batches to flush.** A table's batch closes on size or after `maxWait` (5s by default), so a few seconds after the last write is enough; give it longer if ClickHouse is slow or retrying. -3. **Confirm nothing is left unconsumed** before swapping binaries. Not that the stream is empty: it is dual-use, and deliberately retains ACKed messages for the SSE replay window, so a non-zero depth right after a clean drain is expected. Rows landing in ClickHouse is a success signal, **not proof the queue is drained** — when the DLQ is off for a table, a row that fails its retry is skipped without being acked, so NATS keeps redelivering it while its neighbors land. Check that nothing is still failing or redelivering, and note which signal covers which case: [`GET /v1/ops/dlq/stats`](/api#get-v1opsdlqstats--dlq-statistics) is non-zero only where the DLQ is **on**; `wavehouse_ingest_poison_total` counts unreadable envelopes on **either** setting, told apart by its `disposition` label (`parked` / `dropped`); and for a twice-failed row with the DLQ off — the case just described — the **only** signal is the `ERROR` log (`isolated bad row, DLQ disabled for table`). A clean `dlq/stats` with the DLQ off proves nothing. There is no queue-depth gauge today ([#544](https://github.com/Wave-RF/WaveHouse/issues/544) tracks the related in-flight accounting), and `wavehouse_nats_in_msgs_total` going flat is a supporting signal rather than a guarantee. Enabling the DLQ is not itself a drain — replay from `dlq.{table}` is manual, and has to happen before the upgrade deletes it. +3. **Confirm nothing is left unconsumed** before swapping binaries. Not that the stream is empty: it is dual-use, and deliberately retains ACKed messages for the SSE replay window, so a non-zero depth right after a clean drain is expected. Rows landing in ClickHouse is a success signal, **not proof the queue is drained** — when the DLQ is off for a table, a row that fails its retry is skipped without being acked, so NATS keeps redelivering it while its neighbors land. Check that nothing is still failing or redelivering, and note which signal covers which case: [`GET /v1/ops/dlq/stats`](/api#get-v1opsdlqstats--dlq-statistics) is non-zero only where the DLQ is **on**; `wavehouse_ingest_poison_total` counts unreadable envelopes on **either** setting, told apart by its `disposition` label (`parked` / `dropped`); and for a twice-failed row with the DLQ off — the case just described — the **only** signal is the `ERROR` log (`isolated bad row, DLQ disabled for table`). A clean `dlq/stats` with the DLQ off proves nothing. There is no queue-depth gauge today ([#544](https://github.com/Wave-RF/WaveHouse/issues/544) tracks the related in-flight accounting), and `wavehouse_nats_in_msgs_total` going flat is a supporting signal rather than a guarantee. Enabling the DLQ is not itself a drain: a parked row is not inserted, and the upgrade deletes it. 4. **Upgrade**, then re-enable ingest. -If you skipped the drain, the boot's `WARN` line for each deleted stream (`deleted the stream an earlier build kept for every tenant together`) says how many messages went with it. +If you skipped the drain, the boot's `WARN` line for each deleted stream (`deleted the stream an earlier build kept for every tenant together`) says how many messages went with it: for `WAVEHOUSE_DLQ`, the parked rows lost; for `WAVEHOUSE`, a count that includes the acknowledged history kept for replay, already in ClickHouse — so it bounds the events lost rather than counting them, and is non-zero even after a clean drain. ## Dead Letter Queue (DLQ) diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index b504fef4..72447024 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -22,7 +22,7 @@ The pipeline is **insert-only**. (Upgrading across the v2 envelope? [Drain the q ## High-level shape -Each tenant's events are queued on a JetStream stream of its own. One process holds one durable consumer on each tenant's stream, delivered into one handler, and fans events out to a goroutine per tenant table — the tenant is the subject's leading token. Each tenant's table batches independently and POSTs to ClickHouse over the HTTP interface (`JSONCompactEachRow`). On a bulk-insert failure the batch is re-inserted row by row, so a single poison row can't sink it: clean rows ack, and only the rows that fail again go to the dead-letter stream. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and meets the dead-letter switch once, whole; a tenant no longer served has no switch to read, so its batch is parked. An envelope the worker cannot *read* — malformed JSON, an unknown row `format` (what a pre-v2 message looks like), or columns and a row that don't pair — never reaches a table loop at all: `parseMsg` parks it on the same dead-letter stream, or, where the DLQ is off for the table, acks and drops it rather than redelivering a message that can never insert. A separate sweeper reclaims stream storage. +Each tenant's events are queued on a JetStream stream of its own. One process holds one durable consumer on each tenant's stream, delivered into one handler, and fans events out to a goroutine per tenant table — the tenant is the subject's leading token. Each tenant's table batches independently and POSTs to ClickHouse over the HTTP interface (`JSONCompactEachRow`). On a bulk-insert failure the batch is re-inserted row by row, so a single poison row can't sink it: clean rows ack, and only the rows that fail again go to the dead-letter stream. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips that retry, which no row of it could pass, and meets the dead-letter switch once, whole; a tenant no longer served has no switch to read, so its batch is parked. An envelope the worker cannot *read* — malformed JSON, an unknown row `format`, or columns and a row that don't pair — never reaches a table loop at all: `parseMsg` parks it on the same dead-letter stream, or, where the DLQ is off for the table, acks and drops it rather than redelivering a message that can never insert. A separate sweeper reclaims stream storage. ```mermaid flowchart LR diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 94a7e334..2b37667a 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -116,7 +116,7 @@ A dropped stream reconnects on a jittered exponential backoff, capped at 30s, an :::caution[Resumption is at-least-once, and time-bounded] Delivery across a reconnect is **at-least-once**. The `Last-Event-ID` the client sends is the last event's `received_timestamp`, and the server replays from that instant *inclusively* — so the last event you already saw, and anything sharing its timestamp, arrives again. The SDK does not deduplicate live frames — `liveQuery()` makes one pass at the backfill seam, and only under an ascending order ([#449](https://github.com/Wave-RF/WaveHouse/issues/449)) — so key on `timestamp` plus your own row identity if duplicates matter. -Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. The same silence applies across a server upgrade to this release: the server deletes the previous release's queue at boot, so a replay spanning the upgrade omits the events published before it, without an error — backfill over REST if you need them. +Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. So does a stream a [nested server](/deployment#the-nested-settings-directory) ended because its tenant's folder was rejected, once the folder is fixed: the sweeper purges a tenant's history within a minute of the tenant no longer being served. The same silence applies across a server upgrade to this release: the server deletes the previous release's queue at boot, so a replay spanning the upgrade omits the events published before it, without an error — backfill over REST if you need them. **A column-set change across a gap-fill is a known limitation.** If the table's columns change while you are connected *and* your client replays across that change, live rows arriving after the replay may not be preceded by a fresh `event: schema` frame until the columns next change or you reconnect. The SDK drops a row whose **length** disagrees with the list it was last told, rather than zipping it under the wrong names — so an added or removed column costs you rows, not wrong ones. A **same-length** change is the residual case the arity check cannot see: a `RENAME COLUMN`, or a drop paired with an add, zips values under the wrong names until the next announcement. Reconnecting resynchronizes either way. Full schema-change handling is deferred to the schema-versioning work ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). ::: diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 41e04187..29f842bc 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -213,7 +213,7 @@ The `auth` block is the verifier wiring, minus the secrets. `jwks_url` (absolute A failed batch insert is retried row by row; a row that fails again on its own is a poison row. A batch whose tenant has no ClickHouse connection — one no longer served, or one no pool could be opened for (such as by the connection ceiling) — skips the retry, which no row of it could pass, and every row of it is a poison row. `dlq.enabled` (seed default `true`) decides what happens to it, resolved per table (`dlq.tables.
.enabled` → global) at the moment of the failure, so a reload applies to the next poison row: - `true` — the row is published to the tenant's dead-letter stream (`DLQ_{tenant}`) under `dlq.{tenant}.{table}` (`0` for a directory that holds the four files) with the failure in its headers, and its original is acked. Inspect it with `GET /v1/ops/dlq/stats` (admin-only; `?tenant=` names the tenant). -- `false` — the row is left unacked, so NATS redelivers it and it retries until it inserts or the switch is flipped back. For every row the worker **can read**, nothing is ever dropped either way — the choice is *park it* versus *keep retrying*. **One exception, new in this release:** an envelope the worker cannot read *at all* — malformed JSON, an unknown `format`, or `columns` and `row` that do not pair — can never insert, so redelivering it forever would wedge the consumer. With the DLQ off for the table it is acked and **dropped**, logged at `ERROR` and counted by `wavehouse_ingest_poison_total` with `disposition="dropped"` (also labeled by `table` and `reason`; an envelope parked on the DLQ carries `disposition="parked"`). See [Ingest Pipeline](/ingest-pipeline) — and drain the ingest queue before upgrading. +- `false` — the row is left unacked, so NATS redelivers it and it retries until it inserts or the switch is flipped back. For every row the worker **can read**, nothing is ever dropped either way — the choice is *park it* versus *keep retrying*. **One exception, new in this release:** an envelope the worker cannot read *at all* — malformed JSON, an unknown `format`, or `columns` and `row` that do not pair — can never insert, so redelivering it forever would wedge the consumer. With the DLQ off for the table it is acked and **dropped**, logged at `ERROR` and counted by `wavehouse_ingest_poison_total` with `disposition="dropped"` (also labeled by `table` and `reason`; an envelope parked on the DLQ carries `disposition="parked"`). See [Ingest Pipeline](/ingest-pipeline). For a tenant no longer served — its folder removed or rejected — there is no switch to read: its rows are always parked, so none of them sits unacked in its ingest queue, redelivered for as long as the tenant is away and stopping the [Active Sweeper](/ingest-pipeline#the-active-sweeper) purging that queue. diff --git a/internal/app/app.go b/internal/app/app.go index dc15bf5d..6d39d0fe 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -144,7 +144,8 @@ const ( ) // New wires every component. ctx bounds construction only — the boot-time -// schema refresh and the JetStream stream setup; the loops start in Run. A +// schema refresh and the opening of each served tenant's queue; the loops +// start in Run. A // failure releases whatever was already opened and returns the error, so // the caller never holds a half-built App. func New(ctx context.Context, opts Options) (app *App, err error) { @@ -177,7 +178,7 @@ func New(ctx context.Context, opts Options) (app *App, err error) { if err := a.wireDedupe(); err != nil { return nil, err } - if err := a.wireMQ(); err != nil { + if err := a.wireMQ(ctx); err != nil { return nil, err } if err := a.wireCache(); err != nil { diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 2ecd77d9..cda01e6f 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -571,6 +571,50 @@ func TestNew_DedupeOpenFailure(t *testing.T) { }) } +// A tenant's queue the MQ cannot open follows the registry's rule for the +// shape, as the dedupe store does: a flat directory refuses boot, and a nested +// one boots with that tenant's queue closed and every other tenant's open. +// The obstacle is a regular file where the embedded server keeps a stream's +// store — the embedded implementation's layout, which this test takes on to +// force the failure, as TestNew_DedupeOpenFailure does Pebble's. The failed +// open clears it, so the next publish opens the queue: each one tries again. +func TestNew_QueueOpenFailure(t *testing.T) { + block := func(t *testing.T, dataDir, stream string) { + t.Helper() + p := filepath.Join(dataDir, "nats", "jetstream", "$G", "streams", stream) + require.NoError(t, os.MkdirAll(filepath.Dir(p), 0o750)) + require.NoError(t, os.WriteFile(p, nil, 0o600)) + } + t.Run("flat refuses boot", func(t *testing.T) { + guardGlobals(t) + cfg := testConfig(t, writeSettings(t, nil)) + block(t, cfg.DataDir, "DLQ_0") + _, err := New(t.Context(), Options{Config: cfg}) + require.ErrorContains(t, err, "mq open") + }) + t.Run("nested costs the tenant alone", func(t *testing.T) { + cfg := testConfig(t, writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil})) + block(t, cfg.DataDir, "DLQ_acme") + a := newApp(t, cfg, Options{}) + assert.Zero(t, a.mq.MaxBytes("acme"), "acme's queue did not open") + assert.Equal(t, int64(50<<30), a.mq.MaxBytes("globex"), "and costs globex nothing") + + require.NoError(t, a.MQ().Publish(t.Context(), mq.Topic{Tenant: "acme", Table: "t"}, []byte("x"))) + assert.Equal(t, int64(50<<30), a.mq.MaxBytes("acme"), "a publish opened it at acme's budget") + }) +} + +// Boot opens each served tenant's queue under New's context, as New's doc +// says: a stop signalled during boot is not held up by one open per tenant. +func TestNew_QueueSetupHonorsTheBootContext(t *testing.T) { + guardGlobals(t) + ctx, cancel := context.WithCancel(t.Context()) + cancel() + _, err := New(ctx, Options{Config: testConfig(t, writeSettings(t, nil))}) + require.ErrorIs(t, err, context.Canceled) + require.ErrorContains(t, err, "mq open") +} + // The tenants on the writer's ClickHouse address and database read the same // tables, so an insert invalidates a table's cached results under every one // of them — whatever their user, so across pools — and under no tenant on diff --git a/internal/app/wire.go b/internal/app/wire.go index 60cbdcd8..565e6bc2 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -541,9 +541,11 @@ func (a *App) wireDedupe() error { // resized follows the registry's rule for the shape: a flat directory // refuses boot, like every other store, and on a reload logs it, keeping the // previous budget; a nested directory logs it at boot too, so it never costs -// the process — the tenant's ingest answers 503 until a reload opens its -// queue. The hook is registered before the boot apply, as the dedupe one is. -func (a *App) wireMQ() error { +// the process — the tenant's ingest answers 503 until its queue opens, each +// publish and each reload trying again. The hook is registered before the +// boot apply, as the dedupe one is. The boot apply runs on ctx, New's, so a +// stop signalled during a boot that opens many queues is not held up by them. +func (a *App) wireMQ(ctx context.Context) error { dir := filepath.Join(a.cfg.DataDir, "nats") config.WarnIfFreshDataDir("nats", dir) var broker mq.Broker @@ -565,17 +567,21 @@ func (a *App) wireMQ() error { } } - // Rooted in the App's stop context, so a reload caught mid-hook by - // SIGTERM gives up rather than holding the drain past - // server.shutdown_timeout. - reconcile := func() error { + // The hook's apply is rooted in the App's stop context, so a reload + // caught mid-hook by SIGTERM gives up rather than holding the drain past + // server.shutdown_timeout; a done ctx ends the pass over the tenants. + reconcile := func(ctx context.Context) error { var errs []error for id, store := range a.tenants.All() { + if err := ctx.Err(); err != nil { + errs = append(errs, err) + break + } mb := store.MQMaxBytes() if mb == broker.MaxBytes(id) { continue } - if err := broker.SetMaxBytes(a.stopCtx, id, mb); err != nil { + if err := broker.SetMaxBytes(ctx, id, mb); err != nil { slog.Error("mq queue not reconciled with settings; the next reload retries", "tenant", id, "error", err) errs = append(errs, fmt.Errorf("tenant %s: %w", id, err)) continue @@ -584,8 +590,8 @@ func (a *App) wireMQ() error { } return errors.Join(errs...) } - a.tenants.AfterAdopt(func([]tenant.ID) { _ = reconcile() }) - if err := reconcile(); err != nil && !a.tenants.Nested() { + a.tenants.AfterAdopt(func([]tenant.ID) { _ = reconcile(a.stopCtx) }) + if err := reconcile(ctx); err != nil && !a.tenants.Nested() { return fmt.Errorf("mq open: %w", err) } return nil diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index e082915c..8be913c5 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -96,7 +96,9 @@ const ( // fails: in-process JetStream fails by stalling rather than erroring, so // the likely cause is that resizeTimeout has just run out, and an undo on // that context would fail without touching the stream. SetMaxBytes runs - // for at most the sum of the two. + // for at most the sum of the two when it resizes, and for two + // resizeTimeouts when it opens a queue: the consumers join on a budget of + // their own (apply). rollbackTimeout = 5 * time.Second ) @@ -303,7 +305,8 @@ func (e *EmbeddedNATS) MaxBytes(id tenant.ID) int64 { // call with the new budget reapplies both. // // The JetStream calls are bounded by resizeTimeout, plus rollbackTimeout for -// the undo, both rooted in ctx. That is deliberate: ctx is the process's stop +// the undo — or another resizeTimeout for the consumers joining a queue just +// opened — all rooted in ctx. That is deliberate: ctx is the process's stop // context, so a reload caught mid-hook by a stop gives up — undo included — // rather than holding the drain past server.shutdown_timeout. A cancellation // between the two updates is therefore the one way to leave the pair split, From 999c1db551d8cd34a63a6ca0d019bcfab7f80709 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 19:29:53 -0400 Subject: [PATCH 04/15] fix(mq): boot re-applies a budget to a split queue pair; review fixes --- docs/src/content/docs/architecture.md | 5 ++- docs/src/content/docs/deployment.md | 4 +- docs/src/content/docs/settings-directory.mdx | 2 +- internal/app/app.go | 5 +-- internal/app/app_test.go | 2 +- internal/app/wire.go | 2 +- internal/ingest/worker.go | 9 ++--- internal/mq/embedded.go | 30 ++++++++++++--- internal/mq/embedded_test.go | 40 ++++++++++++++++++++ internal/mq/mq.go | 24 ++++++------ 10 files changed, 92 insertions(+), 31 deletions(-) diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 511a2311..549622f6 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -90,7 +90,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each served tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot and again after every reload, under the App's stop context, which opens that tenant's queue the first time; a queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each served tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out @@ -231,7 +231,8 @@ Client POST /v1/ingest?table={table} → (If the tenant's NATS stream is full, or not open: 503 + Retry-After header) Ingest worker pipeline (StartIngestWorker): - ← JetStream pull consumer (buffer-consumer) on ingest.> + ← JetStream pull consumer (buffer-consumer), one durable per tenant stream + (ingest.{tenant}.>), delivered into one handler → Parse the event envelope (an envelope the worker cannot read — malformed JSON, an unknown or absent format, columns and row that don't pair — is parked on the DLQ, or acked-and-dropped where the DLQ is off for the table; either way diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 996d165f..50bedaaa 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -419,9 +419,9 @@ WaveHouse discovers this schema on startup and refreshes it every `schema.refres ## Upgrading across the v2 ingest envelope -The NATS envelope changed shape in this release: the row now travels positionally, with `format`, `columns` and `row` replacing `data`. **The new worker cannot read a message published by an older version** — it carries no `format`, so there is no way to say which value belongs to which column. +The NATS envelope changed shape in this release: the row now travels positionally, with `format`, `columns` and `row` replacing `data` — and the queue changed layout with it: boot deletes the earlier build's queue (below), so nothing an older version published reaches the new worker, which could not read it anyway (it carries no `format`, so there is no way to say which value belongs to which column). **Drain first** to keep what the old build had not yet inserted. -This affects the streaming surface too, and more quietly. SSE gap-fill (`?since=` / `Last-Event-ID`) replays from the queue, and the upgrade deletes the old one (next paragraph) — the acknowledged history kept for replay included — so this outlives a *correct* drain: any replay spanning the upgrade silently omits the pre-upgrade events, with **no error and no frame**. Clients that need them should backfill over REST. +The streaming surface loses something too, more quietly. SSE gap-fill (`?since=` / `Last-Event-ID`) replays from the queue, so the deletion takes the replay history with it, even after a *correct* drain: any replay spanning the upgrade silently omits the pre-upgrade events, with **no error and no frame**. Clients that need them should backfill over REST. **The upgrade does not carry the old queue over at all.** Boot deletes the earlier build's queue and dead-letter queue (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) and everything in them, logging a `WARN` with each one's message count: an event the old build had not yet inserted, and a row it had already parked, do not survive the upgrade. Draining first keeps the events not yet inserted; a row already parked is lost with the queue, since the earlier build offers no way to read one back (`GET /v1/ops/dlq/stats` returns counts only). diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 29f842bc..a0808514 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -221,7 +221,7 @@ A tenant's dead-letter stream exists from the moment the tenant is first served ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume. A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume. A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. ## Streaming diff --git a/internal/app/app.go b/internal/app/app.go index 6d39d0fe..2131942c 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -145,9 +145,8 @@ const ( // New wires every component. ctx bounds construction only — the boot-time // schema refresh and the opening of each served tenant's queue; the loops -// start in Run. A -// failure releases whatever was already opened and returns the error, so -// the caller never holds a half-built App. +// start in Run. A failure releases whatever was already opened and returns the +// error, so the caller never holds a half-built App. func New(ctx context.Context, opts Options) (app *App, err error) { a := &App{cfg: opts.Config, build: opts.Build, logLevel: opts.LogLevel, listener: opts.Listener} if a.logLevel == nil { diff --git a/internal/app/app_test.go b/internal/app/app_test.go index cda01e6f..8ef1f7c5 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -605,7 +605,7 @@ func TestNew_QueueOpenFailure(t *testing.T) { } // Boot opens each served tenant's queue under New's context, as New's doc -// says: a stop signalled during boot is not held up by one open per tenant. +// says: a stop signaled during boot is not held up by one open per tenant. func TestNew_QueueSetupHonorsTheBootContext(t *testing.T) { guardGlobals(t) ctx, cancel := context.WithCancel(t.Context()) diff --git a/internal/app/wire.go b/internal/app/wire.go index 565e6bc2..e08f4a0e 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -544,7 +544,7 @@ func (a *App) wireDedupe() error { // the process — the tenant's ingest answers 503 until its queue opens, each // publish and each reload trying again. The hook is registered before the // boot apply, as the dedupe one is. The boot apply runs on ctx, New's, so a -// stop signalled during a boot that opens many queues is not held up by them. +// stop signaled during a boot that opens many queues is not held up by them. func (a *App) wireMQ(ctx context.Context) error { dir := filepath.Join(a.cfg.DataDir, "nats") config.WarnIfFreshDataDir("nats", dir) diff --git a/internal/ingest/worker.go b/internal/ingest/worker.go index b4defa36..f6cbaf6a 100644 --- a/internal/ingest/worker.go +++ b/internal/ingest/worker.go @@ -226,11 +226,10 @@ func waitOrDeadline(ctx context.Context, wg *sync.WaitGroup) error { // dispatchLoop owns the one consumer — held on every tenant's queue — and fans // every message out to a tableLoop per tenant table (lazily spawned on first -// sight of one). It does -// no batching itself — it parses just enough to route — so a low-volume table -// can never strand another table's rows behind a shared timer. It is the ONLY -// goroutine that watches ctx; tableLoops stop via channel-close, which gives a -// deterministic drain with no abandoned messages. +// sight of one). It does no batching itself — it parses just enough to route — +// so a low-volume table can never strand another table's rows behind a shared +// timer. It is the ONLY goroutine that watches ctx; tableLoops stop via +// channel-close, which gives a deterministic drain with no abandoned messages. func (w *IngestWorker) dispatchLoop(ctx context.Context, cons mq.Consumer) { defer w.wg.Done() diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index 8be913c5..b621f96e 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -74,8 +74,9 @@ type tenantQueue struct { ingest, dlq bool // maxBytes is the budget last applied in full (MaxBytes); asked is the // budget last asked for, which a publish or park that finds a stream - // missing opens it at. Both are read back from the ingest stream at boot, - // so a tenant no longer served keeps the budget it last had. + // missing opens it at. Boot reads asked back from the ingest stream, so a + // tenant no longer served keeps the budget it last had, and maxBytes too + // when the pair is whole at it (takeStock). maxBytes, asked int64 } @@ -182,20 +183,39 @@ func (e *EmbeddedNATS) takeStock(ctx context.Context) error { return err } } + type dlqState struct { + limit int64 + held uint64 + } + dlqs := map[tenant.ID]dlqState{} streams := e.js.ListStreams(ctx) for info := range streams.Info() { name := info.Config.Name if id, ok := streamTenant(ingestStreamPrefix, name); ok { q := e.queue(id) q.ingest = true - q.maxBytes, q.asked = info.Config.MaxBytes, info.Config.MaxBytes + q.asked = info.Config.MaxBytes } else if id, ok := streamTenant(dlqStreamPrefix, name); ok { e.queue(id).dlq = true + dlqs[id] = dlqState{limit: info.Config.MaxBytes, held: info.State.Bytes} } } if err := streams.Err(); err != nil { return fmt.Errorf("list streams: %w", err) } + // A pair is at its budget when its dead-letter stream is at a tenth of + // the ingest cap, or above it holding more than that: the shrink guard's + // doing. Anything else is a pair a stop or a failed update left split, or + // one missing its dead-letter stream, so its budget stays unapplied and + // the boot's SetMaxBytes applies it to both streams again. + for id, q := range e.queues { + d, ok := dlqs[id] + tenth := q.asked / dlqShare + guarded := d.limit > tenth && d.held <= math.MaxInt64 && int64(d.held) > tenth + if q.ingest && ok && (d.limit == tenth || guarded) { + q.maxBytes = q.asked + } + } return nil } @@ -668,8 +688,8 @@ func sameConsumer(have, want jetstream.ConsumerConfig) bool { } // share is one tenant's part of the fetch-ahead: the total spread over the -// tenants' queues, at least one each. 0 leaves the client default. Under -// e.mu. +// tenants' queues joined so far, at least one each, fixed when that queue's +// delivery starts. 0 leaves the client default. Under e.mu. func (f *fanIn) share() int { if f.prefetch <= 0 { return 0 diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index 06120cc5..e75b62d7 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -1083,6 +1083,46 @@ func TestNewEmbedded_DeletesTheStreamsAnEarlierBuildShared(t *testing.T) { require.NoError(t, e.Publish(ctx, Topic{Tenant: tenant.Default, Table: "events"}, []byte("x"))) } +// A pair a stop or a failed update left split — its dead-letter stream not +// at a tenth of the ingest cap — or one missing its dead-letter stream is not +// at its budget, so the boot's SetMaxBytes applies the budget to both streams +// again; a dead-letter stream kept above its tenth because it holds more (the +// shrink guard) is at its budget and left as it is. +func TestNewEmbedded_ASplitPairIsAppliedAgainAtBoot(t *testing.T) { + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + dir := t.TempDir() + first, err := NewEmbedded(dir) + require.NoError(t, err) + for _, id := range []tenant.ID{"split", "gone", "guarded"} { + require.NoError(t, first.SetMaxBytes(ctx, id, 10<<20)) + } + _, err = first.js.UpdateStream(ctx, dlqStreamConfig("split", 2<<20)) + require.NoError(t, err) + require.NoError(t, first.js.DeleteStream(ctx, "DLQ_gone")) + payload := make([]byte, 1<<10) + for range 200 { + require.NoError(t, first.DeadLetter(ctx, NewMessage(ctx, Topic{Tenant: "guarded", Table: "t"}, payload, time.Now(), nil, nil, nil))) + } + require.NoError(t, first.SetMaxBytes(ctx, "guarded", 1<<20)) + guardedCap := streamConfig(t, first, "DLQ_guarded").MaxBytes + require.Greater(t, guardedCap, int64(1<<20)/10, "the guard kept the parked rows") + require.NoError(t, first.Close()) + + e := openEmbedded(t, dir) + assert.Zero(t, e.MaxBytes("split"), "a split pair is not at its budget") + assert.Zero(t, e.MaxBytes("gone"), "nor one missing its dead-letter stream") + assert.Equal(t, int64(1<<20), e.MaxBytes("guarded"), "a guarded dead-letter stream is") + + for _, id := range []tenant.ID{"split", "gone"} { + require.NoError(t, e.SetMaxBytes(ctx, id, 10<<20)) + assert.Equal(t, int64(10<<20), e.MaxBytes(id)) + assert.Equal(t, int64(10<<20)/10, streamConfig(t, e, dlqStreamName(id)).MaxBytes, "%s: the pair is whole again", id) + } + require.NoError(t, e.SetMaxBytes(ctx, "guarded", 1<<20)) + assert.Equal(t, guardedCap, streamConfig(t, e, "DLQ_guarded").MaxBytes, "left as the guard kept it") +} + // A boot takes stock of the queues on disk: each keeps the budget it last // had, and a consumer created afterwards is held on every one of them — a // tenant no longer served, which is never given a budget again, included — diff --git a/internal/mq/mq.go b/internal/mq/mq.go index 34620f85..663bde3d 100644 --- a/internal/mq/mq.go +++ b/internal/mq/mq.go @@ -195,17 +195,19 @@ type ConsumerConfig struct { // Consumer is a live durable consumer created by ConsumerManager. type Consumer interface { - // Consume delivers each message to handler on a delivery goroutine of - // its tenant's: one per tenant, so a tenant's messages arrive in order, - // one at a time, while different tenants' arrive concurrently — handler - // must be safe for that. A handler that blocks holds back its tenant's - // delivery — that is the backpressure the ingest worker relies on. About - // prefetch messages are fetched ahead across the tenants together, at - // least one per tenant (0 = the client default, per tenant). The returned - // stop asks delivery to end and returns without waiting: a handler - // invocation already in flight, or one for a message already queued - // client-side, may still run after stop returns, so a handler must not - // write to anything the caller tears down right after stopping. + // Consume delivers each message to handler on a delivery goroutine of its + // tenant's: one per tenant, so a tenant's messages arrive in order, one at + // a time, while different tenants' arrive concurrently — handler must be + // safe for that. A handler that blocks holds back its tenant's delivery — + // that is the backpressure the ingest worker relies on. About prefetch + // messages are fetched ahead across the tenants together: the tenants' + // queues when delivery starts split it, and a queue joined later fetches + // ahead its share of it at that point, at least one message each (0 = the + // client default, per tenant). The returned stop asks delivery to end and + // returns without waiting: a handler invocation already in flight, or one + // for a message already queued client-side, may still run after stop + // returns, so a handler must not write to anything the caller tears down + // right after stopping. // // Delivery can also end on its own after Consume has returned: the broker // or the client gives up on the consumer (it was deleted, the connection From db2d20a20a4f3af04ac671d1fcc2e7387215cd4b Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 20:01:33 -0400 Subject: [PATCH 05/15] fix(mq): a failed resize restores the ingest stream's own cap; review fixes --- CHANGELOG.md | 2 +- docs/src/content/docs/deployment.md | 6 ++-- docs/src/content/docs/durability.md | 6 ++-- internal/mq/embedded.go | 21 ++++++++----- internal/mq/embedded_test.go | 49 +++++++++++++++++++++++++++++ 5 files changed, 70 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 23c0c715..86dac970 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -24,7 +24,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), - **Schema discovery captures each table's DDL, its columns' ordinals and default expressions, and the server version** (`internal/discovery/discovery.go`, `internal/testutil/testutil.go`): `Column` gains `DefaultExpression` and `Position` (both from a widened `system.columns` select), `TableSchema` gains `DDL` from `system.tables.create_table_query`, and `SchemaRegistry` gains `ServerVersion()` from a `SELECT version()` probe next to the existing `SELECT timezone()`. Groundwork for the native type layer, captured on the same refresh as the columns so a stale version cannot outlive the schemas it describes. That is a publication guarantee, not a same-server one: `chconn.Manager` resolves the connection per call, so a reload changing `clickhouse.addr` mid-refresh can still pair a version from one server with schemas from another — narrow, and self-correcting on the next refresh. `DDL` is `json:"-"` and does **not** appear in `/v1/ops/schema`: that endpoint marshals `TableSchema` straight to the client, and an external-engine table (S3, MySQL, PostgreSQL, Kafka) renders its wiring there unconditionally — endpoint, bucket or host, database, username, S3 access key id. ClickHouse masks the password itself as `[HIDDEN]` from ~23.9 (verified on 26.7.3), so the exposure is the topology rather than the secret — except on an older server, or one with `display_secrets_in_show_and_select` enabled. `position` and `default_expression` are additive fields in the response. A table listed in `system.tables` with no `system.columns` rows is skipped rather than published column-less, and both new queries fail the refresh on error exactly as `timezone()` and `system.columns` do — callers keep the prior cache and retry. -- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the worker drains, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** — every `config.json` key is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. +- **Settings-directory hot reload — boot loading, three reload triggers, and the config-key migration** (`internal/settings/` (new: `store.go`, `watch.go`, + tests), `internal/api/settings.go` (new, + tests), `internal/api/{router,ingest,structured_query}.go`, `internal/discovery/discovery.go`, `internal/config/config.go`, `cmd/wavehouse/main.go`, `config.yaml`, `deployments/compose/standalone.yaml`, `docs/src/content/docs/settings-directory.mdx` (new — the hot-reloadable half of configuration gets its own page; `configuration.mdx` is boot config only); closes the loop [#500](https://github.com/Wave-RF/WaveHouse/pull/500) opened, tracked by [#48](https://github.com/Wave-RF/WaveHouse/issues/48)): the server now *consumes* the settings directory instead of only validating it. `settings.Store` owns the adopted snapshot: `settings.dir` / `WH_SETTINGS_DIR` is now **required**, boot validates and adopts the directory (missing or invalid refuses to start); a running instance then re-validates and re-adopts on any of three triggers — a **directory watch** (fsnotify on the directory, not the files, so atomic-writer replaces and Kubernetes ConfigMap symlink swaps aren't lost; bursts debounce into one reload), **`SIGHUP`**, and **`POST /v1/ops/settings/reload`** (admin-gated; returns `{"adopted", "findings"}`, `200` adopted / `422` rejected) — all funneling through one serialized reload path. A reload that fails validation keeps the previous good snapshot (an operator mid-edit degrades to a log line, never a broken server); warnings don't block adoption, matching `wavehouse validate`. The tenant tunables **migrate out of boot config** into the directory's `config.json`: `dedupe.id_field` / `dedupe.require_id` (now with the per-table overrides under `dedupe.tables` that [#222](https://github.com/Wave-RF/WaveHouse/issues/222) asked for, resolved per record through the table → global cascade in one atomic snapshot read, so a reload lands at a record boundary and never mixes documents within one record), `query.default_max_rows` and `query.timestamp_bucket_seconds` (read per query), `schema.refresh_interval` (re-read after each tick, so a change applies from the next cycle), `stream.keepalive_interval` / `stream.keepalive_buckets` (a reload calls the new `Heartbeater.Reconfigure`, which rebuilds the keepalive wheel in place with every live subscriber carried over and re-times the running ticker) and `stream.gap_window_minutes` (the sweeper re-reads it every sweep), `mq.max_bytes_gb` (an after-adopt hook updates the tenant's ingest and dead-letter stream limits in place via `mq.Broker.SetMaxBytes` — shrinking below the buffered size backpressures until the sweeper purges it back under the limit, nothing is dropped), `dlq.enabled` with per-table overrides under `dlq.tables` (resolved by the ingest worker at the moment a poison row is isolated: on → park it on the tenant's dead-letter stream and ack; off → leave it unacked for redelivery, never dropped; a served tenant's DLQ stream and `GET /v1/ops/dlq/stats` always exist, so the switch is purely behavioral), the **ClickHouse wiring** (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout`: the new `chconn.Manager` is the one `driver.Conn` every consumer holds and swaps the connection behind it on reload — unconditionally, since the adopted settings are the authority and reachability already surfaces through schema discovery and `/readyz`; the replaced one closes after a `query_timeout` grace; the ingest worker, raw-SQL proxy, and schema registry read the HTTP target, timeout, and database per call), the **auth verifier wiring** (`auth.jwks_url` / `auth.role_claim`: the new `auth.Authenticator` swaps a whole verifier — key source plus its pinned algorithm allowlist — atomically per reload, unconditionally, so an unreachable JWKS fails closed until it can be fetched; `auth.Middleware` is gone — `Authenticator` is the one constructor), and the CORS allowlist (`cors.allowed_origins`, resolved per request). The corresponding YAML/env keys are **removed**: `server.cors_allowed_origins`, `query.default_max_rows`, `schema.refresh_interval`, `dedupe.enabled`, `dedupe.id_field`, `dedupe.require_id`, `stream.keepalive_interval`, `stream.keepalive_buckets`, `mq.gap_window_minutes`, `cache.timestamp_bucket_seconds`, `mq.max_bytes_gb`, `dlq.enabled`, `clickhouse.addr`, `clickhouse.http_port`, `clickhouse.http_scheme`, `clickhouse.database`, `clickhouse.username`, `clickhouse.query_timeout`, `auth.jwks_url`, `auth.role_claim` (and `WH_SERVER_CORS_ALLOWED_ORIGINS`, `WH_QUERY_DEFAULT_MAX_ROWS`, `WH_SCHEMA_REFRESH_INTERVAL`, `WH_DEDUPE_ENABLED`, `WH_DEDUPE_ID_FIELD`, `WH_DEDUPE_REQUIRE_ID`, `WH_STREAM_KEEPALIVE_INTERVAL`, `WH_STREAM_KEEPALIVE_BUCKETS`, `WH_MQ_GAP_WINDOW_MINUTES`, `WH_CACHE_TIMESTAMP_BUCKET_SECONDS`, `WH_MQ_MAX_BYTES_GB`, `WH_DLQ_ENABLED`, `WH_CH_ADDR`, `WH_CH_HTTP_PORT`, `WH_CH_HTTP_SCHEME`, `WH_CH_DATABASE`, `WH_CH_USERNAME`, `WH_CH_QUERY_TIMEOUT`, `WH_AUTH_JWKS_URL`, `WH_AUTH_ROLE_CLAIM`); the secrets — `clickhouse.password`, `auth.jwt_secret`, `auth.operator_key` — stay boot config on purpose (never in a tracked JSON file; combined with the adopted wiring on every reconnect, rotating one is a restart), and boot config is now **strict**: `config.Load` re-reads the YAML against the struct's tags and refuses to start naming every undeclared key, so a `dlq:` or `clickhouse: addr:` left behind can't be read, ignored, and believed; the binary carries **no compiled defaults** — every `config.json` key is required (validation names each missing one), so the adopted snapshot is what the files say, and once adopted it outlives its files (a deleted file or vanished directory is just a rejected reload). Defaults live in one checked-in seed directory (`internal/settings/seed/`, `go:embed`ded): the new **`wavehouse bootstrap [dir]`** writes it (refusing a non-empty directory, the `initdb` contract; the directory resolves exactly as it does for `validate` — the argument, else `WH_SETTINGS_DIR`, usage error with neither — so the two commands are interchangeable on one path and a bare `bootstrap` inside the container images seeds `/app/settings`), the dev `config.yaml` points at a gitignored `./settings` that `make dev` seeds from it, and the e2e fixture ships a copy. The container images ship **no** settings directory: `WH_SETTINGS_DIR` is preset to `/app/settings`, the operator mounts a directory there (`standalone.yaml` bind-mounts the checked-in `deployments/compose/settings/`), and a missing mount refuses to boot rather than running on defaults nobody chose. `dedupe.enabled` moves too: the new `dedupe.Managed` wraps the Pebble store and a `Store.AfterAdopt` hook opens or closes it after every adoption, so flipping the switch is a reload, not a restart (seen ids persist across an off/on cycle; a failed open on reload is logged and ingest fails closed with `500` until the next reload, since the files asked for dedupe — at boot it still refuses to start; a record caught in the instant of the flip is published un-deduped and counted by `wavehouse_ingest_dedupe_disabled_total` rather than failed, and the hook is registered before the boot apply so a reload can never leave the settings and the store out of step). The watcher reloads once as soon as its watch exists, closing the gap between the boot read and the watch — an edit landing in between (a ConfigMap update during a rolling restart) is adopted, not silently missed. `dedupe.enabled` / `WH_DEDUPE_ENABLED` are removed from boot config alongside the other keys. What stays in boot config is only what cannot change under a running process — resource sizing (`data_dir`, `cache.l1_max_cost`), the listeners, the observability exporters — and the secrets. The compose stack now bind-mounts a checked-in `deployments/compose/settings/` (the seed with `clickhouse.addr` pointed at the `clickhouse` service) instead of a volume seeded with `bootstrap`, so the quickstart is `up -d` again; the e2e orchestrator copies the fixture settings per run and patches the testcontainer's ClickHouse ports into `config.json`, since that wiring no longer has an env override. Every after-adopt hook (dedupe, keepalive wheel) is registered before the reload triggers start, so the watcher's first reload can never be missed by a hook. Consumers take functions, not values (`IngestHandler.DedupeSettings`, the structured-query handler's `defaultMaxRows` / `bucketSecs func() int`, the ingest worker's `dlqEnabled func(table) bool`, the sweeper's `gapWindow func() time.Duration`, `corsMiddleware`'s origins getter, `SchemaRegistry`'s database and refresh-interval sources, the query handlers' timeout sources), so `internal/api` stays testable without materializing settings directories. The settings directory is also the **runtime authority for access control and named pipes** (`internal/settings/store.go`, `internal/policy/source.go` (new), `internal/pipes/pipes.go`, `internal/api/{policy,pipes,router}.go`, `internal/stream/hub.go`, `internal/auth/auth.go`, `cmd/wavehouse/main.go`, `Makefile`, `deployments/compose/settings/{policies,roles}.json`, `clients/ts/src/settings.ts` (new); closes [#229](https://github.com/Wave-RF/WaveHouse/issues/229), [#33](https://github.com/Wave-RF/WaveHouse/issues/33), [#461](https://github.com/Wave-RF/WaveHouse/issues/461), [#514](https://github.com/Wave-RF/WaveHouse/issues/514), [#460](https://github.com/Wave-RF/WaveHouse/issues/460), [#363](https://github.com/Wave-RF/WaveHouse/issues/363); advances [#48](https://github.com/Wave-RF/WaveHouse/issues/48) and [#214](https://github.com/Wave-RF/WaveHouse/issues/214)): `roles.json`, `policies.json`, and `pipes.json` are adopted with `config.json` as one snapshot and re-adopted on the same three triggers, and **files are the only write path** — standalone, the operator edits them on the host; on WaveHouse Cloud the control plane writes them — so there is no stored copy that can skip validation: every adoption runs the current rules (strict decode rejecting unknown and duplicate keys, the full policy validation including the claim-template grammar, pipe name/SQL/parameter-type rules, and the cross-file check that every role a grant or `allowed_roles` names is declared in `roles.json`), and a rejected edit keeps the previous good policy and pipes in effect. `policies.json` is one policy document (`{}` = no policy, adopted fail-closed with a warning); `pipes.json` carries full definitions (`allowed_roles`, `parameters`, `description`), so a file-defined pipe is no longer admin-only by construction. Consumers read the adopted snapshot per request through `policy.Source` (a `func() *policy.Policy`; `settings.Store.Policy` in production, `policy.Static(p)` in tests) and `pipes.Source` (`settings.Store`; `pipes.Static(q...)` in tests), so a reload applies to the very next request, including the SSE hub's per-event policy read. `GET /v1/ops/policy`, `POST /v1/ops/policy/validate`, `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, and pipe execution are unchanged; the operator key still passes the `/v1/ops/*` gate under no policy, now as the break-glass that inspects the policy and triggers `POST /v1/ops/settings/reload` after `policies.json` is fixed. The SDK gains `wh.settings.reload()` (`POST /v1/ops/settings/reload`, returning `{ adopted, findings }`). The compose stack's trial `public` policy moves into the bind-mounted `deployments/compose/settings/policies.json` + `roles.json`, and `make dev` copies the same two files into its seeded `./settings` so a fresh dev server works tokenless. **Removed** — the write endpoints `PUT /v1/ops/policy`, `PUT /v1/ops/pipes/{name}`, and `DELETE /v1/ops/pipes/{name}`; the NATS KV buckets `WAVEHOUSE_POLICY` and `WAVEHOUSE_PIPES` and their KV Watch sync (`internal/policy/store.go`, the pipes KV store); the boot-config keys `policy.file_path` / `WH_POLICY_FILE_PATH` and `pipes.dir` / `WH_PIPES_DIR` (a leftover `policy:` or `pipes:` YAML block now refuses boot by name, like the other moved keys) and the `.sql`-directory pipes bootstrap; `deployments/compose/dev-policy.yaml`; the SDK methods `wh.policy.set`, `wh.pipes.set`, and `wh.pipes.delete`; and the test helpers `policy.NewMemoryStore`, `pipes.NewMemoryStore`, and `testutil/natsjs.go`. - **"Was this page helpful?" feedback widget on every docs page** (`docs/src/components/PageFeedback.astro` (new), `docs/src/components/Footer.astro`): a thumbs-up / thumbs-down vote below the page content, captured to PostHog as `docs_feedback` with `{ helpful, page }`. It renders from `Footer.astro`'s sidebar branch — the same indirection the Cloud CTA uses — rather than a per-page import or frontmatter flag, so every content page gets it automatically, including ones not written yet; it sits *below* the Cloud CTA on the pages that carry one, and splash pages (the homepage and 404) take the other footer branch and never render it. One vote per page per visitor: the choice is remembered in `localStorage` keyed by pathname, and a revisit renders the thanks message instead of re-prompting (storage is a nicety, not the record — a browser with storage disabled still votes). - **Settings-directory validation — `wavehouse validate [dir]`** (`internal/settings/` (new: `settings.go`, `validate.go`, `decode.go`, `finding.go`, + tests), `cmd/wavehouse/validate.go` (new, + tests), `cmd/wavehouse/main.go`): first piece of the file-based control plane (settings live in a directory of JSON documents — `roles.json`, `policies.json`, `pipes.json`, `config.json` — that a running instance will hot-reload; this change is validation-only — boot loading and reload wiring land separately). `settings.Validate(dir)` is the single gate every consumer of the directory runs: deliberately pure (no network, no ClickHouse — table/column existence stays with schema discovery, per Bring-Your-Own-Schema), and it collects **all** findings in one pass instead of failing on the first. Checks, layered: the directory holds exactly the four files (a missing file is an error — an empty document is `{}`, so absence always means deletion or a wrong path; any unexpected entry — file or directory — is an error so a typoed `polices.json` or a stray backup can't be silently ignored; dot-prefixed entries are the one carve-out, since erroring on vim swap files or the `..data` machinery Kubernetes ConfigMap mounts publish through would break hand editing and the cloud fan-out's mount pattern alike); strict JSON syntax (unknown fields rejected — the JSON form of the retired-config-key trap; empty/truncated files rejected, never read as an empty document; a leading UTF-8 byte order mark named as such instead of surfacing as a cryptic invalid-character error; a directory, unreadable file, or non-regular file (a FIFO would hang the read forever waiting for a writer; a stat gate rejects it — following symlinks, so Kubernetes ConfigMap mounts' symlink layout still passes) squatting on a settings filename named as the one real problem, not double-reported as "missing"; a top-level `null` rejected — the one well-formed document that decodes into a zero value without error, so it would silently read as "no settings"; trailing content rejected; duplicated object keys detected by a token-level pass, since `encoding/json` silently keeps the last copy); per-file shape rules (role names non-empty/unique, pipe names/SQL/param types, `config.json` bounds mirroring boot-config validation — its sections are the *tenant-owned* behavioral tunables (dedupe id_field/require_id plus per-table overrides under `dedupe.tables` — each entry overrides only the fields it names, resolving table → global → compiled default per field, so the effective id_field can never be empty — an explicit empty, whitespace-only, or whitespace-padded id_field is rejected at both levels, since an exact-match JSON key lookup would silently miss every row ([#222](https://github.com/Wave-RF/WaveHouse/issues/222)'s shape, unblocked by the file design since table names are runtime-resolved like policy grants); query default_max_rows, schema refresh_interval, CORS origins); platform-owned knobs like the SSE keepalives deliberately stay boot config); and cross-file referential integrity (every role a policy grant, `default_role`/`admin_role`, or pipe allowlist references must be declared in `roles.json`; an empty role string in a grant or allowlist is named as such — it matches no request and authorizes nobody). Warnings don't invalidate: a grant scoping the admin role (an unconditional bypass — dead config), `default_role` = admin, and a `default` on a required pipe parameter are flagged but legal. An empty `policies.json` means no policy — fail closed, matching deleted-policy semantics — and draws a warning naming the total lockout, so it announces itself at validation time instead of one 403 at a time. The CLI (`cmd/wavehouse/validate.go`, following the `health` subcommand pattern) takes the directory as an argument or from `WH_SETTINGS_DIR`, prints findings, and exits 0/1/2 (valid/invalid/usage) so CI and operators can gate config changes before they reach a running instance. The dispatch in `main.go` also grows `help` and `version` subcommands, and an unknown command is now a usage error instead of silently falling through and starting the server (`wavehouse validat` booting a listener is not a typo anyone wants); each subcommand parses its arguments with a stdlib `flag.FlagSet`, so `wavehouse -h` prints command-specific help and a stray flag or argument is a usage error rather than being silently swallowed. `WH_SETTINGS_DIR` has a single authority: `config.EnvSettingsDir`, with a reflection test pinning the `settings.dir` struct tag to it. The directory's location joins boot config as `settings.dir` (`WH_SETTINGS_DIR`; `internal/config/config.go`, `config.yaml`, `docs/src/content/docs/configuration.mdx`) — boot-tier by necessity, since it's the pointer the reload machinery follows; no default, same silent-misconfiguration reasoning as `policy.file_path`. diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 50bedaaa..2029bc0d 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -383,13 +383,13 @@ That is the layout a control plane writes. Each folder's `clickhouse` block is i The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe store needs no restructuring: it keys every tenant's seen ids by tenant, and the four files are tenant `0`, as a `0` folder is. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id is a finding of its own — that folder is skipped, and the rest of the directory still loads. -**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). Its message queue is kept, at the budget it last had, but the history gap-fill replays is purged from it at the next sweep, as a removed tenant's is, so a stream resumed after the fix has a hole where that history was. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. +**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). Its message queue is kept, at the budget it last had, but the history that gap-fill replays is purged from it at the next sweep, as a removed tenant's is, so a stream resumed after the fix has a hole where that history was. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. -**Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. That is how a tenant is removed: delete its folder, then reload the whole directory. Its open streams end, its routes answer `404`, and its queued rows are parked on the DLQ under its own subject; nothing it stored is deleted — its message queue is kept at the budget it last had, and only the history gap-fill replays goes from it, at the next sweep — so restoring the folder restores the tenant, seen ids and parked rows included. Reloading a deleted folder by name instead leaves its tenant rejected, answering `503`. The last folder can be removed the same way, with two catches, since `wavehouse validate` and boot both read an emptied directory as the four files missing: `validate` exits `1`, so a writer that gates each reload on it has to skip the check for that one reload, and a server restarted before a folder is written back refuses to boot. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. +**Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. That is how a tenant is removed: delete its folder, then reload the whole directory. Its open streams end, its routes answer `404`, and its queued rows are parked on the DLQ under its own subject; nothing it stored is deleted — its message queue is kept at the budget it last had, and only the history that gap-fill replays goes from it, at the next sweep — so restoring the folder restores the tenant, seen ids and parked rows included. Reloading a deleted folder by name instead leaves its tenant rejected, answering `503`. The last folder can be removed the same way, with two catches, since `wavehouse validate` and boot both read an emptied directory as the four files missing: `validate` exits `1`, so a writer that gates each reload on it has to skip the check for that one reload, and a server restarted before a folder is written back refuses to boot. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. **The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` takes it too, and reads a rejected or removed tenant's dead-letter queue like a served one's, since the queue is kept; a tenant that has none is a `404`. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. +**What a tenant's folder decides.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does (every tenant's store is a share of the one Pebble instance at `/pebble`, each key led by its tenant), so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. Its message queue is its own as well: its events are queued on a stream of their own, capped at its own `mq.max_bytes_gb` — at that budget its ingest answers `503` while every other tenant's keeps publishing — beside a dead-letter stream of its own at a tenth of it, and the history that gap-fill replays from it is kept for its own `stream.gap_window_minutes`. Nothing checks what the tenants' budgets add up to against the disk, so size them together ([Message Queue](/settings-directory#message-queue)). An event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. One setting weighs every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`. **What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. Tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its seen ids kept for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its open `GET /v1/stream` connections are ended, as any tenant's are when it stops being served — the other tenants' events are untouched. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 353608f2..4f7cc1b7 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -33,7 +33,7 @@ WaveHouse does not currently expose a knob to relax this — `SyncAlways` is alw Because the publish blocks on `fsync`, **your typical ingest latency is your storage's typical `fsync` latency, and your worst-case publish is your storage's worst-case `fsync`.** When that tail is healthy (sub-millisecond to single-digit milliseconds) the guarantee is essentially free. When it is not, the same code path that handles every production message stalls: - Publishes block for the duration of the `fsync`, so a multi-second `fsync` tail is a multi-second ingest tail. -- The embedded server's stream/consumer setup and every publish run under the JetStream client's request timeout; a slow-enough substrate makes them exceed it. The symptom at a first boot, which opens every tenant's queue, is `open dlq stream: ... context deadline exceeded`; a later boot writes nothing, so the first publish is where it shows. +- The embedded server's consumer setup and every publish run under the JetStream client's request timeout, and opening or resizing a tenant's queue under a ten-second budget of WaveHouse's own; a slow-enough substrate makes them exceed it. The symptom when a tenant's queue first opens — at the boot or reload that first serves the tenant — is `open dlq stream: ... context deadline exceeded`, or `open ingest stream: ...` (the two share the budget); a boot that finds every queue already at its budget writes nothing, so there the first publish is where it shows. - If the worker cannot drain to ClickHouse faster than producers publish, a tenant's stream fills toward its [`mq.max_bytes_gb`](/settings-directory#message-queue) and the API returns `503` to that tenant ([backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs)). ## Where `SyncAlways` is cheap vs. expensive @@ -81,7 +81,7 @@ Read the measured p99 against these bands, which track WaveHouse's `SyncAlways` | 1–5 ms | **Good** | | 5–50 ms | **Workable** — watch bursty load | | 50 ms – 1 s | **Marginal** — relax durability once `mq.sync_interval` ([#139](https://github.com/Wave-RF/WaveHouse/issues/139)) lands, or move to faster storage | -| > 1 s | **Broken** — opening a tenant's queue (`open dlq stream`) will time out under load; fix the storage substrate | +| > 1 s | **Broken** — opening a tenant's queue (`open dlq stream` / `open ingest stream`) will time out under load; fix the storage substrate | :::caution[macOS `fsync` lies by default] A plain `fsync()` on macOS returns once data is in the drive's volatile cache — it does **not** force a flush to NAND; only `fcntl(fd, F_FULLFSYNC)` does (NATS, Postgres, and SQLite all use it). On a Mac, any per-flush number under ~1 ms is almost certainly not a real flush — the gap between plain `fsync()` and `F_FULLFSYNC` can be ~180× on the same consumer NVMe. `fio` on macOS calls plain `fsync()`, so don't trust Mac `fio` numbers for tail-latency planning. This mostly matters when benchmarking a dev machine; production WaveHouse runs on Linux, where `fio` is honest. @@ -93,7 +93,7 @@ A self-contained `wavehouse storage-check` preflight subcommand that bakes this If you see any of these, benchmark the `/nats` volume as above: -- `open dlq stream: ... context deadline exceeded` when a tenant's queue first opens, at the first boot or at the reload that adopts the tenant. +- `open dlq stream: ... context deadline exceeded`, or `open ingest stream: ...`, when a tenant's queue first opens, at the boot or reload that first serves the tenant. - Ingest p99 latency in the seconds, or occasional `200`s that take multiple seconds to return. - Intermittent `503 Service Unavailable` from `/v1/ingest` when ClickHouse is healthy (the worker can't drain fast enough because acking is `fsync`-bound). - Flaky CI or load tests that pass on fast storage and fail on a shared/virtualized host. diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index b621f96e..c3762a8f 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -78,6 +78,10 @@ type tenantQueue struct { // tenant no longer served keeps the budget it last had, and maxBytes too // when the pair is whole at it (takeStock). maxBytes, asked int64 + // ingestCap is the cap the ingest stream has — what a failed resize + // restores it to. Not maxBytes: a pair boot found split has a cap but no + // budget applied in full, and a cap of 0 would be none at all. + ingestCap int64 } // EmbeddedNATS is the one implementation of every mq interface. @@ -194,7 +198,7 @@ func (e *EmbeddedNATS) takeStock(ctx context.Context) error { if id, ok := streamTenant(ingestStreamPrefix, name); ok { q := e.queue(id) q.ingest = true - q.asked = info.Config.MaxBytes + q.asked, q.ingestCap = info.Config.MaxBytes, info.Config.MaxBytes } else if id, ok := streamTenant(dlqStreamPrefix, name); ok { e.queue(id).dlq = true dlqs[id] = dlqState{limit: info.Config.MaxBytes, held: info.State.Bytes} @@ -310,10 +314,10 @@ func (e *EmbeddedNATS) MaxBytes(id tenant.ID) int64 { // JetStream applies a limit change to a live stream without touching its // messages: growing takes effect immediately; shrinking the ingest stream // below its current size makes DiscardNew refuse new publishes until the -// worker drains it — nothing buffered is dropped. The dead-letter stream is -// DiscardOld, which would delete its oldest parked rows to fit a smaller cap, -// so it is never capped below the bytes it holds (#532): it keeps what it -// has, and that is logged. +// sweeper purges it back under the cap — nothing buffered is dropped. The +// dead-letter stream is DiscardOld, which would delete its oldest parked rows +// to fit a smaller cap, so it is never capped below the bytes it holds (#532): +// it keeps what it has, and that is logged. // // The pair moves together where it can. If the dead-letter update fails after // the ingest one succeeded, the ingest resize is undone so the pair stays at @@ -358,7 +362,7 @@ func (e *EmbeddedNATS) apply(ctx context.Context, id tenant.ID, q *tenantQueue, if _, err := e.js.CreateOrUpdateStream(resizeCtx, ingestStreamConfig(id, maxBytes)); err != nil { return fmt.Errorf("open ingest stream: %w", err) } - q.ingest, q.maxBytes = true, maxBytes + q.ingest, q.maxBytes, q.ingestCap = true, maxBytes, maxBytes // The joins run on a budget of their own: a queue that opened but no // consumer holds fails every consumer (fail), so a slow open must not // leave them no time. @@ -371,17 +375,20 @@ func (e *EmbeddedNATS) apply(ctx context.Context, id tenant.ID, q *tenantQueue, } return nil } + prevCap := q.ingestCap if _, err := e.js.UpdateStream(resizeCtx, ingestStreamConfig(id, maxBytes)); err != nil { return fmt.Errorf("resize ingest stream: %w", err) } + q.ingestCap = maxBytes if err := e.applyDLQ(resizeCtx, id, q, maxBytes); err != nil { // The undo runs on its own budget, not the one the dead-letter call // has likely just exhausted. rollbackCtx, cancelRollback := context.WithTimeout(ctx, rollbackTimeout) defer cancelRollback() - if _, rollbackErr := e.js.UpdateStream(rollbackCtx, ingestStreamConfig(id, q.maxBytes)); rollbackErr != nil { + if _, rollbackErr := e.js.UpdateStream(rollbackCtx, ingestStreamConfig(id, prevCap)); rollbackErr != nil { return fmt.Errorf("%w (ingest stream rollback failed, so it stays at the new limit and the dlq at the previous: %w)", err, rollbackErr) } + q.ingestCap = prevCap return fmt.Errorf("%w (ingest stream restored to the previous limit)", err) } q.maxBytes = maxBytes diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index e75b62d7..3aeeb1c3 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -458,6 +458,55 @@ func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) } +// A resize whose dead-letter update fails undoes the ingest one, back to the +// cap the ingest stream had. That is not the budget applied in full: a boot +// that found the pair split applied none, and a cap of 0 would leave the +// ingest stream with no cap at all. +func TestEmbeddedNATS_SetMaxBytes_UndoRestoresTheIngestStreamsCap(t *testing.T) { + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + dir := t.TempDir() + first, err := NewEmbedded(dir) + require.NoError(t, err) + require.NoError(t, first.SetMaxBytes(ctx, "acme", 8<<20)) + require.NoError(t, first.js.DeleteStream(ctx, "DLQ_acme")) + require.NoError(t, first.Close()) + // The dead-letter stream cannot open again: a file where its store goes. + require.NoError(t, os.WriteFile(filepath.Join(dir, "jetstream", "$G", "streams", dlqStreamName("acme")), nil, 0o600)) + + e := openEmbedded(t, dir) + require.Zero(t, e.MaxBytes("acme"), "a pair without its dead-letter stream is not at its budget") + err = e.SetMaxBytes(ctx, "acme", 16<<20) + require.ErrorContains(t, err, "ingest stream restored to the previous limit") + assert.Equal(t, int64(8<<20), streamConfig(t, e, "INGEST_acme").MaxBytes, "back at the cap it had, not unlimited") + assert.Zero(t, e.MaxBytes("acme"), "and the next call retries") +} + +// A consumer that cannot join a tenant's queue opened after it started says so +// on failed — the one report that stops the ingest worker, which would +// otherwise let the tenant's ingest answer 200 for rows nobody reads. The +// queue itself is open, so SetMaxBytes succeeds. +func TestEmbeddedNATS_Consume_ReportsAQueueItCannotJoin(t *testing.T) { + e := openEmbedded(t, t.TempDir()) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + // A durable name the client refuses: with no queue yet, nothing checks it. + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: "bad.name", MaxAckPending: 10}) + require.NoError(t, err) + stop, failed, err := cons.Consume(func(*Message) {}, 4) + require.NoError(t, err) + t.Cleanup(stop) + + require.NoError(t, e.SetMaxBytes(ctx, "acme", testBudget)) + select { + case err := <-failed: + require.ErrorIs(t, err, ErrDeliveryEnded) + assert.Contains(t, err.Error(), "acme") + case <-time.After(5 * time.Second): + t.Fatal("a queue the consumer could not join was not reported") + } +} + // A budget that shrinks a tenant's dead-letter stream below what it holds // would have DiscardOld delete the oldest parked rows to fit (#532), so the // stream keeps what it holds, capped at that, and every row survives. From 07c6a91ec042fbdb33e01ea3a0379d0f418d7e54 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 20:35:20 -0400 Subject: [PATCH 06/15] fix(app): a rejected tenant keeps its replay history; review fixes --- AGENTS.md | 4 +-- CHANGELOG.md | 2 +- docs/src/content/docs/api.md | 4 +-- docs/src/content/docs/architecture.md | 8 ++--- docs/src/content/docs/deployment.md | 2 +- docs/src/content/docs/ingest-pipeline.md | 2 +- docs/src/content/docs/sdk/admin.md | 2 +- docs/src/content/docs/sdk/streaming.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- internal/app/app_test.go | 29 +++++++++++++----- internal/app/wire.go | 24 ++++++++++++--- internal/ingest/sweeper.go | 8 ++--- internal/mq/mq.go | 8 ++--- internal/settings/registry.go | 18 ++++++----- internal/settings/registry_test.go | 32 ++++++++++++++++++-- internal/stream/hub.go | 10 +++--- 16 files changed, 108 insertions(+), 49 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index dbf19e84..33935174 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,7 +29,7 @@ One binary: Eighteen internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): - **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers -- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` and the `mq.max_bytes_gb` reconcile handing the MQ each served tenant's own gap window and byte budget, and `defaultSetting`/`onDefaultAdopt` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it +- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` handing the sweeper each tenant's own gap window (a rejected tenant's as its folder last had it, unbounded for one rejected since boot) and the `mq.max_bytes_gb` reconcile each served tenant's byte budget, and `defaultSetting`/`onDefaultAdopt` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index). Every key leads with the tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for a result and its singleflight, `..
.
.` for a namespace — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` advances the tenant version that leads every namespace key of one tenant, orphaning every cached query keyed by its tables in one step (a pipe result names no table and keeps its TTL, [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the table-keyed cache (the structured-query results) of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) - **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config @@ -45,7 +45,7 @@ Eighteen internal packages under `internal/` (plus `internal/testutil/` for shar - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, down to none, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes - **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Prune` evicts the subscribers of every tenant a reload stopped serving; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable; `Evict` asks its handler to end the stream), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each served tenant's own gap window (`gapWindows`); each served tenant has a schema registry of its own (story 6) +- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`, `GET /v1/ops/dlq/stats`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions diff --git a/CHANGELOG.md b/CHANGELOG.md index 86dac970..90d647e0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,7 +32,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Changed -- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/subscriber.go`, `internal/settings/{settings,store}.go`, `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch is shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes`, keeping no acknowledged history for a tenant no longer served, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. +- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch is shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. - **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 9b675b4c..ecc0651b 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -612,7 +612,7 @@ Opens a persistent SSE connection for real-time event streaming. Supports histor | ------ | ----------- | | `Last-Event-ID` | RFC 3339 timestamp of the last received event. If present, overrides the `since` query parameter for automatic reconnection (standard `EventSource` behavior). | -**Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back — with a hole where the tenant's history was, which the sweeper purges within a minute of the tenant no longer being served — while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. +**Response:** SSE stream (`text/event-stream`). Data events include an `id:` field set to the event's `received_timestamp`. The stream opens with a `: connected` comment and emits a minimal `:` keepalive comment periodically (every 30 seconds by default), which keeps a quiet connection from being closed by a proxy; both are standard SSE comments that `EventSource` ignores (raw consumers should skip `:`-prefixed lines). When the server stops (see [Stopping](/deployment#stopping)) it ends every open stream immediately rather than holding it for the drain; `EventSource` reconnects on its own and resumes from `Last-Event-ID`. A reload that stops serving the stream's tenant — its folder removed or rejected, over a [nested settings directory](/deployment#the-nested-settings-directory) — ends that tenant's open streams the same way, and the reconnect then gets its `404` (removed) or `503` (rejected): the SDK stops on the `404` and retries the `503`, resuming from `Last-Event-ID` once the folder is back, while a browser `EventSource` treats either as fatal. A browser going cross-origin reads either refusal only when it passes CORS: it is decorated from tenant `0`'s list ([multi-tenant deployments](/deployment#multi-tenant-deployments)), so where tenant `0` is not served or its list does not admit the page's origin, the SDK sees a network error instead and keeps re-dialing. **Row values arrive positionally, and the column names are announced separately.** Before the first row, and again whenever the column list changes, the stream sends an `event: schema` frame naming the columns of the rows that follow — in order, already reduced to what the caller's role may read. That re-announcement is **not** guaranteed after a gap-fill across a column change; see the arity note below. Every data frame's `row` array then has exactly one value per announced column, in that order. `schema` is a **named** SSE event, so a browser `EventSource` must `addEventListener('schema', …)` — it never reaches `onmessage`. A schema frame carries **no** `id:` line, so it never moves the client's `Last-Event-ID`. In the example below the table has its own `received_timestamp` **column**, which collides by name with the frame's top-level `received_timestamp` **field** — they are different values: the field is when WaveHouse received the event, the row slot is that column as published (`null` where the record omitted it, which ClickHouse replaces with the column's default on insert). @@ -745,7 +745,7 @@ Triggers an immediate re-discovery of the `?tenant=`'s ClickHouse table schemas #### `GET /v1/ops/dlq/stats` — DLQ Statistics -Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The queue is read from the message queue rather than the settings, so a tenant whose folder was rejected or removed is read like one being served, for as long as its queue is kept. The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream exists from the moment the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. +Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The queue is read from the message queue rather than the settings, so a tenant whose folder was rejected or removed is read like one being served, since its queue is kept (nothing deletes it). The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream exists from the moment the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. **Error responses:** diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 549622f6..30a04ae2 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -90,7 +90,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each served tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out @@ -139,7 +139,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ - **worker.go** — `StartIngestWorker` launches an ingest pipeline: a durable `buffer-consumer` consumer of the ingest queue (created through `mq.ConsumerManager`) reads events, batches them per tenant table — the tenant read off each message's `mq.Topic` — and performs bulk INSERTs to ClickHouse. The pipeline is **insert-only**. The wire format `EventMessage` carries `{table_name, scope, received_timestamp, format, columns, row}` — the row positionally as one `JSONCompactEachRow` line, with `columns` naming its positions (the table's insertable columns — a computed one cannot be named in an `INSERT`); the worker batches per (tenant, table, column list) and writes `INSERT INTO … (cols) FORMAT JSONCompactEachRow`. It accepts any table name (events are addressed by `mq.Topic{Tenant, Table, Scope}` with raw names; `internal/mq` encodes them into subject tokens), then bulk-INSERTs. The embedded NATS server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only publishers that can reach the ingest queue are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (`policy.admin_role`) — see the Query Path section below; the `/v1/ops/*` `RequireAdmin` middleware enforces the check at the API layer, so a no/invalid-token request (resolved to `default_role`, not admin in a production config) never reaches the proxy. On a bulk-insert failure the batch is re-inserted row by row — except a batch whose tenant has no ClickHouse connection (no longer served, or no pool could be opened for it, such as by the connection ceiling), which no row could pass and `parkBatch` takes to the DLQ switch whole, logging once per batch rather than twice per row; rows that succeed are acked, and only the rows that fail again are routed to the DLQ (`sendToDLQ` → `mq.DeadLetterer.DeadLetter`), which parks the as-published `EventMessage` envelope under the topic it arrived on (`dlq.{tenant}.{table}` subjects inside `internal/mq`) with the failure context in `X-DLQ-*` headers when the tenant's `dlq.enabled` is on for the table — see [Ingest Pipeline](/ingest-pipeline) for the worker internals. - **types.go** — `EventMessage` struct (TableName, Scope — reserved, always empty today, ReceivedTimestamp, Format, Columns, Row; `Format` is `FormatJSONCompactEachRow` and `Row` is one positional line whose slots `Columns` names) and `BufferConsumerName` constant, shared across API handlers and the ingest pipeline. - **compact.go** — `EncodeCompactRow`, the positional row encoder every published row goes through, rendering one record over the table's **insertable** columns in declaration order. Serialization only: it validates nothing and judges no value. -- **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: each served tenant's own `stream.gap_window_minutes` — `internal/app`'s `gapWindows` — and none for a tenant no longer served). Finding the purge point is `internal/mq`'s (`purge.go`). +- **sweeper.go** — `Sweeper` implements the Active Sweeper pattern. It runs every minute and asks the MQ (`mq.Purger.PurgeAcked`) to drop the ingest events that are **both** ACKed by the buffer consumer (written to ClickHouse) **and** older than the gap window (re-read every sweep: each tenant's own `stream.gap_window_minutes`, a rejected tenant's as its folder last had it (unbounded for one rejected since boot) — `internal/app`'s `gapWindows` — and none for a removed tenant). Finding the purge point is `internal/mq`'s (`purge.go`). ### `mq/` — Message Queue @@ -190,7 +190,7 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi ### `tenant/` — Tenant Identifier -- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper hands the MQ each served tenant's own gap window (`gapWindows`); each served tenant has a schema registry of its own, built with its id (story 6). +- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper hands the MQ each tenant's own gap window (`gapWindows`, a rejected tenant's included); each served tenant has a schema registry of its own, built with its id (story 6). ### `chconn/` — ClickHouse Connection Pools @@ -255,7 +255,7 @@ Ingest worker pipeline (StartIngestWorker): Active Sweeper (async goroutine, every 60s), on each tenant's stream: → Read buffer consumer's AckFloor (highest contiguous ACKed seq) → Binary search for first message within that tenant's own gap window - (none for a tenant no longer served) + (a rejected tenant's as its folder last had it, unbounded for one rejected since boot; none for a removed tenant) → Purge target = MIN(ack_floor + 1, gap_window_seq) → Purge all messages below target from JetStream ``` diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index 2029bc0d..6ad7f34e 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -383,7 +383,7 @@ That is the layout a control plane writes. Each folder's `clickhouse` block is i The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe store needs no restructuring: it keys every tenant's seen ids by tenant, and the four files are tenant `0`, as a `0` folder is. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id is a finding of its own — that folder is skipped, and the rest of the directory still loads. -**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). Its message queue is kept, at the budget it last had, but the history that gap-fill replays is purged from it at the next sweep, as a removed tenant's is, so a stream resumed after the fix has a hole where that history was. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. +**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with, except an open `GET /v1/stream`, which is ended at once: its reconnect gets the `503` until the folder is fixed — the SDK keeps retrying and then resumes from `Last-Event-ID`, while a browser `EventSource` gives up on the `503` and has to be reopened. The rows the tenant had already accepted but not yet inserted, those of an ingest request in flight included, which still answers `200`, are parked on the DLQ under the tenant's own subject rather than held for the fix, as a removed tenant's are (see [Dead Letter Queue](#dead-letter-queue-dlq)). Its message queue is kept, at the budget it last had, and so is the history that gap-fill replays, for the `stream.gap_window_minutes` its folder last had (all of it, for a folder rejected since the server started, whose window the server never read): a stream resumed after a fix within that window picks up where it left off. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. **Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. That is how a tenant is removed: delete its folder, then reload the whole directory. Its open streams end, its routes answer `404`, and its queued rows are parked on the DLQ under its own subject; nothing it stored is deleted — its message queue is kept at the budget it last had, and only the history that gap-fill replays goes from it, at the next sweep — so restoring the folder restores the tenant, seen ids and parked rows included. Reloading a deleted folder by name instead leaves its tenant rejected, answering `503`. The last folder can be removed the same way, with two catches, since `wavehouse validate` and boot both read an emptied directory as the four files missing: `validate` exits `1`, so a writer that gates each reload on it has to skip the check for that one reload, and a server restarted before a folder is written back refuses to boot. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index 72447024..448fa6a0 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -228,7 +228,7 @@ Several layers throttle the pipeline, inner to outer: ## The Active Sweeper -The worker advances the consumer's `AckFloor` by acking; the sweep observes it to decide what is safe to purge. They never call each other — the consumer's `AckFloor` is their only contract. The sweeper (`internal/ingest`) owns the schedule and the window: each tick it calls `mq.Purger.PurgeAcked(buffer-consumer, cutoffs)` with each served tenant's cutoff at now − its own `stream.gap_window_minutes`; a tenant no longer served — its folder removed or rejected — is given none, and keeps none of the history it has acknowledged. The steps after the tick below are the embedded broker's implementation of that call, run on each tenant's stream at that tenant's cutoff. +The worker advances the consumer's `AckFloor` by acking; the sweep observes it to decide what is safe to purge. They never call each other — the consumer's `AckFloor` is their only contract. The sweeper (`internal/ingest`) owns the schedule and the window: each tick it calls `mq.Purger.PurgeAcked(buffer-consumer, cutoffs)` with each tenant's cutoff at now − its own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it, or one before anything it holds if its folder has been rejected since boot, so its clients resume once the folder is fixed; a removed tenant is given none, and keeps none of the history it has acknowledged. The steps after the tick below are the embedded broker's implementation of that call, run on each tenant's stream at that tenant's cutoff. ```mermaid flowchart TD diff --git a/docs/src/content/docs/sdk/admin.md b/docs/src/content/docs/sdk/admin.md index 59db0e5a..1dee059d 100644 --- a/docs/src/content/docs/sdk/admin.md +++ b/docs/src/content/docs/sdk/admin.md @@ -72,7 +72,7 @@ const { data } = await wh.dlq.list({ tenant: 'acme' }); const { data: clicks } = await wh.dlq.table('clicks', { tenant: 'acme' }); ``` -`wh.dlq.stream()` exists in the API but is **not yet functional**: there is no server-side DLQ stream today (the SSE bridge only carries `ingest.>` subjects), so it connects and receives no events rather than failing. Live DLQ streaming is tracked in [#197](https://github.com/Wave-RF/WaveHouse/issues/197). +`wh.dlq.stream()` exists in the API but is **not yet functional**: there is no server-side SSE route for dead-lettered events today (the SSE bridge only carries `ingest.>` subjects), so it connects and receives no events rather than failing. Live DLQ streaming is tracked in [#197](https://github.com/Wave-RF/WaveHouse/issues/197). --- diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 2b37667a..94a7e334 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -116,7 +116,7 @@ A dropped stream reconnects on a jittered exponential backoff, capped at 30s, an :::caution[Resumption is at-least-once, and time-bounded] Delivery across a reconnect is **at-least-once**. The `Last-Event-ID` the client sends is the last event's `received_timestamp`, and the server replays from that instant *inclusively* — so the last event you already saw, and anything sharing its timestamp, arrives again. The SDK does not deduplicate live frames — `liveQuery()` makes one pass at the backfill seam, and only under an ascending order ([#449](https://github.com/Wave-RF/WaveHouse/issues/449)) — so key on `timestamp` plus your own row identity if duplicates matter. -Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. So does a stream a [nested server](/deployment#the-nested-settings-directory) ended because its tenant's folder was rejected, once the folder is fixed: the sweeper purges a tenant's history within a minute of the tenant no longer being served. The same silence applies across a server upgrade to this release: the server deletes the previous release's queue at boot, so a replay spanning the upgrade omits the events published before it, without an error — backfill over REST if you need them. +Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. The same silence applies across a server upgrade to this release: the server deletes the previous release's queue at boot, so a replay spanning the upgrade omits the events published before it, without an error — backfill over REST if you need them. **A column-set change across a gap-fill is a known limitation.** If the table's columns change while you are connected *and* your client replays across that change, live rows arriving after the replay may not be preceded by a fresh `event: schema` frame until the columns next change or you reconnect. The SDK drops a row whose **length** disagrees with the list it was last told, rather than zipping it under the wrong names — so an added or removed column costs you rows, not wrong ones. A **same-length** change is the residual case the arity check cannot see: a `RENAME COLUMN`, or a drop paired with an add, zips values under the wrong names until the next announcement. Reconnecting resynchronizes either way. Full schema-change handling is deferred to the schema-versioning work ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). ::: diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index a0808514..99b1f679 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -221,7 +221,7 @@ A tenant's dead-letter stream exists from the moment the tenant is first served ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume. A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume — counting every tenant ever served on it, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream (up to a tenth of its last budget). A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. ## Streaming diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 8ef1f7c5..17e92812 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -730,10 +730,12 @@ func gapWindow(minutes int) map[string]any { return map[string]any{"stream": map[string]any{"keepalive_interval": 30, "keepalive_buckets": 3, "gap_window_minutes": minutes}} } -// Each tenant being served keeps its own stream.gap_window_minutes, since -// each has a queue of its own; a rejected tenant is not served, so it is not -// named and keeps no history (mq.Purger.PurgeAcked). A flat directory's single -// tenant gets exactly its own window. +// Each tenant keeps its own stream.gap_window_minutes, since each has a queue +// of its own — a rejected tenant the window its folder last had, so its +// clients resume once the folder is fixed, and everything while that window +// is unknown. A removed tenant is not named and keeps no history +// (mq.Purger.PurgeAcked). A flat directory's single tenant gets exactly its +// own window. func TestGapWindows(t *testing.T) { open := func(t *testing.T, dir string) *settings.Registry { t.Helper() @@ -754,11 +756,24 @@ func TestGapWindows(t *testing.T) { rewriteSettings(t, filepath.Join(root, "globex"), invalidQuery) tenants.Reload("test") - assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute, "initech": 30 * time.Minute}, gapWindows(tenants)) + assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute, "globex": 60 * time.Minute, "initech": 30 * time.Minute}, gapWindows(tenants), + "a rejected tenant keeps the window its folder last had") + + require.NoError(t, os.RemoveAll(filepath.Join(root, "globex"))) + tenants.Reload("test") + assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute, "initech": 30 * time.Minute}, gapWindows(tenants), + "a removed tenant keeps none") }) - t.Run("no tenant served names none", func(t *testing.T) { - assert.Empty(t, gapWindows(open(t, writeNestedSettings(t, map[string]map[string]any{"acme": invalidQuery})))) + t.Run("a folder rejected since boot keeps everything", func(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": invalidQuery}) + tenants := open(t, root) + assert.Equal(t, map[tenant.ID]time.Duration{"acme": keepEverything}, gapWindows(tenants)) + + rewriteSettings(t, filepath.Join(root, "acme"), gapWindow(15)) + tenants.Reload("test") + assert.Equal(t, map[tenant.ID]time.Duration{"acme": 15 * time.Minute}, gapWindows(tenants), + "its own window once its folder validates") }) } diff --git a/internal/app/wire.go b/internal/app/wire.go index e08f4a0e..ed3b98ec 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -6,6 +6,7 @@ import ( "fmt" "log/slog" "maps" + "math" "net" "net/http" "os" @@ -130,18 +131,31 @@ func shortestKeepalive(tenants *settings.Registry) (period time.Duration, bucket return period, buckets } -// gapWindows is the history the sweeper keeps for each tenant being served: -// its own stream.gap_window_minutes, since each tenant's events have a queue -// of their own. A tenant it does not name — removed or rejected — keeps no -// history (mq.Purger.PurgeAcked). +// gapWindows is the history the sweeper keeps for each tenant: its own +// stream.gap_window_minutes, since each tenant's events have a queue of their +// own — for a rejected tenant, the window its folder last had, because a +// rejection is the common reload failure (a typo, fixed minutes later) and +// its clients resume from Last-Event-ID once it is served again. A removed +// tenant is not named, so it keeps no history (mq.Purger.PurgeAcked). func gapWindows(tenants *settings.Registry) map[tenant.ID]time.Duration { windows := map[tenant.ID]time.Duration{} - for id, store := range tenants.All() { + for id, store := range tenants.Known() { + if store == nil { + windows[id] = keepEverything + continue + } windows[id] = store.GapWindow() } return windows } +// keepEverything is the window of a tenant whose folder has been rejected +// since boot: this process has never read its stream.gap_window_minutes, so +// none of the history its queue holds is known to be past it. A rejected +// tenant is sent no new events, so what it keeps is what its queue held at +// boot. +const keepEverything = time.Duration(math.MaxInt64) + // served reports whether the registry is serving tenant id: what the // per-tenant resources — verifiers, dedupe stores, open streams — are pruned // by once a reload removes or rejects their tenant. diff --git a/internal/ingest/sweeper.go b/internal/ingest/sweeper.go index 367e22af..b0d4a2d1 100644 --- a/internal/ingest/sweeper.go +++ b/internal/ingest/sweeper.go @@ -22,10 +22,10 @@ import ( // (see mq.Purger). type Sweeper struct { purger mq.Purger - // gapWindows is the history to keep for each tenant being served, read on - // every sweep so a reload of stream.gap_window_minutes applies from the - // next sweep without a restart. A tenant it does not name — one removed - // or rejected — keeps no history (mq.Purger.PurgeAcked). + // gapWindows is the history to keep for each tenant, read on every sweep + // so a reload of stream.gap_window_minutes applies from the next sweep + // without a restart. A tenant it does not name keeps no history + // (mq.Purger.PurgeAcked). gapWindows func() map[tenant.ID]time.Duration } diff --git a/internal/mq/mq.go b/internal/mq/mq.go index 663bde3d..47897978 100644 --- a/internal/mq/mq.go +++ b/internal/mq/mq.go @@ -276,10 +276,10 @@ type Purger interface { // its first unacked event) AND stored before that tenant's cutoff in // olderThan. Either bound alone keeps the event: unacked events are not // yet written, and recent ones are still needed for replay. A tenant - // olderThan does not name — one no longer served — keeps no history: - // everything it has acknowledged goes. Reports whether anything was - // removed. ErrConsumerNotFound when the consumer has not been created on - // some tenant's queue; the other tenants' are purged all the same. + // olderThan does not name keeps no history: everything it has + // acknowledged goes. Reports whether anything was removed. + // ErrConsumerNotFound when the consumer has not been created on some + // tenant's queue; the other tenants' are purged all the same. PurgeAcked(ctx context.Context, consumer string, olderThan map[tenant.ID]time.Time) (purged bool, err error) } diff --git a/internal/settings/registry.go b/internal/settings/registry.go index 663e0a23..e624a780 100644 --- a/internal/settings/registry.go +++ b/internal/settings/registry.go @@ -154,13 +154,17 @@ func (r *Registry) All() iter.Seq2[tenant.ID, *Store] { } // Known iterates over every tenant the registry holds, served or rejected, -// in id order — for a consumer that must keep a rejected tenant's resources -// current too: the tenant comes back into service with them, and a rejection -// is the common reload failure (a typo, fixed and reloaded minutes later). -func (r *Registry) Known() iter.Seq[tenant.ID] { - return func(yield func(tenant.ID) bool) { - for _, id := range slices.Sorted(maps.Keys(*r.tenants.Load())) { - if !yield(id) { +// in id order, each with the store holding its last adopted settings — nil +// for a tenant whose folder has not validated since boot — for a consumer +// that must keep a rejected tenant's resources current too: the tenant comes +// back into service with them, and a rejection is the common reload failure +// (a typo, fixed and reloaded minutes later). A rejected tenant's store +// serves no request; it is handed out for what those resources read of it. +func (r *Registry) Known() iter.Seq2[tenant.ID, *Store] { + return func(yield func(tenant.ID, *Store) bool) { + tenants := *r.tenants.Load() + for _, id := range slices.Sorted(maps.Keys(tenants)) { + if !yield(id, tenants[id].store) { return } } diff --git a/internal/settings/registry_test.go b/internal/settings/registry_test.go index 0bdff569..06e2a3b9 100644 --- a/internal/settings/registry_test.go +++ b/internal/settings/registry_test.go @@ -5,7 +5,6 @@ import ( "log/slog" "os" "path/filepath" - "slices" "testing" "github.com/stretchr/testify/assert" @@ -307,18 +306,45 @@ func TestRegistry_HooksRunOnAReloadThatAdoptsNothing(t *testing.T) { } // Known is every tenant the registry holds, rejected ones included, in id -// order: what a resource a rejected tenant comes back to is kept current for. +// order, each with the store of its last adopted settings: what a resource a +// rejected tenant comes back to is kept current from. func TestRegistry_Known(t *testing.T) { t.Parallel() root := writeTree(t, map[string]map[string]string{"globex": maxRowsFiles(222), "acme": maxRowsFiles(111), "broken": brokenFiles()}) reg, _ := Open(root) require.NotNil(t, reg) - assert.Equal(t, []tenant.ID{"acme", "broken", "globex"}, slices.Collect(reg.Known())) + known := func() ([]tenant.ID, map[tenant.ID]*Store) { + var ids []tenant.ID + stores := map[tenant.ID]*Store{} + for id, store := range reg.Known() { + ids = append(ids, id) + stores[id] = store + } + return ids, stores + } + ids, stores := known() + assert.Equal(t, []tenant.ID{"acme", "broken", "globex"}, ids) + assert.Nil(t, stores["broken"], "a folder that has not validated since boot has no settings to hand out") + acme, _ := reg.For("acme") + assert.Same(t, acme, stores["acme"]) var served []tenant.ID for id := range reg.All() { served = append(served, id) } assert.Equal(t, []tenant.ID{"acme", "globex"}, served, "All leaves the rejected tenant out; Known does not") + + // A tenant rejected after an adoption still comes with that adoption's + // settings: the store keeps its last document. + for name, content := range brokenFiles() { + require.NoError(t, os.WriteFile(filepath.Join(root, "acme", name), []byte(content), 0o600)) + } + reg.Reload("test") + _, ok := reg.For("acme") + require.False(t, ok) + _, stores = known() + require.Same(t, acme, stores["acme"]) + assert.Equal(t, 111, stores["acme"].DefaultMaxRows()) + // Stopping early is the iterator's contract, not the caller's problem. for id := range reg.Known() { assert.Equal(t, tenant.ID("acme"), id) diff --git a/internal/stream/hub.go b/internal/stream/hub.go index e7ce2d4d..b13221bf 100644 --- a/internal/stream/hub.go +++ b/internal/stream/hub.go @@ -43,8 +43,9 @@ type Hub struct { // the default implementation, which delegates to ResolvedPermissions.RowVisible // — today's behavior unchanged. Wired once before the Hub serves traffic and // not safe to mutate afterwards: rowAdmitted reads it from the consumer - // goroutine and from SSE handler goroutines without holding h.mu. Every - // delivery path reaches it through rowAdmitted, never directly. + // goroutines (one per tenant) and from SSE handler goroutines without + // holding h.mu. Every delivery path reaches it through rowAdmitted, never + // directly. RowEvaluator RowEvaluator } @@ -393,9 +394,8 @@ func newEventView(raw []byte) *eventView { // The hub is a second consumer of the same events as the ingest worker, and // acks independently of it, so a format only the worker refuses would stream // to clients while the worker parks it on the DLQ. Refusing it here keeps the - // two readers agreeing on what the bytes mean. Today only a pre-v2 envelope - // declares anything else, and it would fail pairing anyway on its empty - // column list — this is what holds once a second format exists. + // two readers agreeing on what the bytes mean. Today every envelope + // declares that one format — this is what holds once a second exists. if ev.evt.Format != ingest.FormatJSONCompactEachRow { return ev } From 04aad9ffb4751b8719637b1bb214cb683154fd8a Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 21:02:27 -0400 Subject: [PATCH 07/15] fix(mq): share the hub bridge's fetch-ahead across tenants; review fixes --- CHANGELOG.md | 2 +- docs/src/content/docs/api.md | 4 ++-- docs/src/content/docs/architecture.md | 4 ++-- docs/src/content/docs/settings-directory.mdx | 2 +- internal/mq/embedded.go | 10 ++++++---- internal/mq/embedded_test.go | 16 ++++++++++++++++ internal/mq/mq.go | 4 +++- internal/settings/settings.go | 4 ++-- 8 files changed, 33 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 90d647e0..76564ceb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,7 +32,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Changed -- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch is shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. +- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. - **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index ecc0651b..df65cfe4 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -745,7 +745,7 @@ Triggers an immediate re-discovery of the `?tenant=`'s ClickHouse table schemas #### `GET /v1/ops/dlq/stats` — DLQ Statistics -Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The queue is read from the message queue rather than the settings, so a tenant whose folder was rejected or removed is read like one being served, since its queue is kept (nothing deletes it). The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream exists from the moment the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. +Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The queue is read from the message queue rather than the settings, so a tenant whose folder was rejected or removed is read like one being served, since its queue is kept (nothing deletes it). The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream is opened when the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. **Error responses:** @@ -754,7 +754,7 @@ Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant] | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason) | | 400 | `{"error":"invalid query string: …"}` / `{"error":"invalid ?tenant: …"}` | The query string does not parse (`?tenant=acme;x=1`, a bad `%` escape), or `tenant` is empty, repeated, or not a tenant id | | 403 | `{"error":"forbidden"}` | Caller's role is not the policy `admin_role` (`"admin"` by default) | -| 404 | `{"error":"no dead-letter queue for tenant: "}` | The tenant has no dead-letter queue: it has never been served on this data directory, or the id names no tenant | +| 404 | `{"error":"no dead-letter queue for tenant: "}` | The tenant has no dead-letter queue: it has never been served on this data directory, its queue could not be opened (see [Message Queue](/settings-directory#message-queue)), or the id names no tenant | | 500 | `{"error":"stream info failed"}` | NATS JetStream stream-info lookup failed | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while tenant `0`'s JWKS has not been fetched yet (the ops tree verifies as tenant `0`); refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 30a04ae2..51ab00d0 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -90,7 +90,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where they would hold the ack floor and stop the sweeper. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out @@ -145,7 +145,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ The **only** package that imports NATS/JetStream — a `depguard` rule in `.golangci.yml` fails `make lint` on any `github.com/nats-io` import in every package golangci-lint builds; the `integration`-tagged files under `tests/` sit outside its default build context, so the boundary there rests on convention (AGENTS.md Key Design Decision #20). Every other package talks to the broker through the types below, so a subject, stream, or broker change lands here once. -- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. +- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. - **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts the sweep. It runs on each tenant's stream at that tenant's cutoff. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. - **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, or is refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 99b1f679..caf5b685 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -217,7 +217,7 @@ A failed batch insert is retried row by row; a row that fails again on its own i For a tenant no longer served — its folder removed or rejected — there is no switch to read: its rows are always parked, so none of them sits unacked in its ingest queue, redelivered for as long as the tenant is away and stopping the [Active Sweeper](/ingest-pipeline#the-active-sweeper) purging that queue. -A tenant's dead-letter stream exists from the moment the tenant is first served (an empty stream costs nothing) and the stats endpoint is always registered — the switch is purely behavioral, which is what makes it safe to reload. +A tenant's dead-letter stream is opened when the tenant is first served (an empty stream costs nothing) and the stats endpoint is always registered — the switch is purely behavioral, which is what makes it safe to reload. ## Message Queue diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index c3762a8f..2a990960 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -547,9 +547,11 @@ func wrapMsg(ctx context.Context, m jetstream.Msg) *Message { // Subscribe holds a durable explicit-ack consumer named consumerName on every // tenant's queue, those opened later included, and delivers each message to -// handler with the trace context its headers carry, until ctx is done. A -// tenant's queue that cannot be joined when it opens is logged: its events -// reach handler from the next boot. +// handler with the trace context its headers carry, until ctx is done. It +// fetches the client's default number of messages ahead across the tenants +// together (see fanIn.share), so what sits client-side does not grow with +// the tenants. A tenant's queue that cannot be joined when it opens is +// logged: its events reach handler from the next boot. func (e *EmbeddedNATS) Subscribe(ctx context.Context, consumerName string, handler func(msg *Message) error) error { f := e.newFanIn(ctx, jetstream.ConsumerConfig{Durable: consumerName, AckPolicy: jetstream.AckExplicitPolicy}) f.fail = func(err error) { @@ -563,7 +565,7 @@ func (e *EmbeddedNATS) Subscribe(ctx context.Context, consumerName string, handl if err := handler(msg); err != nil { _ = msg.Nak() } - }, 0, false) + }, jetstream.DefaultMaxMessages, false) if err != nil { return fmt.Errorf("consume: %w", err) } diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index 3aeeb1c3..f22fb7b0 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -1070,6 +1070,22 @@ func TestFanIn_SharesThePrefetch(t *testing.T) { assert.Zero(t, (&fanIn{handles: handles(3)}).share(), "0 leaves the client default") } +// The hub bridge's fetch-ahead is the client default split across the +// tenants' queues, like the worker's prefetch, so what it holds client-side +// does not grow with the number of tenants. +func TestEmbeddedNATS_Subscribe_SharesTheClientDefault(t *testing.T) { + e := newTestEmbedded(t, "acme", "globex") + ctx, cancel := context.WithCancel(t.Context()) + defer cancel() + require.NoError(t, e.Subscribe(ctx, "hub-bridge", func(*Message) error { return nil })) + + e.mu.Lock() + defer e.mu.Unlock() + require.Len(t, e.consumers, 1) + assert.Equal(t, jetstream.DefaultMaxMessages, e.consumers[0].prefetch) + assert.Equal(t, jetstream.DefaultMaxMessages/2, e.consumers[0].share()) +} + // Nothing lands on the default tenant by omission (#583): the tenant is a // required token, checked against its grammar before anything is sent. func TestEmbeddedNATS_Publish_RefusesATopicWithoutATenant(t *testing.T) { diff --git a/internal/mq/mq.go b/internal/mq/mq.go index 47897978..6626cfa8 100644 --- a/internal/mq/mq.go +++ b/internal/mq/mq.go @@ -164,7 +164,9 @@ type Subscriber interface { // consumer named consumerName, held on every tenant's queue — those // opened after Subscribe included. The handler runs on one delivery // goroutine per tenant, one message at a time, so it must be safe to - // call concurrently for different tenants. + // call concurrently for different tenants. The messages fetched ahead of + // it are a fixed number split across the tenants, as Consumer.Consume's + // prefetch is, so they do not grow with the number of tenants. // // CONTRACT: If the handler intends to return an error to trigger automatic // redelivery, it MUST NOT manually call msg.Ack() or msg.Nak() beforehand. diff --git a/internal/settings/settings.go b/internal/settings/settings.go index 2ce9b120..7da6c3fa 100644 --- a/internal/settings/settings.go +++ b/internal/settings/settings.go @@ -165,8 +165,8 @@ type TableDedupe struct { // DLQConfig gates the Dead Letter Queue: whether a row that still fails // after the row-by-row isolation retry is parked on the tenant's dead-letter // queue (and its original acked) or left unacked to be redelivered -// indefinitely. The queue exists from the moment the tenant is first served — -// empty until something lands on it — so the switch is purely behavioral and +// indefinitely. The queue is opened when the tenant is first served — empty +// until something lands on it — so the switch is purely behavioral and // resolves per table through the same override cascade as dedupe. type DLQConfig struct { Enabled *bool `json:"enabled"` From 57870c4d19dea281845e85c386ee580894070645 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 21:33:51 -0400 Subject: [PATCH 08/15] fix(mq): publish only into a queue the broker recorded open; review fixes --- AGENTS.md | 2 +- CHANGELOG.md | 2 +- docs/src/content/docs/architecture.md | 4 +- docs/src/content/docs/settings-directory.mdx | 4 +- internal/api/ingest.go | 4 +- internal/app/app.go | 10 ++-- internal/app/wire.go | 48 +++++------------- internal/mq/embedded.go | 51 +++++++++++++++----- internal/mq/embedded_test.go | 42 ++++++++++++++++ 9 files changed, 105 insertions(+), 62 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 33935174..3d780d92 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,7 +29,7 @@ One binary: Eighteen internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): - **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers -- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` handing the sweeper each tenant's own gap window (a rejected tenant's as its folder last had it, unbounded for one rejected since boot) and the `mq.max_bytes_gb` reconcile each served tenant's byte budget, and `defaultSetting`/`onDefaultAdopt` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it +- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive` for the one setting folded over every tenant served, `gapWindows` handing the sweeper each tenant's own gap window (a rejected tenant's as its folder last had it, unbounded for one rejected since boot) and the `mq.max_bytes_gb` reconcile each served tenant's byte budget, and `defaultPolicy` for the one setting that still follows tenant `0`, a flat directory's ops-gate admin role; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`, and the same hook's `Hub.Prune` ends the open streams of a tenant no longer served), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's - **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index). Every key leads with the tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for a result and its singleflight, `..
.
.` for a namespace — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` advances the tenant version that leads every namespace key of one tenant, orphaning every cached query keyed by its tables in one step (a pipe result names no table and keeps its TTL, [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the table-keyed cache (the structured-query results) of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) - **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config diff --git a/CHANGELOG.md b/CHANGELOG.md index 76564ceb..c24d0df9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,7 +32,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Changed -- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. +- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it, and so is the tracking of tenant `0`'s last adopted store: a flat directory's ops gate, its one reader left, reads tenant `0` through the registry). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. - **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 51ab00d0..9da998d1 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -90,7 +90,7 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request ### `app/` — Process wiring - **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`, tracked by an `onDefaultAdopt` hook that runs only after a reload that adopted it), so a `0` folder that a reload rejects or removes leaves it as it was. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped, and a removed or rejected tenant's queued rows are parked rather than left unacked, where each would be redelivered every ack wait for as long as the tenant is away and would hold that tenant's ack floor, so the sweeper could purge none of its queue past it. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there; before that, a diagnostic naming a tenant a reload stopped serving goes back to the no-tenant one. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one setting that still follows the default tenant is read per request, the admin role of a flat directory's ops gate: `defaultPolicy` reads it through the registry, where a flat directory's tenant `0` is always served. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. `wireStreaming`'s hook prunes the stream hub the same way (`Hub.Prune`, with the one `served` predicate the auth and dedupe hooks use too), ending the open streams of a tenant no longer served. One setting is shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)). The sweeper is handed each tenant's own `stream.gap_window_minutes` (`gapWindows`, read every sweep over `Registry.Known`, so a rejected tenant keeps the window its folder last had, and all of its history if the folder has been rejected since boot), since each tenant's events have a queue of their own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over the `Tenant` factory of the embedded Pebble implementation (`dedupe.NewEmbedded`), handing it `data_dir` once; the implementation decides where every tenant's store lives — one instance, each key led by its tenant (story 3) — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its seen ids kept when the tenant is switched off, rejected, or removed. An instance that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for every tenant with dedupe on over a nested one. The system gauges report that one instance's figures (`Embedded.Stats`), not a sum over tenants. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. `wireMQ` hands each served tenant's `mq.max_bytes_gb` to `mq.Broker.SetMaxBytes` at boot, under `New`'s context (so a stop signaled mid-boot is not held up by opening many queues), and again after every reload, under the App's stop context; the first apply opens that tenant's queue. A queue that cannot be opened or resized follows the registry's rule for the shape — fatal at boot over a flat directory, logged over a nested one — and is retried by the next reload, a queue that did not open by the next publish too. How the budget is split across the tenant's streams, the time bounds, the rollback, and the dead-letter shrink guard are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out @@ -148,7 +148,7 @@ The **only** package that imports NATS/JetStream — a `depguard` rule in `.gola - **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. - **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts the sweep. It runs on each tenant's stream at that tenant's cutoff. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. -- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, or is refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. +- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. ### `observability/` — OpenTelemetry Pipeline diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index caf5b685..d34c0a65 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -221,7 +221,9 @@ A tenant's dead-letter stream is opened when the tenant is first served (an empt ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume — counting every tenant ever served on it, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream (up to a tenth of its last budget). A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. + +**Sizing the volume.** Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume — counting every tenant ever served on it, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream (up to a tenth of its last budget). A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. ## Streaming diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 029799b6..10daddea 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -707,10 +707,10 @@ func (h *IngestHandler) processRecord( slog.DebugContext(ctx, "publishing event to the ingest queue", "table", table, "scope", scope) if err := h.Publisher.Publish(ctx, mq.Topic{Tenant: store.Tenant(), Table: table, Scope: scope}, payload); err != nil { if errors.Is(err, mq.ErrQueueFull) { - slog.WarnContext(ctx, "ingest queue is full", "error", err, "table", table, "scope", scope) + slog.WarnContext(ctx, "ingest queue is full", "tenant", store.Tenant(), "error", err, "table", table, "scope", scope) return false, nil, &requestAbort{Status: http.StatusServiceUnavailable, Message: "service unavailable", RetryAfter: "30"} } - slog.ErrorContext(ctx, "failed to publish to the ingest queue", "error", err, "table", table, "scope", scope) + slog.ErrorContext(ctx, "failed to publish to the ingest queue", "tenant", store.Tenant(), "error", err, "table", table, "scope", scope) return false, nil, &requestAbort{Status: http.StatusInternalServerError, Message: "publish failed"} } diff --git a/internal/app/app.go b/internal/app/app.go index 2131942c..a7aec9d1 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -18,7 +18,7 @@ // handed whole to each component's wiring function, which derives the // per-call getters the internal packages take: keyed by the request's store // for the handlers, by tenant id for the async paths (perTenant), and fixed -// to the default tenant for the ops gate of a flat directory (defaultSetting). +// to the default tenant for the ops gate of a flat directory (defaultPolicy). package app import ( @@ -30,7 +30,6 @@ import ( "net/http" "os" "os/signal" - "sync/atomic" "time" "golang.org/x/sync/errgroup" @@ -89,11 +88,8 @@ type App struct { listener net.Listener // tenants is the registry every tenant-aware path resolves through, and - // the owner of every reload. defaultStore is tenant 0's store as of its - // last adoption, which the ops gate of a flat directory reads its admin - // role from (defaultSetting). - tenants *settings.Registry - defaultStore atomic.Pointer[settings.Store] + // the owner of every reload. + tenants *settings.Registry // policies is the default tenant's policy, for the ops gate of a flat // directory. policies policy.Source diff --git a/internal/app/wire.go b/internal/app/wire.go index ed3b98ec..968d5f34 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -66,50 +66,24 @@ func (a *App) wireSettings() error { return fmt.Errorf("settings directory %s invalid, refusing to start — findings above; `wavehouse validate` reproduces them, `wavehouse bootstrap` writes a starter directory", a.cfg.Settings.Dir) } a.tenants = tenants - // Registered first: hooks run in registration order, so every reload - // updates the tracked store before any other hook runs. - a.trackDefaultStore() - a.onDefaultAdopt(a.trackDefaultStore) - a.policies = func() *policy.Policy { return defaultSetting(a, (*settings.Store).Policy) } + a.policies = func() *policy.Policy { return defaultPolicy(tenants) } if !tenants.Nested() && a.policies() == nil { slog.Warn("no policy adopted — every token-based request is denied until policies.json defines one (fail closed)") } return nil } -// trackDefaultStore remembers tenant 0's store as of its last adoption. The -// registry stops handing out a rejected tenant's store and forgets a removed -// one, but the store keeps its last adopted document either way — and that is -// what defaultSetting goes on reading. -func (a *App) trackDefaultStore() { - if store, ok := a.tenants.For(tenant.Default); ok { - a.defaultStore.Store(store) +// defaultPolicy is the default tenant's access-control policy, which the ops +// gate of a flat directory reads its admin role from per request. There +// tenant 0 is the whole directory, always served: a reload that fails keeps +// the previous document. A nested directory's ops gate reads no policy at all +// (api.NewRouter). +func defaultPolicy(tenants *settings.Registry) *policy.Policy { + store, ok := tenants.For(tenant.Default) + if !ok { + return nil } -} - -// defaultSetting reads one setting of the default tenant: the admin role the -// ops gate of a flat directory reads per request. It reads tenant 0's last -// adopted document, so a 0 folder a reload rejected or removed leaves its -// reader as it was. A nested directory that has never served a tenant 0 reads -// T's zero value, and its ops gate reads no policy at all. -func defaultSetting[T any](a *App, get func(*settings.Store) T) T { - store := a.defaultStore.Load() - if store == nil { - var zero T - return zero - } - return get(store) -} - -// onDefaultAdopt registers fn to run after each reload that adopts the -// default tenant, so a nested directory's other tenants never move what -// follows it, and a rejected 0 folder leaves that as it was. -func (a *App) onDefaultAdopt(fn func()) { - a.tenants.AfterAdopt(func(adopted []tenant.ID) { - if slices.Contains(adopted, tenant.Default) { - fn() - } - }) + return store.Policy() } // shortestKeepalive is the shape of the one keepalive wheel every tenant's diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index 2a990960..bebcfa63 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -57,15 +57,20 @@ type EmbeddedNATS struct { conn *nats.Conn js jetstream.JetStream - // mu guards queues and consumers, and serializes opening or resizing a - // tenant's queue with registering a consumer, so a queue opened while a - // consumer registers is never missed by it. It is held across the - // JetStream calls that open or resize a queue. + // mu guards queues, consumers and writes to opened, and serializes + // opening or resizing a tenant's queue with registering a consumer, so a + // queue opened while a consumer registers is never missed by it. It is + // held across the JetStream calls that open or resize a queue. mu sync.Mutex queues map[tenant.ID]*tenantQueue // consumers are the durable consumers held on every tenant's queue, each // joined to a queue as it opens. consumers []*fanIn + // opened holds the tenants whose queue has both streams and every + // registered consumer joined — what Publish trusts, rather than a stream + // answering: an open that gave up can leave behind a stream JetStream goes + // on to create, which no consumer holds. Written under mu, read without it. + opened sync.Map // tenant.ID → struct{} } // tenantQueue is what the broker knows of one tenant's queue. @@ -219,6 +224,7 @@ func (e *EmbeddedNATS) takeStock(ctx context.Context) error { if q.ingest && ok && (d.limit == tenth || guarded) { q.maxBytes = q.asked } + e.record(id, q) } return nil } @@ -267,6 +273,19 @@ func (e *EmbeddedNATS) ingestTenants() []tenant.ID { return ids } +// record brings opened in line with what the broker knows of tenant id's +// queue. Both streams known means every consumer holds the queue too: apply +// joins the consumers to a queue it opens before this records it, and a +// consumer registered later joins every ingest stream there is. Under e.mu +// (or before e is shared). +func (e *EmbeddedNATS) record(id tenant.ID, q *tenantQueue) { + if q.ingest && q.dlq { + e.opened.Store(id, struct{}{}) + } else { + e.opened.Delete(id) + } +} + // ingestStreamConfig is tenant id's ingest stream. LimitsPolicy: standard // append-only log; the Active Sweeper handles message purging. MaxBytes caps // the tenant's share of the disk. DiscardNew rejects new messages when full, @@ -353,6 +372,7 @@ func (e *EmbeddedNATS) SetMaxBytes(ctx context.Context, id tenant.ID, maxBytes i // apply brings tenant id's queue to maxBytes: opening it when its ingest // stream is missing, resizing it otherwise (see SetMaxBytes). Under e.mu. func (e *EmbeddedNATS) apply(ctx context.Context, id tenant.ID, q *tenantQueue, maxBytes int64) error { + defer e.record(id, q) resizeCtx, cancel := context.WithTimeout(ctx, resizeTimeout) defer cancel() if !q.ingest { @@ -426,9 +446,10 @@ func (e *EmbeddedNATS) applyDLQ(ctx context.Context, id tenant.ID, q *tenantQueu } // reopen opens tenant id's queue at the budget last asked for it, for a -// publish or park that found one of its streams missing. errNoQueue when no -// budget has been asked for the tenant yet: a reload can make a tenant -// resolvable an instant before its budget arrives. +// publish that finds the queue not recorded open, or a publish or park that +// found one of its streams missing. errNoQueue when no budget has been asked +// for the tenant yet: a reload can make a tenant resolvable an instant before +// its budget arrives. // // It runs detached from ctx's cancellation, bounded by its own timeouts: // ctx is one caller's — an ingest request — while the queue is every @@ -442,6 +463,7 @@ func (e *EmbeddedNATS) reopen(ctx context.Context, id tenant.ID) error { if q == nil || q.asked == 0 { return fmt.Errorf("tenant %s: %w", id, errNoQueue) } + defer e.record(id, q) // What is missing is asked of JetStream rather than read off the flags, // which may still say the stream the publish just missed exists — or it // may be back already, opened by a caller that held mu first. @@ -467,15 +489,22 @@ func (e *EmbeddedNATS) reopen(ctx context.Context, id tenant.ID) error { // Publish stores data on topic's ingest subject, in its tenant's queue. A // topic without a valid tenant is refused before anything is sent (see // subject). A tenant with no queue has one opened at the budget last asked -// for it (see SetMaxBytes). A queue that cannot be opened — none asked for -// yet, or JetStream refused it — and a queue at its byte budget (DiscardNew) -// are reported as ErrQueueFull: either way the tenant's queue takes nothing -// now, and a retry is the caller's answer. +// for it (see SetMaxBytes) — and so does one whose stream exists but whose +// queue the broker has not recorded open, since no consumer may hold that +// stream. A queue that cannot be opened — none asked for yet, or JetStream +// refused it — and a queue at its byte budget (DiscardNew) are reported as +// ErrQueueFull: either way the tenant's queue takes nothing now, and a retry +// is the caller's answer. func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, opts ...PublishOpt) error { subj, err := subject(ingestPrefix, topic) if err != nil { return err } + if _, ok := e.opened.Load(topic.Tenant); !ok { + if openErr := e.reopen(ctx, topic.Tenant); openErr != nil { + return fmt.Errorf("%w: %w", ErrQueueFull, openErr) + } + } err = e.publish(ctx, subj, data, opts) if errors.Is(err, jetstream.ErrNoStreamResponse) { if openErr := e.reopen(ctx, topic.Tenant); openErr != nil { diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index f22fb7b0..edb99d97 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -458,6 +458,48 @@ func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) } +// An open that gives up on the ingest stream can leave one behind that +// JetStream goes on to create — in-process, a call fails by timing out — and +// no consumer holds it. A publish goes by the broker's record of the queue, +// not by the stream answering: it opens the queue properly first, consumers +// joined, so its row reaches them rather than a stream nobody reads. +func TestEmbeddedNATS_Publish_OpensAQueueItsOpenGaveUpOn(t *testing.T) { + dir := t.TempDir() + block := filepath.Join(dir, "jetstream", "$G", "streams", ingestStreamName("acme")) + require.NoError(t, os.MkdirAll(filepath.Dir(block), 0o750)) + require.NoError(t, os.WriteFile(block, nil, 0o600)) + e := openEmbedded(t, dir) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + cons, err := e.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer"}) + require.NoError(t, err) + got := make(chan string, 1) + stop, _, err := cons.Consume(func(msg *Message) { + got <- string(msg.Data) + _ = msg.Ack() + }, 10) + require.NoError(t, err) + defer stop() + + require.Error(t, e.SetMaxBytes(ctx, "acme", testBudget), "the ingest stream cannot open") + if err := os.Remove(block); err != nil { + require.ErrorIs(t, err, os.ErrNotExist) + } + // JetStream creates it after all, behind the broker's back. + _, err = e.js.CreateStream(ctx, ingestStreamConfig("acme", testBudget)) + require.NoError(t, err) + + require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"))) + select { + case data := <-got: + assert.Equal(t, "x", data) + case <-ctx.Done(): + t.Fatal("the row reached no consumer") + } + assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) +} + // A resize whose dead-letter update fails undoes the ingest one, back to the // cap the ingest stream had. That is not the budget applied in full: a boot // that found the pair split applied none, and a cap of 0 would leave the From 7cdb794d0e946d6f6edc83e1e9142f0191f4f9fe Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 22:02:13 -0400 Subject: [PATCH 09/15] docs(mq): a consumer that cannot join a queue opened at runtime; review fixes --- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/durability.md | 3 ++- docs/src/content/docs/settings-directory.mdx | 2 +- internal/mq/embedded.go | 17 +++++++++-------- 4 files changed, 13 insertions(+), 11 deletions(-) diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 9da998d1..c91e7303 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -147,7 +147,7 @@ The **only** package that imports NATS/JetStream — a `depguard` rule in `.gola - **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. -- **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts the sweep. It runs on each tenant's stream at that tenant's cutoff. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. +- **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts that tenant's purge. It runs on each tenant's stream at that tenant's cutoff, and a failure on one tenant's stream is reported without stopping the sweep of the others. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. - **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. ### `observability/` — OpenTelemetry Pipeline diff --git a/docs/src/content/docs/durability.md b/docs/src/content/docs/durability.md index 4f7cc1b7..8e57d823 100644 --- a/docs/src/content/docs/durability.md +++ b/docs/src/content/docs/durability.md @@ -33,7 +33,7 @@ WaveHouse does not currently expose a knob to relax this — `SyncAlways` is alw Because the publish blocks on `fsync`, **your typical ingest latency is your storage's typical `fsync` latency, and your worst-case publish is your storage's worst-case `fsync`.** When that tail is healthy (sub-millisecond to single-digit milliseconds) the guarantee is essentially free. When it is not, the same code path that handles every production message stalls: - Publishes block for the duration of the `fsync`, so a multi-second `fsync` tail is a multi-second ingest tail. -- The embedded server's consumer setup and every publish run under the JetStream client's request timeout, and opening or resizing a tenant's queue under a ten-second budget of WaveHouse's own; a slow-enough substrate makes them exceed it. The symptom when a tenant's queue first opens — at the boot or reload that first serves the tenant — is `open dlq stream: ... context deadline exceeded`, or `open ingest stream: ...` (the two share the budget); a boot that finds every queue already at its budget writes nothing, so there the first publish is where it shows. +- The embedded server's consumer setup at boot and every publish run under the JetStream client's request timeout, and opening or resizing a tenant's queue — and joining the consumers to one that opens while the server runs — under ten-second budgets of WaveHouse's own; a slow-enough substrate makes them exceed it. The symptom when a tenant's queue first opens — at the boot or reload that first serves the tenant — is `open dlq stream: ... context deadline exceeded`, or `open ingest stream: ...` (the two share the budget); a boot that finds every queue already at its budget writes nothing, so there the first publish is where it shows. - If the worker cannot drain to ClickHouse faster than producers publish, a tenant's stream fills toward its [`mq.max_bytes_gb`](/settings-directory#message-queue) and the API returns `503` to that tenant ([backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs)). ## Where `SyncAlways` is cheap vs. expensive @@ -94,6 +94,7 @@ A self-contained `wavehouse storage-check` preflight subcommand that bakes this If you see any of these, benchmark the `/nats` volume as above: - `open dlq stream: ... context deadline exceeded`, or `open ingest stream: ...`, when a tenant's queue first opens, at the boot or reload that first serves the tenant. +- `ingest consumer delivery ended; ingestion has stopped` with `join its queue: ... context deadline exceeded`, and the process exiting, when a tenant's queue opens while the server runs and the ingest worker's consumer cannot join it in time; the stream hub's consumer failing the same way logs `a tenant's events do not reach this consumer until the next boot` instead. - Ingest p99 latency in the seconds, or occasional `200`s that take multiple seconds to return. - Intermittent `503 Service Unavailable` from `/v1/ingest` when ClickHouse is healthy (the worker can't drain fast enough because acking is `fsync`-bound). - Flaky CI or load tests that pass on fast storage and fail on a shared/virtualized host. diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index d34c0a65..15f4d24e 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -221,7 +221,7 @@ A tenant's dead-letter stream is opened when the tenant is first served (an empt ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. A queue that opens while the server runs but that a consumer cannot join is different: if the ingest worker's cannot, the process exits with the error, and its restart joins the queue at boot; if the stream hub's cannot, that is logged (`a tenant's events do not reach this consumer until the next boot`), and the tenant's streams get no live rows, gap-fill aside, until a restart. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. **Sizing the volume.** Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume — counting every tenant ever served on it, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream (up to a tenth of its last budget). A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index bebcfa63..aa102b85 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -66,10 +66,11 @@ type EmbeddedNATS struct { // consumers are the durable consumers held on every tenant's queue, each // joined to a queue as it opens. consumers []*fanIn - // opened holds the tenants whose queue has both streams and every - // registered consumer joined — what Publish trusts, rather than a stream - // answering: an open that gave up can leave behind a stream JetStream goes - // on to create, which no consumer holds. Written under mu, read without it. + // opened holds the tenants whose queue has both streams, every registered + // consumer joined to it or told it could not be (fanIn.fail) — what + // Publish trusts, rather than a stream answering: an open that gave up can + // leave behind a stream JetStream goes on to create, which no consumer + // holds. Written under mu, read without it. opened sync.Map // tenant.ID → struct{} } @@ -274,10 +275,10 @@ func (e *EmbeddedNATS) ingestTenants() []tenant.ID { } // record brings opened in line with what the broker knows of tenant id's -// queue. Both streams known means every consumer holds the queue too: apply -// joins the consumers to a queue it opens before this records it, and a -// consumer registered later joins every ingest stream there is. Under e.mu -// (or before e is shared). +// queue. Both streams known means every consumer has been joined to the +// queue too, or told it could not be: apply joins the consumers to a queue it +// opens before this records it, and a consumer registered later joins every +// ingest stream there is. Under e.mu (or before e is shared). func (e *EmbeddedNATS) record(id tenant.ID, q *tenantQueue) { if q.ingest && q.dlq { e.opened.Store(id, struct{}{}) From e199a035f28dea5b23d2809fdd6f4a1639b4cc07 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 22:31:47 -0400 Subject: [PATCH 10/15] fix(mq): pace publish-side retries of a queue that cannot open; review fixes --- CHANGELOG.md | 2 +- docs/src/content/docs/architecture.md | 2 +- docs/src/content/docs/ingest-pipeline.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- internal/app/wire.go | 7 ++- internal/mq/embedded.go | 56 ++++++++++++++++-- internal/mq/embedded_test.go | 60 +++++++++++++++++++- 7 files changed, 117 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c24d0df9..27cbd578 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,7 +32,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Changed -- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it, and so is the tracking of tenant `0`'s last adopted store: a flat directory's ops gate, its one reader left, reads tenant `0` through the registry). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each publish and reload trying again. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. +- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it, and so is the tracking of tenant `0`'s last adopted store: a flat directory's ops gate, its one reader left, reads tenant `0` through the registry). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each reload trying again, and a publish at most once every five seconds, so its clients retrying never hold up another tenant's queue. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. - **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index c91e7303..fb6f03fc 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -148,7 +148,7 @@ The **only** package that imports NATS/JetStream — a `depguard` rule in `.gola - **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. - **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts that tenant's purge. It runs on each tenant's stream at that tenant's cutoff, and a failure on one tenant's stream is reported without stopping the sweep of the others. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. -- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. +- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. Publishes that find the same queue not open share one attempt (`singleflight`), and after one fails the tenant's publishes are refused at once for five seconds rather than each trying again under the broker's lock, which every tenant's open, resize and reload takes; a reload retries regardless. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. ### `observability/` — OpenTelemetry Pipeline diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index 448fa6a0..f314ca49 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -202,7 +202,7 @@ Messages still sitting in `msgChan` or the consumer's prefetch buffer at shutdow ### When the consumer dies -Delivery can end underneath a running worker: the durable consumer is deleted, or the MQ connection closes. The broker client reports that only through an asynchronous error callback and then stops delivering — no message ever arrives to say so, so a loop that only watches `msgChan` would wait forever while the API kept accepting events nothing writes. `mq.Consumer.Consume` therefore returns a `failed` channel next to `stop` (`mq.ErrDeliveryEnded`, wrapping the broker's reason), and `dispatchLoop` selects on it beside `ctx.Done()` and `msgChan`. On a failure it runs the same bottom-up drain as a shutdown — the rows already in hand are flushed and acked, not abandoned — and then reports the error on the worker's own `failed` channel. A consumer that cannot start at all takes the same path. +Delivery can end underneath a running worker: the durable consumer is deleted, the MQ connection closes, or a tenant's queue opened while the server runs cannot be joined. The broker client reports the first two only through an asynchronous error callback and then stops delivering, and `internal/mq` reports the third when it opens the queue — no message ever arrives to say so, so a loop that only watches `msgChan` would wait forever while the API kept accepting events nothing writes. `mq.Consumer.Consume` therefore returns a `failed` channel next to `stop` (`mq.ErrDeliveryEnded`, wrapping the broker's reason), and `dispatchLoop` selects on it beside `ctx.Done()` and `msgChan`. On a failure it runs the same bottom-up drain as a shutdown — the rows already in hand are flushed and acked, not abandoned — and then reports the error on the worker's own `failed` channel. A consumer that cannot start at all takes the same path. The worker does not try to revive the consumer. The app's ingest-worker component returns the error from `app.Run`, which stops every other component and exits non-zero, the same way any failed component does; the supervisor's restart recreates the durable consumer at boot, and everything unacked is redelivered (at-least-once). Passing conditions the client also reports through that callback (a missed heartbeat, a leadership change) are logged at `WARN` and do not end the worker. With the embedded broker (`DontListen`, no external client that could delete a durable) this path is hard to reach; the likeliest way in is a tenant's queue, opened at runtime, that the consumer cannot join. It matters more once a remote broker exists. diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 15f4d24e..c1580690 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -221,7 +221,7 @@ A tenant's dead-letter stream is opened when the tenant is first served (an empt ## Message Queue -- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each publish and each reload trying the queue again — while every other tenant carries on. A queue that opens while the server runs but that a consumer cannot join is different: if the ingest worker's cannot, the process exits with the error, and its restart joins the queue at boot; if the stream hub's cannot, that is logged (`a tenant's events do not reach this consumer until the next boot`), and the tenant's streams get no live rows, gap-fill aside, until a restart. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. +- `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each reload trying the queue again, and so does a publish, at most once every five seconds — while every other tenant carries on. A queue that opens while the server runs but that a consumer cannot join is different: if the ingest worker's cannot, the process exits with the error, and its restart joins the queue at boot; if the stream hub's cannot, that is logged (`a tenant's events do not reach this consumer until the next boot`), and the tenant's `GET /v1/stream` connections get no live rows, gap-fill aside, until a restart. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. **Sizing the volume.** Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume — counting every tenant ever served on it, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream (up to a tenth of its last budget). A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. diff --git a/internal/app/wire.go b/internal/app/wire.go index 968d5f34..ac494bea 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -530,9 +530,10 @@ func (a *App) wireDedupe() error { // refuses boot, like every other store, and on a reload logs it, keeping the // previous budget; a nested directory logs it at boot too, so it never costs // the process — the tenant's ingest answers 503 until its queue opens, each -// publish and each reload trying again. The hook is registered before the -// boot apply, as the dedupe one is. The boot apply runs on ctx, New's, so a -// stop signaled during a boot that opens many queues is not held up by them. +// reload trying again, and publishes too at the pace the MQ allows. The hook +// is registered before the boot apply, as the dedupe one is. The boot apply +// runs on ctx, New's, so a stop signaled during a boot that opens many queues +// is not held up by them. func (a *App) wireMQ(ctx context.Context) error { dir := filepath.Join(a.cfg.DataDir, "nats") config.WarnIfFreshDataDir("nats", dir) diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index aa102b85..32d5047d 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -18,6 +18,7 @@ import ( natsserver "github.com/nats-io/nats-server/v2/server" "github.com/nats-io/nats.go" "github.com/nats-io/nats.go/jetstream" + "golang.org/x/sync/singleflight" ) // slogNATSLogger adapts the default slog logger to the natsserver.Logger @@ -72,6 +73,20 @@ type EmbeddedNATS struct { // leave behind a stream JetStream goes on to create, which no consumer // holds. Written under mu, read without it. opened sync.Map // tenant.ID → struct{} + // reopening merges into one attempt the publishes that find the same + // tenant's queue not open, and failedOpen holds, for a tenant whose last + // such attempt failed, its error and until when its publishes take that + // as their answer (openForPublish). + reopening singleflight.Group + failedOpen sync.Map // tenant.ID → openFailure +} + +// openFailure is a publish's failed attempt to open a tenant's queue, and +// until when the tenant's publishes are refused with its error rather than +// trying again. +type openFailure struct { + until time.Time + err error } // tenantQueue is what the broker knows of one tenant's queue. @@ -111,6 +126,9 @@ const ( // resizeTimeouts when it opens a queue: the consumers join on a budget of // their own (apply). rollbackTimeout = 5 * time.Second + // publishRetry is how long a tenant's publishes are refused at once after + // one failed to open its queue (openForPublish). + publishRetry = 5 * time.Second ) // errNoQueue is why a publish or park finds no queue it can open: no budget @@ -282,6 +300,7 @@ func (e *EmbeddedNATS) ingestTenants() []tenant.ID { func (e *EmbeddedNATS) record(id tenant.ID, q *tenantQueue) { if q.ingest && q.dlq { e.opened.Store(id, struct{}{}) + e.failedOpen.Delete(id) } else { e.opened.Delete(id) } @@ -492,23 +511,24 @@ func (e *EmbeddedNATS) reopen(ctx context.Context, id tenant.ID) error { // subject). A tenant with no queue has one opened at the budget last asked // for it (see SetMaxBytes) — and so does one whose stream exists but whose // queue the broker has not recorded open, since no consumer may hold that -// stream. A queue that cannot be opened — none asked for yet, or JetStream -// refused it — and a queue at its byte budget (DiscardNew) are reported as -// ErrQueueFull: either way the tenant's queue takes nothing now, and a retry -// is the caller's answer. +// stream (see openForPublish for how often a publish tries). A queue that +// cannot be opened — none asked for yet, or JetStream refused it — and a +// queue at its byte budget (DiscardNew) are reported as ErrQueueFull: either +// way the tenant's queue takes nothing now, and a retry is the caller's +// answer. func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, opts ...PublishOpt) error { subj, err := subject(ingestPrefix, topic) if err != nil { return err } if _, ok := e.opened.Load(topic.Tenant); !ok { - if openErr := e.reopen(ctx, topic.Tenant); openErr != nil { + if openErr := e.openForPublish(ctx, topic.Tenant); openErr != nil { return fmt.Errorf("%w: %w", ErrQueueFull, openErr) } } err = e.publish(ctx, subj, data, opts) if errors.Is(err, jetstream.ErrNoStreamResponse) { - if openErr := e.reopen(ctx, topic.Tenant); openErr != nil { + if openErr := e.openForPublish(ctx, topic.Tenant); openErr != nil { return fmt.Errorf("%w: %w", ErrQueueFull, openErr) } err = e.publish(ctx, subj, data, opts) @@ -521,6 +541,30 @@ func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, op return err } +// openForPublish opens tenant id's queue for a publish that found it not open +// (reopen). The publishes that find it so at the same time share one +// attempt, and after an attempt fails the tenant's publishes get its error at +// once, without taking mu, until publishRetry has passed: under clients +// retrying, a queue that cannot open would otherwise hold mu for attempt +// after attempt, and every other tenant's open, resize and reload waits on +// mu. A reload that applies the tenant's budget retries it regardless +// (SetMaxBytes). +func (e *EmbeddedNATS) openForPublish(ctx context.Context, id tenant.ID) error { + if v, ok := e.failedOpen.Load(id); ok { + if f := v.(openFailure); time.Now().Before(f.until) { + return f.err + } + } + _, err, _ := e.reopening.Do(string(id), func() (any, error) { + err := e.reopen(ctx, id) + if err != nil { + e.failedOpen.Store(id, openFailure{until: time.Now().Add(publishRetry), err: err}) + } + return nil, err + }) + return err +} + // DeadLetter stores msg's data on its topic's dead-letter subject, in its // tenant's queue — the subject it arrived on with the ingest prefix swapped // for the dead-letter one, nothing decoded or re-encoded. The dead-letter diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index edb99d97..b27d1a05 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -424,7 +424,8 @@ func TestEmbeddedNATS_SetMaxBytes_IngestFailureChangesNothing(t *testing.T) { // errors and applies no budget, and a publish is refused as a full queue, // while every other tenant's queue opens after it (which a store limit at // the very top of the int64 range would refuse: see NewEmbedded). Once the -// cause is gone, a publish opens the queue at the budget last asked for it. +// cause is gone, a reload opens the queue at the budget last asked for it, +// however recently a publish tried. func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { dir := t.TempDir() // The dead-letter stream is the first of the pair to open. A failed open @@ -454,10 +455,67 @@ func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { if err := os.Remove(block); err != nil { require.ErrorIs(t, err, os.ErrNotExist) } + require.NoError(t, e.SetMaxBytes(ctx, "acme", testBudget)) require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"))) assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) } +// After a publish fails to open its tenant's queue, the tenant's publishes +// are refused at once, without waiting on the broker's lock, until +// publishRetry has passed: under clients retrying, one tenant's broken queue +// would otherwise hold the lock that every other tenant's open, resize and +// reload takes. Once the window has passed, a publish tries again. +func TestEmbeddedNATS_Publish_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T) { + dir := t.TempDir() + block := filepath.Join(dir, "jetstream", "$G", "streams", dlqStreamName("acme")) + obstruct := func() { + t.Helper() + require.NoError(t, os.MkdirAll(filepath.Dir(block), 0o750)) + require.NoError(t, os.WriteFile(block, nil, 0o600)) + } + obstruct() + e := openEmbedded(t, dir) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + acme := Topic{Tenant: "acme", Table: "t"} + + require.Error(t, e.SetMaxBytes(ctx, "acme", testBudget)) + obstruct() + require.ErrorIs(t, e.Publish(ctx, acme, []byte("x")), ErrQueueFull, "the publish's own attempt fails") + if err := os.Remove(block); err != nil { + require.ErrorIs(t, err, os.ErrNotExist) + } + + // The queue could open now, but within the window a publish tries + // nothing: it is refused while the lock is held elsewhere. + e.mu.Lock() + var paced error + done := make(chan struct{}) + go func() { + defer close(done) + paced = e.Publish(ctx, acme, []byte("x")) + }() + var returned bool + select { + case <-done: + returned = true + case <-time.After(2 * time.Second): + } + e.mu.Unlock() + <-done + require.True(t, returned, "a paced publish waited on the broker's lock") + require.ErrorIs(t, paced, ErrQueueFull) + assert.Zero(t, e.MaxBytes("acme")) + + v, ok := e.failedOpen.Load(tenant.ID("acme")) + require.True(t, ok) + failed := v.(openFailure) + failed.until = time.Now() + e.failedOpen.Store(tenant.ID("acme"), failed) + require.NoError(t, e.Publish(ctx, acme, []byte("x")), "once the window has passed") + assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) +} + // An open that gives up on the ingest stream can leave one behind that // JetStream goes on to create — in-process, a call fails by timing out — and // no consumer holds it. A publish goes by the broker's record of the queue, From 55137d8ec23ff7278323f119a40ef9e9ec17f9e0 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 22:55:33 -0400 Subject: [PATCH 11/15] test(mq): one tenant's failed purge stops no other; review fixes --- docs/src/content/docs/sdk/streaming.md | 2 +- internal/mq/embedded_test.go | 110 +++++++++++++++++++++++++ 2 files changed, 111 insertions(+), 1 deletion(-) diff --git a/docs/src/content/docs/sdk/streaming.md b/docs/src/content/docs/sdk/streaming.md index 94a7e334..27ee3a95 100644 --- a/docs/src/content/docs/sdk/streaming.md +++ b/docs/src/content/docs/sdk/streaming.md @@ -116,7 +116,7 @@ A dropped stream reconnects on a jittered exponential backoff, capped at 30s, an :::caution[Resumption is at-least-once, and time-bounded] Delivery across a reconnect is **at-least-once**. The `Last-Event-ID` the client sends is the last event's `received_timestamp`, and the server replays from that instant *inclusively* — so the last event you already saw, and anything sharing its timestamp, arrives again. The SDK does not deduplicate live frames — `liveQuery()` makes one pass at the backfill seam, and only under an ascending order ([#449](https://github.com/Wave-RF/WaveHouse/issues/449)) — so key on `timestamp` plus your own row identity if duplicates matter. -Replay is also bounded by the server's [`stream.gap_window_minutes`](/settings-directory#streaming) — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. The same silence applies across a server upgrade to this release: the server deletes the previous release's queue at boot, so a replay spanning the upgrade omits the events published before it, without an error — backfill over REST if you need them. +Replay is also bounded by the [`stream.gap_window_minutes`](/settings-directory#streaming) of the tenant you stream from — 15 minutes by default. A drop longer than that resumes with a hole and no signal, because the purged messages are simply gone. The same silence applies to a replay spanning the upgrade across the v2 ingest envelope, whose boot deletes the earlier build's queue — see [Upgrading across the v2 ingest envelope](/deployment#upgrading-across-the-v2-ingest-envelope); backfill over REST if you need those events. **A column-set change across a gap-fill is a known limitation.** If the table's columns change while you are connected *and* your client replays across that change, live rows arriving after the replay may not be preceded by a fresh `event: schema` frame until the columns next change or you reconnect. The SDK drops a row whose **length** disagrees with the list it was last told, rather than zipping it under the wrong names — so an added or removed column costs you rows, not wrong ones. A **same-length** change is the residual case the arity check cannot see: a `RENAME COLUMN`, or a drop paired with an add, zips values under the wrong names until the next announcement. Reconnecting resynchronizes either way. Full schema-change handling is deferred to the schema-versioning work ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)). ::: diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index b27d1a05..85fd5719 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -1000,6 +1000,56 @@ func TestEmbeddedNATS_PurgeAcked_EachTenantAtItsOwnCutoff(t *testing.T) { assert.Zero(t, msgs("initech"), "a tenant the cutoffs do not name keeps nothing it has acknowledged") } +// One tenant's purge failing stops no other tenant's: the errors say which +// failed, and the sweep goes on to the next tenant at its own cutoff — here +// after one whose durable is gone and one whose stream is. A sweep whose +// context has already ended touches no tenant. +func TestEmbeddedNATS_PurgeAcked_OneTenantsFailureStopsNoOther(t *testing.T) { + ids := []tenant.ID{"acme", "globex", "initech", "umbrella"} + e := newTestEmbedded(t, ids...) + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + + for _, id := range ids { + for i := range 2 { + require.NoError(t, e.Publish(ctx, Topic{Tenant: id, Table: "p"}, []byte{byte(i)})) + } + } + ackAll(t, e, "buffer", 8) + for _, id := range ids { + s, err := e.stream(ctx, ingestStreamName(id)) + require.NoError(t, err) + require.Eventually(t, func() bool { + floor, err := s.consumerAckFloor(ctx, "buffer") + return err == nil && floor == 2 + }, 5*time.Second, 20*time.Millisecond, id) + } + msgs := func(id tenant.ID) uint64 { + s, err := e.stream(ctx, ingestStreamName(id)) + require.NoError(t, err) + st, err := s.state(ctx, "") + require.NoError(t, err) + return st.Msgs + } + + ended, end := context.WithCancel(ctx) + end() + purged, err := e.PurgeAcked(ended, "buffer", nil) + require.ErrorIs(t, err, context.Canceled) + assert.False(t, purged) + assert.Equal(t, uint64(2), msgs("umbrella"), "a sweep whose context has ended touches nothing") + + require.NoError(t, e.js.DeleteConsumer(ctx, ingestStreamName("acme"), "buffer")) + require.NoError(t, e.js.DeleteStream(ctx, ingestStreamName("globex"))) + purged, err = e.PurgeAcked(ctx, "buffer", map[tenant.ID]time.Time{"initech": time.Now().Add(-time.Hour)}) + require.ErrorIs(t, err, ErrConsumerNotFound, "acme's durable is gone") + require.ErrorContains(t, err, "tenant globex: get stream") + assert.True(t, purged, "the tenants after them are purged all the same") + assert.Equal(t, uint64(2), msgs("acme")) + assert.Equal(t, uint64(2), msgs("initech"), "kept for its own window") + assert.Zero(t, msgs("umbrella")) +} + // The isolation per-tenant queues buy: a tenant at MaxAckPending, or one // whose handler is stuck, holds back its own delivery and no other tenant's — // each tenant's messages arrive on a delivery of their own, in order. @@ -1328,3 +1378,63 @@ func TestNewEmbedded_TakesStockOfTheQueuesOnDisk(t *testing.T) { require.NoError(t, e.Publish(ctx, Topic{Tenant: "acme", Table: "t"}, []byte("x"))) assert.Equal(t, int64(8<<20), streamConfig(t, e, "INGEST_acme").MaxBytes) } + +// A durable found on disk is kept as it stands when it holds the settings +// asked for — a boot over many queues writes nothing it need not — and is +// updated in place when they differ; either way delivery resumes past what it +// acknowledged before the restart. +func TestEmbeddedNATS_ADurableOnDiskIsReusedAcrossARestart(t *testing.T) { + for _, tt := range []struct { + name string + maxAckPending int + }{ + {"same settings", 10}, + {"other settings", 20}, + } { + t.Run(tt.name, func(t *testing.T) { + dir := t.TempDir() + ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) + defer cancel() + topic := Topic{Tenant: "acme", Table: "t"} + + first, err := NewEmbedded(dir) + require.NoError(t, err) + require.NoError(t, first.SetMaxBytes(ctx, "acme", 8<<20)) + require.NoError(t, first.Publish(ctx, topic, []byte{0})) + cons, err := first.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer", MaxAckPending: 10}) + require.NoError(t, err) + acked := make(chan error, 2) + stop, _, err := cons.Consume(func(msg *Message) { acked <- msg.DoubleAck(ctx) }, 1) + require.NoError(t, err) + select { + case err := <-acked: + require.NoError(t, err) + case <-ctx.Done(): + t.Fatal("the first row was not delivered") + } + stop() + require.NoError(t, first.Close()) + + e := openEmbedded(t, dir) + require.NoError(t, e.Publish(ctx, topic, []byte{1})) + cons, err = e.CreateConsumer(ctx, ConsumerConfig{Durable: "buffer", MaxAckPending: tt.maxAckPending}) + require.NoError(t, err) + got := make(chan byte, 2) + stop, _, err = cons.Consume(func(msg *Message) { + _ = msg.Ack() + got <- msg.Data[0] + }, 4) + require.NoError(t, err) + t.Cleanup(stop) + select { + case b := <-got: + assert.Equal(t, byte(1), b, "delivery resumes past what was acknowledged before the restart") + case <-ctx.Done(): + t.Fatal("the row published after the restart was not delivered") + } + c, err := e.js.Consumer(ctx, ingestStreamName("acme"), "buffer") + require.NoError(t, err) + assert.Equal(t, tt.maxAckPending, c.CachedInfo().Config.MaxAckPending) + }) + } +} From 060ca9ea040984864da01ec553a1433109208523 Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 23:17:19 -0400 Subject: [PATCH 12/15] docs(mq): size a dead-letter stream the shrink guard kept; review fixes --- docs/src/content/docs/api.md | 2 +- docs/src/content/docs/settings-directory.mdx | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index df65cfe4..6fc71291 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -745,7 +745,7 @@ Triggers an immediate re-discovery of the `?tenant=`'s ClickHouse table schemas #### `GET /v1/ops/dlq/stats` — DLQ Statistics -Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The queue is read from the message queue rather than the settings, so a tenant whose folder was rejected or removed is read like one being served, since its queue is kept (nothing deletes it). The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream is opened when the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. +Returns per-table message counts in one tenant's Dead Letter Queue: the [tenant](/deployment#the-nested-settings-directory) an optional `?tenant=` names, the default tenant `0` without it, which is the whole settings directory unless it is nested. The tenant is looked up in the message queue, not the settings, so a tenant whose folder was rejected or removed is read like one being served, since its queue is kept (nothing deletes it). The query string is parsed strictly, as on the other admin reads. Admin-only, like the rest of this section. Whether a poison row lands here is the settings directory's [`dlq.enabled`](/settings-directory#dead-letter-queue) switch (global or per table); a tenant's dead-letter stream is opened when the tenant is first served, and this endpoint always exists. Before any failure has ever occurred, the endpoint returns `200` with `{"tables":{},"total":0}`. **Error responses:** diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index c1580690..93bfdd6a 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -223,7 +223,7 @@ A tenant's dead-letter stream is opened when the tenant is first served (an empt - `mq.max_bytes_gb` (seed default `50`) — disk budget for the tenant's embedded JetStream ingest stream (`INGEST_{tenant}`), which buffers its ingested events until the worker writes them to ClickHouse; its dead-letter stream (`DLQ_{tenant}`) gets a tenth of it. Each tenant's pair of streams is its own, opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed. The ingest stream runs `DiscardNew`, so when it's full new publishes are rejected and `POST /v1/ingest` returns `503` for that tenant alone — [backpressure by construction](/ingest-pipeline#backpressure-and-durability-knobs). A reload updates both streams' limits in place without touching what's buffered: growing takes effect immediately; shrinking below what's currently on disk makes the ingest stream refuse new publishes until the Active Sweeper purges it back under the limit — what it purges is what is both written to ClickHouse and past the tenant's `stream.gap_window_minutes`, and nothing already accepted is dropped — and a dead-letter stream holding more than a tenth of the new budget is kept at what it holds rather than shrunk, since shrinking it would delete its oldest parked rows; that is logged, and the stream then makes room for each new row by dropping its oldest, as a full one always does. If NATS rejects the update, the rest of the reload is still adopted, the failure is logged, and the next reload retries it. A queue NATS will not open at all refuses boot, like every other store; over [a nested directory](/deployment#the-nested-settings-directory) it costs that tenant alone, at boot or on reload — its ingest answers `503`, each reload trying the queue again, and so does a publish, at most once every five seconds — while every other tenant carries on. A queue that opens while the server runs but that a consumer cannot join is different: if the ingest worker's cannot, the process exits with the error, and its restart joins the queue at boot; if the stream hub's cannot, that is logged (`a tenant's events do not reach this consumer until the next boot`), and the tenant's `GET /v1/stream` connections get no live rows, gap-fill aside, until a restart. The two streams are resized as a pair: a failed DLQ resize undoes the ingest one so both stay on the previous budget, but if that undo fails too the ingest stream keeps the new limit and the DLQ the previous one until a later reload succeeds — the log line says which happened. -**Sizing the volume.** Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep the tenants' budgets, plus a tenth of each for their dead-letter streams, within the free space of the `/nats` volume — counting every tenant ever served on it, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream (up to a tenth of its last budget). A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. +**Sizing the volume.** Nothing checks the budget against the disk — neither one tenant's nor what the tenants' add up to ([#138](https://github.com/Wave-RF/WaveHouse/issues/138)) — so keep within the free space of the `/nats` volume every tenant's budget plus its dead-letter stream's cap: a tenth of the budget, or what the stream held when a smaller budget arrived, if that is more. Count every tenant ever served on the volume, not only those served now: a rejected or removed tenant's queue is kept and nothing deletes it, so what it holds goes on holding disk — a rejected tenant's replay history, and the rows parked on either one's dead-letter stream, up to that cap. A disk that fills before a budget does fails every tenant's writes, not one: the failed write is logged, the publish goes unanswered until it times out, and ingest answers `500` (`publish failed`) for every tenant on that volume, not the `503` with `Retry-After` of a full budget. It does not clear on its own: a stream that failed a write refuses every later one until WaveHouse restarts, so free the space and then restart. ## Streaming From 9344f1b8cd33b50dc3534fdb0607b0a201df973a Mon Sep 17 00:00:00 2001 From: taitelee Date: Thu, 24 Sep 2026 23:53:39 -0400 Subject: [PATCH 13/15] fix(mq): pace a park's reopen, and warn only on a missing consumer; review fixes --- CHANGELOG.md | 2 +- docs/src/content/docs/architecture.md | 4 +-- internal/ingest/sweeper.go | 18 +++++++++- internal/ingest/sweeper_test.go | 27 ++++++++++++++ internal/mq/embedded.go | 51 ++++++++++++++------------- internal/mq/embedded_test.go | 23 ++++++------ internal/mq/mq.go | 7 ++-- 7 files changed, 90 insertions(+), 42 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 27cbd578..70ce51e1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -32,7 +32,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Changed -- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it, and so is the tracking of tenant `0`'s last adopted store: a flat directory's ops gate, its one reader left, reads tenant `0` through the registry). A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each reload trying again, and a publish at most once every five seconds, so its clients retrying never hold up another tenant's queue. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. +- **Each tenant has a message queue of its own** (`internal/mq/{mq,subject,embedded}.go` (+ tests), `internal/ingest/{sweeper,worker}.go` (+ tests), `internal/api/{dlq,ingest}.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `internal/stream/{subscriber,hub}.go`, `internal/settings/{settings,store,registry}.go` (+ tests), `internal/testutil/{mocks,testutil}.go`, `clients/ts/src/{dlq,types}.ts` (+ tests), `docs/src/content/docs/{deployment,api,architecture,ingest-pipeline,durability,why-wavehouse}.md`, `docs/src/content/docs/sdk/{admin,reference,streaming}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `AGENTS.md`): story 5b of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). The embedded NATS server keeps each tenant's events on a pair of JetStream streams of its own — `INGEST_` (`ingest..>`, `DiscardNew`) at the tenant's own `mq.max_bytes_gb`, and `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — opened when the tenant is first served and kept, at the budget it last had, when its folder is rejected or removed; subjects are unchanged, and nothing outside `internal/mq` names a stream. A tenant at its budget gets `503` while the others keep publishing, and the ingest worker's and the hub bridge's durables are held on every tenant's stream, each with its own ack floor and `MaxAckPending`, so one tenant's backlog holds back neither another's delivery nor its purge; the worker's prefetch and the hub bridge's fetch-ahead are each shared across the tenants' streams. A removed or rejected tenant's stream is still consumed, so its queued rows reach the worker and are parked on its own dead-letter queue. The sweeper purges each tenant's stream at that tenant's own `stream.gap_window_minutes` — a rejected tenant's as its folder last had it (`settings.Registry.Known` now yields each tenant's last adopted store), and all of its history if the folder has been rejected since boot, so its clients resume once the folder is fixed — keeping no acknowledged history for a removed tenant, and every served tenant's `mq.max_bytes_gb` is applied after each reload rather than tenant `0`'s alone (the boot warning about a nested directory with no tenant `0` is gone with it, and so is the tracking of tenant `0`'s last adopted store: a flat directory's ops gate, its one reader left, reads tenant `0` through the registry). One tenant's failed purge holds up no other tenant's, and the sweep logs it at `ERROR` unless every failure in it is a buffer consumer not created yet. A reload that shrinks a budget no longer deletes dead letters: a dead-letter stream holding more than a tenth of the new budget keeps what it holds, and that is logged — the interim guard of [#532](https://github.com/Wave-RF/WaveHouse/issues/532). JetStream's own check of the streams' caps against the disk, which made them fit 75% of the free disk at boot together, is lifted: a budget is a cap and never a reservation, what the budgets add up to against the disk is [#138](https://github.com/Wave-RF/WaveHouse/issues/138)'s, and a flat directory whose budget exceeds three quarters of the free disk now boots where it used to be refused. A queue that cannot be opened refuses a flat boot like any other store and, over a nested directory, costs its tenant alone — its ingest answers `503`, each reload trying again, and a publish at most once every five seconds, so its clients retrying never hold up another tenant's queue. `GET /v1/ops/dlq/stats` reads one tenant's dead-letter queue — the one `?tenant=` names, parsed strictly like the other admin reads, and tenant `0`'s without it, no longer the sum across tenants — answering for a rejected or removed tenant too and `404` for a tenant with no queue; the SDK's `wh.dlq.list()` and `.table()` take a `tenant` option. The streams an earlier build kept for every tenant together (`WAVEHOUSE`, `WAVEHOUSE_DLQ`) overlap every tenant's subjects and are deleted at boot, what they held with them, and a subject with no tenant token no longer reads as tenant `0`'s. - **Message-queue subjects lead with the tenant, and the async paths read it off each message** (`internal/mq/{mq,subject,embedded}.go`, `internal/api/{ingest,stream}.go`, `internal/stream/hub.go`, `internal/ingest/{worker,sweeper}.go`, `internal/app/wire.go`, `docs/src/content/docs/{architecture,ingest-pipeline,deployment,api}.md`, `docs/src/content/docs/{settings-directory,access-control}.mdx`, `AGENTS.md`): story 5a of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `mq.Topic` gains a `Tenant`, the leading token of every subject — `ingest..
[.]`, `dlq..
[.]` — placed verbatim, since the tenant-id grammar makes it one token, so one wildcard selects a tenant's traffic (`ingest.acme.>`); a settings directory that holds the four files produces the same subjects with `0` as the token, and nothing else about them changes. The ingest and stream handlers address the request's tenant, which the resolved store carries (`settings.Store.Tenant`, from story 8). The stream hub indexes subscribers by the full topic and evaluates each event under its own tenant's policy, so a subscriber on one tenant's table never receives another tenant's rows for a table of the same name; a gap-fill and the opening schema frame read the connection's tenant. The ingest worker reads each message's tenant off its topic, batches per tenant table, and resolves the dead-letter switch under the row's own tenant — an envelope it cannot read is parked or dropped under the topic's tenant too — and bumps that tenant's cache namespaces (story 8's `invalidate` now receives the message's tenant rather than tenant `0`; the wiring's `sharedTables` still repeats each bump under every tenant the directory holds, since every tenant reads the same ClickHouse table until story 6). `Publish` and gap-fill refuse a topic whose tenant is empty or outside the grammar, so nothing lands on tenant `0` by omission. diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index fb6f03fc..fe0c94c9 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -145,10 +145,10 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ The **only** package that imports NATS/JetStream — a `depguard` rule in `.golangci.yml` fails `make lint` on any `github.com/nats-io` import in every package golangci-lint builds; the `integration`-tagged files under `tests/` sit outside its default build context, so the boundary there rests on convention (AGENTS.md Key Design Decision #20). Every other package talks to the broker through the types below, so a subject, stream, or broker change lands here once. -- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; `ErrConsumerNotFound` when the consumer has not been created yet) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. +- **mq.go** — The owned surface, stated as intent rather than broker mechanics. `Topic{Tenant, Table, Scope}` is the only address the rest of the process handles (a validated tenant id and raw names; comparable, so the SSE hub keys its index by the value). `Message` carries `Data`, its topic (`Topic()` decodes the delivered key on demand — the tenant included, which is how the hub bridge and the worker learn whose event it is; `TopicKey()` is the delivered form, for log lines), and the ack family (`DoubleAck(ctx)`, `Ack()`, `Nak()`); `Headers` is the message header map (`Add`/`Set`/`Get`, exact-key) that `PublishOpt`s such as `WithHeader` shape. Interfaces, each speaking per tenant and never per stream: `Publisher` (`ErrQueueFull` when the tenant's ingest queue is at its byte budget or not open yet — the API's 503 + `Retry-After`), `Subscriber` (every ingest event of every tenant, under a named durable consumer, its fetch-ahead split across the tenants — the hub bridge), `ConsumerManager` → `Consumer` (a durable explicit-ack consumer from a `ConsumerConfig`, whose `MaxAckPending` holds per tenant; `Consume` delivers each tenant's messages on a goroutine of that tenant's, in order, so a blocking handler is backpressure on its own tenant alone, spreads the prefetch across the tenants, and returns a `stop` plus a `failed` channel that reports delivery ending on its own — `ErrDeliveryEnded`, e.g. a deleted consumer, a closed connection, or a tenant's queue that could not be joined — since no message would ever say so) for the ingest worker, `DeadLetterer.DeadLetter` (park a message under its own topic, in its tenant's dead-letter queue; the caller acks), `DeadLetterStats.DeadLetterCounts` (one tenant's; `ErrNoDeadLetterQueue` when it has none), `Purger.PurgeAcked` (drop what is both acked by a consumer and stored before its tenant's cutoff, everything acked for a tenant given none; one error per failed tenant, joined — `ErrConsumerNotFound` for a queue the consumer has not been created on yet, the one failure the sweeper logs as a warning rather than an error) for the sweeper, and `Replayer.ReplaySince` for SSE gap-fill. `Broker` composes them with each tenant's byte budget (`SetMaxBytes`/`MaxBytes`), `Stats`, and `Close`; it is what `internal/app` holds. - **subject.go** — The embedded broker's naming, private to the package: the stream names (`INGEST_` and `DLQ_` — prefixes that differ in their first letter, so no tenant id makes one kind's name the other's — and the one pair an earlier build kept for every tenant together, `WAVEHOUSE`/`WAVEHOUSE_DLQ`, which boot deletes), the `ingest.`/`dlq.` prefixes and `>` wildcards, the subject-token encoder (alphanumerics and `_` pass, everything else is percent-encoded, so a name can never split or wildcard a subject), and `Topic` ↔ subject conversion. A subject is `.
[.]`: the tenant verbatim — its grammar (`tenant.Parse`) makes it one token, and it is checked on the way to the wire, so a topic without one has no subject — then the table and scope as encoded tokens; tenant first so one wildcard selects a tenant's traffic (`ingest.acme.>`). A topic has the same tail on both streams, so parking on the DLQ is a prefix swap on the delivered subject — nothing is decoded or re-encoded — and the tail's first token picks the tenant's stream. - **purge.go** — The Active Sweeper's arithmetic over JetStream sequences: purge target = `MIN(consumer ack floor + 1, first sequence stored at or after the cutoff)`, the latter found by binary search over message timestamps (~15 lookups). Every uncertainty resolves toward purging less: a sequence that holds no message is kept as a candidate bound rather than discarding the half below it, and a lookup that fails outright aborts that tenant's purge. It runs on each tenant's stream at that tenant's cutoff, and a failure on one tenant's stream is reported without stopping the sweep of the others. Healthy state keeps exactly the gap window; ClickHouse down freezes purging; a catastrophic outage fills the stream to `MaxBytes` and `DiscardNew` pushes back. -- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. Publishes that find the same queue not open share one attempt (`singleflight`), and after one fails the tenant's publishes are refused at once for five seconds rather than each trying again under the broker's lock, which every tenant's open, resize and reload takes; a reload retries regardless. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. +- **embedded.go** — `EmbeddedNATS`, the one `Broker`: an in-process NATS server with JetStream, giving each tenant a queue of its own — stream `INGEST_` with subjects `ingest..>`, capped at the tenant's `mq.max_bytes_gb` (`DiscardNew`), and stream `DLQ_` (`dlq..>`, `DiscardOld`) at a tenth of it — with the durable consumers on the ingest one; nothing outside the package sees that layout. JetStream's own check of the streams' caps against the disk (75% of the free disk by default) is set out of reach, so a budget is a cap and never a reservation. Boot deletes the pair an earlier build kept for every tenant together — its subjects overlap every tenant's — and takes stock of the tenants' streams on disk with their budgets, so a consumer created later is held on every one, a tenant no longer served included. `SetMaxBytes` opens a tenant's queue the first time — its dead-letter stream first, so no row is queued that could not be parked — and every registered consumer joins it; a publish or park that finds a stream missing reopens it at the budget last asked for the tenant, and so does a publish to a queue the broker has not recorded open — an open that timed out can leave a stream JetStream creates after all, which no consumer holds — either refused as a full queue with none asked yet. Publishes and parks that find the same queue not open share one attempt (`singleflight`), and after one fails the tenant's publishes and parks are refused at once for five seconds rather than each trying again under the broker's lock, which every tenant's open, resize and reload takes; a reload retries regardless. After that `SetMaxBytes` applies a reloaded budget to the tenant's two live streams as a pair: if the DLQ update fails after the ingest one succeeded, the ingest resize is undone so both stay on the previous budget — best effort, since if that undo also fails the ingest stream stays at the new limit and the DLQ at the previous, and the error says so. A dead-letter stream is never capped below the bytes it holds, which `DiscardOld` would delete to fit ([#532](https://github.com/Wave-RF/WaveHouse/issues/532)): it keeps what it holds, and that is logged. Its JetStream calls are bounded to ten seconds — plus ten more for the consumers joining a queue it has just opened, and five for the rollback of a failed resize, each a budget of its own rather than the one that just expired — since a reload holds the settings store's lock while its hooks run; `MaxBytes` reports the budget last applied in full, so a failed resize is retried by the next reload. The consumers `CreateConsumer` and `Subscribe` build hold one durable on each tenant's stream, looked up before anything is written so a boot over many queues writes nothing it need not, each delivering on a goroutine of its own into the one handler. `PurgeAcked` and `DeadLetterCounts` run per tenant stream. `Stats` reports connection and inbound-message counters for `observability.RegisterSystemMetrics`. Trace context rides in the message headers: `Publish` applies `observability.InjectHeaders`, and a message delivered through `Subscribe` (the hub bridge) carries `observability.ExtractHeaders` on its `Ctx`; the worker's `Consumer` path skips the extraction, since it batches across messages and reads no per-message context. ### `observability/` — OpenTelemetry Pipeline diff --git a/internal/ingest/sweeper.go b/internal/ingest/sweeper.go index b0d4a2d1..4024b9de 100644 --- a/internal/ingest/sweeper.go +++ b/internal/ingest/sweeper.go @@ -61,7 +61,7 @@ func (s *Sweeper) sweep(ctx context.Context) { } _, err := s.purger.PurgeAcked(ctx, BufferConsumerName, cutoffs) if err != nil { - if errors.Is(err, mq.ErrConsumerNotFound) { + if onlyConsumerNotFound(err) { // Consumer may not exist yet if no messages have been ingested. slog.WarnContext(ctx, "sweeper: buffer consumer not found (may not exist yet)", "error", err) return @@ -69,3 +69,19 @@ func (s *Sweeper) sweep(ctx context.Context) { slog.ErrorContext(ctx, "sweeper: purge", "error", err) } } + +// onlyConsumerNotFound reports whether every tenant's failure err joins is a +// missing buffer consumer — the one failure expected before the worker has +// created it. Any other failure among them keeps the sweep's report at +// ERROR: a tenant whose purge keeps failing fills toward its budget. +func onlyConsumerNotFound(err error) bool { + if joined, ok := err.(interface{ Unwrap() []error }); ok { + for _, e := range joined.Unwrap() { + if !onlyConsumerNotFound(e) { + return false + } + } + return true + } + return errors.Is(err, mq.ErrConsumerNotFound) +} diff --git a/internal/ingest/sweeper_test.go b/internal/ingest/sweeper_test.go index 9b1eabab..3d585561 100644 --- a/internal/ingest/sweeper_test.go +++ b/internal/ingest/sweeper_test.go @@ -3,12 +3,15 @@ package ingest import ( "context" "errors" + "fmt" + "log/slog" "testing" "time" "github.com/Wave-RF/WaveHouse/internal/mq" "github.com/Wave-RF/WaveHouse/internal/tenant" "github.com/Wave-RF/WaveHouse/internal/testutil" + "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" ) @@ -63,6 +66,30 @@ func TestSweep_ErrorsDoNotPanic(t *testing.T) { } } +// A missing buffer consumer is the expected failure, before the worker has +// created it, and only a warning; any other tenant's failure in the same +// sweep — the purger joins one per tenant — keeps the report at ERROR. +func TestSweep_OnlyAMissingConsumerIsAWarning(t *testing.T) { + missing := fmt.Errorf("tenant acme: %w", mq.ErrConsumerNotFound) + for _, tt := range []struct { + name string + err error + want, not string + }{ + {"a missing consumer", errors.Join(missing), "WARN", "ERROR"}, + {"a missing consumer beside another failure", errors.Join(missing, errors.New("tenant globex: get stream: stream not found")), "ERROR", "WARN"}, + {"another failure", errors.New("broker unavailable"), "ERROR", "WARN"}, + } { + t.Run(tt.name, func(t *testing.T) { + logs := logtest.Capture(t, slog.LevelDebug) + s := NewSweeper(&testutil.MockPurger{Err: tt.err}, func() map[tenant.ID]time.Duration { return nil }) + s.sweep(context.Background()) + assert.Contains(t, logs.String(), `"level":"`+tt.want+`"`) + assert.NotContains(t, logs.String(), `"level":"`+tt.not+`"`) + }) + } +} + // --------------------------------------------------------------------------- // Start() context cancellation test // --------------------------------------------------------------------------- diff --git a/internal/mq/embedded.go b/internal/mq/embedded.go index 32d5047d..830be900 100644 --- a/internal/mq/embedded.go +++ b/internal/mq/embedded.go @@ -73,17 +73,17 @@ type EmbeddedNATS struct { // leave behind a stream JetStream goes on to create, which no consumer // holds. Written under mu, read without it. opened sync.Map // tenant.ID → struct{} - // reopening merges into one attempt the publishes that find the same - // tenant's queue not open, and failedOpen holds, for a tenant whose last - // such attempt failed, its error and until when its publishes take that - // as their answer (openForPublish). + // reopening merges into one attempt the publishes and parks that find + // the same tenant's queue not open, and failedOpen holds, for a tenant + // whose last such attempt failed, its error and until when its publishes + // and parks take that as their answer (reopenPaced). reopening singleflight.Group failedOpen sync.Map // tenant.ID → openFailure } -// openFailure is a publish's failed attempt to open a tenant's queue, and -// until when the tenant's publishes are refused with its error rather than -// trying again. +// openFailure is a publish's or park's failed attempt to open a tenant's +// queue, and until when the tenant's publishes and parks are refused with its +// error rather than trying again. type openFailure struct { until time.Time err error @@ -126,9 +126,9 @@ const ( // resizeTimeouts when it opens a queue: the consumers join on a budget of // their own (apply). rollbackTimeout = 5 * time.Second - // publishRetry is how long a tenant's publishes are refused at once after - // one failed to open its queue (openForPublish). - publishRetry = 5 * time.Second + // reopenRetry is how long a tenant's publishes and parks are refused at + // once after one failed to open its queue (reopenPaced). + reopenRetry = 5 * time.Second ) // errNoQueue is why a publish or park finds no queue it can open: no budget @@ -511,7 +511,7 @@ func (e *EmbeddedNATS) reopen(ctx context.Context, id tenant.ID) error { // subject). A tenant with no queue has one opened at the budget last asked // for it (see SetMaxBytes) — and so does one whose stream exists but whose // queue the broker has not recorded open, since no consumer may hold that -// stream (see openForPublish for how often a publish tries). A queue that +// stream (see reopenPaced for how often a publish tries). A queue that // cannot be opened — none asked for yet, or JetStream refused it — and a // queue at its byte budget (DiscardNew) are reported as ErrQueueFull: either // way the tenant's queue takes nothing now, and a retry is the caller's @@ -522,13 +522,13 @@ func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, op return err } if _, ok := e.opened.Load(topic.Tenant); !ok { - if openErr := e.openForPublish(ctx, topic.Tenant); openErr != nil { + if openErr := e.reopenPaced(ctx, topic.Tenant); openErr != nil { return fmt.Errorf("%w: %w", ErrQueueFull, openErr) } } err = e.publish(ctx, subj, data, opts) if errors.Is(err, jetstream.ErrNoStreamResponse) { - if openErr := e.openForPublish(ctx, topic.Tenant); openErr != nil { + if openErr := e.reopenPaced(ctx, topic.Tenant); openErr != nil { return fmt.Errorf("%w: %w", ErrQueueFull, openErr) } err = e.publish(ctx, subj, data, opts) @@ -541,15 +541,15 @@ func (e *EmbeddedNATS) Publish(ctx context.Context, topic Topic, data []byte, op return err } -// openForPublish opens tenant id's queue for a publish that found it not open -// (reopen). The publishes that find it so at the same time share one -// attempt, and after an attempt fails the tenant's publishes get its error at -// once, without taking mu, until publishRetry has passed: under clients -// retrying, a queue that cannot open would otherwise hold mu for attempt -// after attempt, and every other tenant's open, resize and reload waits on -// mu. A reload that applies the tenant's budget retries it regardless -// (SetMaxBytes). -func (e *EmbeddedNATS) openForPublish(ctx context.Context, id tenant.ID) error { +// reopenPaced opens tenant id's queue for a publish or park that found it not +// open (reopen). The callers that find it so at the same time share one +// attempt, and after an attempt fails the tenant's publishes and parks get +// its error at once, without taking mu, until reopenRetry has passed: under +// clients retrying, or the worker parking row after row, a queue that cannot +// open would otherwise hold mu for attempt after attempt, and every other +// tenant's open, resize and reload waits on mu. A reload that applies the +// tenant's budget retries it regardless (SetMaxBytes). +func (e *EmbeddedNATS) reopenPaced(ctx context.Context, id tenant.ID) error { if v, ok := e.failedOpen.Load(id); ok { if f := v.(openFailure); time.Now().Before(f.until) { return f.err @@ -558,7 +558,7 @@ func (e *EmbeddedNATS) openForPublish(ctx context.Context, id tenant.ID) error { _, err, _ := e.reopening.Do(string(id), func() (any, error) { err := e.reopen(ctx, id) if err != nil { - e.failedOpen.Store(id, openFailure{until: time.Now().Add(publishRetry), err: err}) + e.failedOpen.Store(id, openFailure{until: time.Now().Add(reopenRetry), err: err}) } return nil, err }) @@ -570,13 +570,14 @@ func (e *EmbeddedNATS) openForPublish(ctx context.Context, id tenant.ID) error { // for the dead-letter one, nothing decoded or re-encoded. The dead-letter // stream is DiscardOld, so a full one drops its oldest parked rows rather than // refusing. A dead-letter stream found missing is opened again with its -// tenant's queue, as Publish does. +// tenant's queue, paced as Publish's is (reopenPaced); a park refused leaves +// its row unacked, to be redelivered. func (e *EmbeddedNATS) DeadLetter(ctx context.Context, msg *Message, opts ...PublishOpt) error { subj := dlqPrefix + msg.topicKey err := e.publish(ctx, subj, msg.Data, opts) if errors.Is(err, jetstream.ErrNoStreamResponse) { if id, ok := keyTenant(msg.topicKey); ok { - if err = e.reopen(ctx, id); err == nil { + if err = e.reopenPaced(ctx, id); err == nil { err = e.publish(ctx, subj, msg.Data, opts) } } diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index 85fd5719..6e87ff7b 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -460,12 +460,13 @@ func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { assert.Equal(t, int64(testBudget), e.MaxBytes("acme")) } -// After a publish fails to open its tenant's queue, the tenant's publishes -// are refused at once, without waiting on the broker's lock, until -// publishRetry has passed: under clients retrying, one tenant's broken queue -// would otherwise hold the lock that every other tenant's open, resize and -// reload takes. Once the window has passed, a publish tries again. -func TestEmbeddedNATS_Publish_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T) { +// After a publish fails to open its tenant's queue, the tenant's publishes — +// and its parks, which find the dead-letter stream missing — are refused at +// once, without waiting on the broker's lock, until reopenRetry has passed: +// under clients retrying, or the worker parking row after row, one tenant's +// broken queue would otherwise hold the lock that every other tenant's open, +// resize and reload takes. Once the window has passed, a publish tries again. +func TestEmbeddedNATS_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T) { dir := t.TempDir() block := filepath.Join(dir, "jetstream", "$G", "streams", dlqStreamName("acme")) obstruct := func() { @@ -486,14 +487,15 @@ func TestEmbeddedNATS_Publish_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T require.ErrorIs(t, err, os.ErrNotExist) } - // The queue could open now, but within the window a publish tries - // nothing: it is refused while the lock is held elsewhere. + // The queue could open now, but within the window a publish or park + // tries nothing: each is refused while the lock is held elsewhere. e.mu.Lock() - var paced error + var paced, parked error done := make(chan struct{}) go func() { defer close(done) paced = e.Publish(ctx, acme, []byte("x")) + parked = e.DeadLetter(ctx, NewMessage(ctx, acme, []byte("x"), time.Now(), nil, nil, nil)) }() var returned bool select { @@ -503,8 +505,9 @@ func TestEmbeddedNATS_Publish_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T } e.mu.Unlock() <-done - require.True(t, returned, "a paced publish waited on the broker's lock") + require.True(t, returned, "a paced publish or park waited on the broker's lock") require.ErrorIs(t, paced, ErrQueueFull) + require.Error(t, parked) assert.Zero(t, e.MaxBytes("acme")) v, ok := e.failedOpen.Load(tenant.ID("acme")) diff --git a/internal/mq/mq.go b/internal/mq/mq.go index 6626cfa8..3f1c45c1 100644 --- a/internal/mq/mq.go +++ b/internal/mq/mq.go @@ -279,9 +279,10 @@ type Purger interface { // olderThan. Either bound alone keeps the event: unacked events are not // yet written, and recent ones are still needed for replay. A tenant // olderThan does not name keeps no history: everything it has - // acknowledged goes. Reports whether anything was removed. - // ErrConsumerNotFound when the consumer has not been created on some - // tenant's queue; the other tenants' are purged all the same. + // acknowledged goes. Reports whether anything was removed, and joins + // each failed tenant's error — ErrConsumerNotFound for one whose queue the + // consumer has not been created on; the other tenants' are purged all the + // same. PurgeAcked(ctx context.Context, consumer string, olderThan map[tenant.ID]time.Time) (purged bool, err error) } From f5d8f4843a750b9d42b18928e022c2d3ec9153c9 Mon Sep 17 00:00:00 2001 From: taitelee Date: Fri, 25 Sep 2026 06:23:19 -0400 Subject: [PATCH 14/15] test(mq): keep the streams directory occupied through a failed open --- internal/app/app_test.go | 14 +++++++++----- internal/mq/embedded_test.go | 6 ++++++ 2 files changed, 15 insertions(+), 5 deletions(-) diff --git a/internal/app/app_test.go b/internal/app/app_test.go index 17e92812..98e50e1c 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -593,14 +593,18 @@ func TestNew_QueueOpenFailure(t *testing.T) { require.ErrorContains(t, err, "mq open") }) t.Run("nested costs the tenant alone", func(t *testing.T) { + // globex, not acme: opened first, acme's streams keep the streams + // directory occupied through globex's failed open, which the server + // would otherwise remove on a goroutine of its own while the next + // open writes there (mq's TestEmbeddedNATS_PacesTheRetriesOfAQueueThatCannotOpen). cfg := testConfig(t, writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil})) - block(t, cfg.DataDir, "DLQ_acme") + block(t, cfg.DataDir, "DLQ_globex") a := newApp(t, cfg, Options{}) - assert.Zero(t, a.mq.MaxBytes("acme"), "acme's queue did not open") - assert.Equal(t, int64(50<<30), a.mq.MaxBytes("globex"), "and costs globex nothing") + assert.Zero(t, a.mq.MaxBytes("globex"), "globex's queue did not open") + assert.Equal(t, int64(50<<30), a.mq.MaxBytes("acme"), "and costs acme nothing") - require.NoError(t, a.MQ().Publish(t.Context(), mq.Topic{Tenant: "acme", Table: "t"}, []byte("x"))) - assert.Equal(t, int64(50<<30), a.mq.MaxBytes("acme"), "a publish opened it at acme's budget") + require.NoError(t, a.MQ().Publish(t.Context(), mq.Topic{Tenant: "globex", Table: "t"}, []byte("x"))) + assert.Equal(t, int64(50<<30), a.mq.MaxBytes("globex"), "a publish opened it at globex's budget") }) } diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index 6e87ff7b..15b837f6 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -479,6 +479,12 @@ func TestEmbeddedNATS_PacesTheRetriesOfAQueueThatCannotOpen(t *testing.T) { ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel() acme := Topic{Tenant: "acme", Table: "t"} + // Another tenant's streams keep the streams directory occupied: after a + // failed open the server, on a goroutine of its own, removes that + // directory and the account's once they are empty, and the obstacle put + // back below would race it — a file written into a directory being + // removed. + require.NoError(t, e.SetMaxBytes(ctx, "globex", testBudget)) require.Error(t, e.SetMaxBytes(ctx, "acme", testBudget)) obstruct() From eb9f74096da0965fca33b6af52f5804fcb8ffbc3 Mon Sep 17 00:00:00 2001 From: taitelee Date: Fri, 25 Sep 2026 08:35:06 -0400 Subject: [PATCH 15/15] test(mq): keep the streams directory occupied through the first failed open too --- internal/mq/embedded_test.go | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/internal/mq/embedded_test.go b/internal/mq/embedded_test.go index 15b837f6..2a7c483a 100644 --- a/internal/mq/embedded_test.go +++ b/internal/mq/embedded_test.go @@ -438,6 +438,15 @@ func TestEmbeddedNATS_SetMaxBytes_AQueueThatCannotOpen(t *testing.T) { require.NoError(t, os.WriteFile(block, nil, 0o600)) } obstruct() + // A directory JetStream ignores (no metafile, so recovery skips it) + // keeps the streams directory occupied through acme's failed open, + // which would otherwise leave it empty: the server then removes it on a + // goroutine of its own, and globex's open right after would race that + // inside its own MkdirAll (see + // TestEmbeddedNATS_PacesTheRetriesOfAQueueThatCannotOpen). Not another + // tenant's streams: those reserve bytes, and the refusal guarded + // against below needs the reserved count to have gone negative. + require.NoError(t, os.Mkdir(filepath.Join(filepath.Dir(block), "occupied"), 0o750)) e := openEmbedded(t, dir) ctx, cancel := context.WithTimeout(t.Context(), 10*time.Second) defer cancel()