diff --git a/AGENTS.md b/AGENTS.md index dcbcf6b3..faa5b15b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -29,23 +29,23 @@ One binary: Eighteen internal packages under `internal/` (plus `internal/testutil/` for shared test helpers): - **`api/`** — Chi HTTP router, JWT/JWKS middleware (from `auth/`), ingest/query/structured-query/SSE/schema/DLQ/pipes handlers -- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker, `tenant.Default` for the schema registry until story 6), `shortestKeepalive`/`longestGapWindow` for the two settings folded over every tenant served, and `defaultSetting`/`onDefaultAdopt` for the resources a process still has one of, which follow tenant `0`; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it +- **`app/`** — the process wiring: `New` builds every component from the boot config and the settings directory (each one wired in one place — what it opens, what it loops, what it releases — with the settings registry handed to its wiring function whole, the injection point of the per-tenant registry of #583: store-keyed getters for the handlers, `perTenant` for the async paths (with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker), the `chconn.Pools` and the per-tenant `discoveries` reconciled from `AfterAdopt`, `shortestKeepalive`/`longestGapWindow` for the two settings folded over every tenant served, and `defaultSetting`/`onDefaultAdopt` for the one resource a process still has one of, the MQ, which follows tenant `0`; the auth verifiers are per tenant, reconfigured (rebuilt only on changed wiring) and pruned from `AfterAdopt`), `Run` drives the long-lived ones under one `errgroup` until the context is cancelled or one fails, `Close` releases them in reverse order. `cmd/wavehouse` and `tests/integration` both boot through it - **`auth/`** — JWT auth middleware: HMAC **or** JWKS verification with `alg` pinned to the active verifier, role extraction from a configurable claim path; always runs, never rejects (bad token → empty role + stashed reason). One verifier per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 9): `Authenticator` keys them by `tenant.ID` — the request store's `settings.Store.Tenant()`, through an injected `TenantSource`; `tenant.Default` on the tenant-exempt routes — built from each tenant's `auth` block by `Reconfigure`, dropped by `Prune` once the tenant stops being served (rejected or removed), released by `Close`; the secrets (`Config`) are boot-level and shared. A JWKS key set is fetched off the boot and reload paths: until one has been stored the verifier is pending and a token-bearing request gets `503` + `Retry-After` from `api.refuseUnverifiable` (`auth.ErrVerifierPending`), never a `default_role` evaluation; refresh is library-managed (Eric, 2026-09-22), response capped at 1 MiB; the operator key's admin role is the request tenant's -- **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index). Every key leads with the tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for a result and its singleflight, `...` for a namespace — so no cached read or coalesced flight crosses tenants, and a bump through `Invalidate` names one tenant's namespaces and no other's; the one crossing is the wiring's, above the package: while the tenants share one ClickHouse (until story 6) `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant the registry knows (`Registry.Known`, a rejected tenant included), so an insert is felt by all of them -- **`chconn/`** — `Manager`, the one ClickHouse `driver.Conn` every consumer holds; `Reconfigure` swaps the connection behind it after a settings reload changes the wiring (never dials; the old connection closes after a `query_timeout` grace) +- **`cache/`** — `Cache` interface → `LocalCache` (Ristretto: one pool for every tenant) + `VersionManager` (the invalidation index). Every key leads with the tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8) — `:query:` for a result and its singleflight, `..
.
.` for a namespace — so no cached read or coalesced flight crosses tenants, a bump through `Invalidate` names one tenant's namespaces and no other's, and `InvalidateTenant` advances the tenant version that leads every namespace key of one tenant, orphaning every cached query keyed by its tables in one step (a pipe result names no table and keeps its TTL, [#343](https://github.com/Wave-RF/WaveHouse/pull/343)); the one crossing is the wiring's, above the package: `internal/app` hands the ingest worker the cache through `sharedTables`, which repeats each of the worker's bumps under every tenant on the same ClickHouse address and database (`chconn.Pools.SharingTables`, whatever their user or tls block — they read the same tables), and orphans the table-keyed cache (the structured-query results) of a tenant back on a pool after an absence, since it was out of that fan-out while away, or moved to another address or database, since it now reads other tables (story 6) +- **`chconn/`** — `Pools`, one `Manager` (a `driver.Conn`) per distinct `Identity{Addr, Database, Username, Password, TLS}` tuple among the served tenants, reconciled from the settings registry's `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to their largest `max_open_conns`/`max_idle_conns`; a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had (never dials; a resize swaps the connection with the same grace). The boot config's `clickhouse.max_total_conns` bounds the open pools' `max_open_conns` together: boot refuses naming sum and ceiling; at a reload a resize above it keeps the pool's size, and a tuple that cannot be opened (the ceiling, an unreadable certificate, or options the driver refuses) leaves its tenants on the pool they had or on none — logged, retried by the next reload. Every consumer resolves its tenant's pool per call: `For` (nil for a tenant on no pool, a `503`), `Target` (the tenant's own HTTP wiring over its pool's TLS config), `SharingTables`, `Ping` (every pool at once, ready at the first answer). `HTTPClients` keeps one `http.Client` per TLS config - **`chsql/`** — dependency-free ClickHouse SQL helpers shared by `query`/`policy` (avoids an import cycle): `QuoteIdent` (backtick-quote every identifier) + `BindUnsafe` (reject names with a literal `?`) - **`config/`** — YAML + env var config loading (cleanenv); strict on both sides (undeclared YAML key, unbound `WH_*` variable) and probes `data_dir` writability — boot is the validator, there is no dry run - **`dedupe/`** — `Deduplicator` interface → `Embedded` (Pebble), wrapped by `Managed` whose open/closed state follows the hot-reloadable `dedupe.enabled` in the settings directory's `config.json`; `Stores` holds one `Managed` per tenant, built through a `Factory` (`func(tenant.ID) *Managed`, the seam a shared backend later slots into; `Managed` opens its store through a function, `Embedded(dir)` for Pebble), rooted at `data_dir//dedupe` whatever the directory's shape (the four files are tenant `0`; an earlier layout's `data_dir/pebble` is moved there at boot when `data_dir/0/dedupe` is absent, and left alone and warned about when both exist), and reconciled from the registry's `AfterAdopt` hook — open exactly when the tenant is served with its switch on, closed with its data kept on disk otherwise ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7) -- **`discovery/`** — `SchemaRegistry` that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) -- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`; the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`), so non-admin callers never reach the proxy. A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) +- **`discovery/`** — `SchemaRegistry`, one per served tenant over a `Source` read once per refresh — the tenant's pool's connection and the database that pool was opened for, one snapshot, so a refused move keeps discovering the database the tenant's queries still use (`internal/app`'s `discoveries` builds, runs and stops them from `AfterAdopt` and `App.Close`: `RetryRefresh` until the first success, then `StartAutoRefresh` with a random first tick; `Lookup` answers `ErrNotLoaded` before the first success — the handlers' `503` with `Retry-After` — and `ErrUnknownTable` after; a failed loop attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`), that introspects ClickHouse `system.columns` (name/type/nullability plus `default_expression` and 1-based `position`) and `system.tables` (each table's `create_table_query`, kept in-process and never serialized — an external-engine table renders its wiring there unconditionally — endpoint, bucket/host, database, username, S3 access key id; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so the exposure is the topology, not the secret), records the server version, + `Validate()` for ingest payloads + `CanonicalizeTimestamps()` rewriting top-level `DateTime`/`DateTime64` column values to the canonical RFC 3339 UTC wire form pre-publish (Key Design Decision #19) +- **`ingest/`** — Ingest worker pipeline (`worker.go`: JetStream input → per-table batch INSERT with DLQ output). The pipeline is **insert-only**. The wire format `EventMessage` (`types.go`) carries `{table_name, scope, received_timestamp, format, columns, row}` and nothing else — `row` is one positional `JSONCompactEachRow` line and `columns` names its slots, the table's **insertable** columns (a `MATERIALIZED`/`ALIAS` column cannot be named in an `INSERT`); the worker batches per (tenant, table, column list), the tenant read off each message's `mq.Topic`, and inserts each batch into its tenant's own ClickHouse (`chconn.Pools.Target`); the worker accepts whatever table name the envelope carries (table existence was already checked by the HTTP ingest handler, which `404`s an unknown table before publish; the worker doesn't re-validate), then bulk-INSERTs. In the embedded-NATS deployment (the default), the server runs with `DontListen: true` (`internal/mq/embedded.go`), so the only Publishers reachable on the `ingest.>` subjects are in-process Go code — today, only the HTTP `/v1/ingest?table={table}` handler. Non-insert mutations (`DELETE`/`UPDATE`/`TRUNCATE`/…) must go through `POST /v1/ops/query` under the admin role (the same `RequireAdmin` gate as the rest of `/v1/ops/*`), so non-admin callers never reach the proxy. A request with no token (or an invalid one) resolves to the `default_role`, which in a production config is not the admin role (setting them equal is a loudly-warned dev-only setting), so it can't reach this endpoint. Plus `Sweeper` (Active Sweeper for NATS message lifecycle) + `EventMessage`/`BufferConsumerName` types (`types.go`) - **`mq/`** — the message-queue boundary: the **only** package that imports NATS/JetStream (Key Design Decision #20). and the only one that knows how the broker works. Everything else addresses events by `Topic{Tenant, Table, Scope}` (a validated tenant id and raw names — the tenant leads every subject, `ingest..
`, so one wildcard selects a tenant's traffic, a topic without one is refused, and a pre-tenant subject reads as tenant `0`'s) and states intent through the interfaces — `Publisher` (`ErrQueueFull` is the backpressure signal), `Subscriber`, `ConsumerManager`/`Consumer`/`ConsumerConfig` (the ingest worker's durable consumer), `DeadLetterer` and `DeadLetterStats` (park a message, count what is parked), `Purger` (drop what is both acked and older than a cutoff — the sweeper), `Replayer` (SSE gap-fill) — composed into `Broker`, which adds the byte budget (`SetMaxBytes`/`MaxBytes`: the `mq.max_bytes_gb` reload) and `Stats` (the system gauges' source). Subjects, prefixes, wildcards, token encoding, stream names, sequences, and ack floors are private to the one implementation, `EmbeddedNATS` (`embedded.go`, `subject.go`, `purge.go`); `internal/app` constructs it and hands everything else a `mq.Broker` - **`observability/`** — OpenTelemetry pipeline: `InitProvider` wires trace/metric/log providers via OTLP gRPC (each signal independently gated). A top-level `Prometheus` config block drives an optional `/metrics` scrape endpoint that runs independently of OTLP push — standalone (Alloy/Mimir scrape, no collector), alongside OTLP, or off. `NewLogger` produces a slog handler that fans out to stdout AND OTLP (stdout always 100%, OTLP sample-rate-aware). `TraceHandler` injects trace_id/span_id from active spans. `tracer.go` provides W3C trace context propagation over message headers (`InjectHeaders`/`ExtractHeaders` on a plain header map; `internal/mq` injects on every publish and extracts on the `Subscribe` path, so this package never sees a NATS type). - **`pipes/`** — Named query pipes: `NamedQuery` type + `BindParams` + `Source` (read per request; `settings.Store` in production, `Static(q...)` in tests) - **`policy/`** — Hasura-style access control, **role-first**: `TablePolicy` is `map[string]RolePermissions`, and a role's grant splits by operation into `SelectPermissions` (columns, row `filter`, aggregations, the `max_*` limits) and `InsertPermissions` (columns, `check`) — so a field only one side honors does not exist on the other. `Evaluate()` resolves ONE operation and leaves the other side **nil** (`Select *ResolvedSelect` / `Insert *ResolvedInsert`), which every accessor fails closed on — nil is "not resolved", distinct from an empty side, which is "unrestricted" (what the admin return builds). Claim templating (`{{ jwt.claim.path }}`) resolves during that call. Policies come from `Source`, a `func() *Policy` read per call (`settings.Store.Policy` in production, `Static(p)` in tests) - **`query/`** — Structured query AST types + SQL builder with schema validation, structural policy predicate/limit emission, timestamp bucketing - **`settings/`** — the settings directory, in either shape ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): flat (the four files: tenant `0` alone) or nested (one folder per tenant, never mixed). `Validate` detects the shape and checks it — `ValidateDir` per directory (strict JSON, per-file rules, cross-file role references), folder names against `tenant.Parse`, a nested finding's `File` led by its folder; `Store` is a passive holder (one tenant's adopted snapshot, typed accessors read per call); `Registry` (tenant id → `Store`) owns `Open`, the serialized `Reload`/`ReloadTenant`, the `AfterAdopt` hooks, and the fsnotify `Watch` (flat only). Flat refuses an invalid directory at boot and keeps the previous snapshot on a rejected reload; nested fails closed per tenant (a rejected folder stops being served, the rest carry on, a whole-tree reload mirrors the folders, and a finding about the root itself rejects the reload whole). Plus the embedded (`go:embed`) seed `wavehouse bootstrap` writes -- **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) -- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes; not `nats` or `pebble` in any letter case, the entries `data_dir` keeps for itself — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper folds over the tenants served (`longestGapWindow`); the schema registry is still constructed with `tenant.Default` (story 6) +- **`stream/`** — SSE fan-out: rows travel POSITIONALLY, so each connection is told its projected column list in an `event: schema` frame before its first row and again on drift — **not** guaranteed after a gap-fill across a column change, which can leave a connection reading live rows against a stale list until it reconnects ([#543](https://github.com/Wave-RF/WaveHouse/issues/543)) — (tracked per connection; replay tracks its own). The event `Hub` (registers subscribers by `(mq.Topic, role)` — one tenant's table — and evaluates each event under its own tenant's policy and schema registry; `Broadcast` projects + serializes each event once per role, the #294 delivery hot path — a role carrying a row-level `filter` keeps the shared projection but delivers per subscriber, each subscriber's claims evaluated against the row, #319), `Subscriber` (per-connection outbound `Frame` queue, `Send`/`Frames`; claims fixed at construction, immutable), the `Bucket` fan-out set (`subscriberSet`, one per `(topic, role)`), the `Heartbeater` keepalive wheel, and `Metrics` (the `wavehouse_sse_*` stream instruments) +- **`tenant/`** — the tenant identifier ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)): `ID` (a validated string), `Parse` (letters, digits, `_`, `-`; ≤ 64 bytes; not `nats` or `pebble` in any letter case, the entries `data_dir` keeps for itself — safe as a folder name and as an MQ subject token), `Default` (`"0"`), and `Header` (`X-Tenant-ID`). Imports nothing from the rest of the repo. `api.TenantMW` resolves the header against `settings.Registry` before auth on every `/v1` route outside `/v1/ops/*` (`400` malformed, `404` unknown, a bare `503` for a nested tenant whose folder was rejected) and puts the resolved `*settings.Store` in the request context; the ops routes that address one tenant (`GET /v1/ops/pipes[/{name}]`, `POST /v1/ops/settings/reload`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh`, `POST /v1/ops/query`) take a strictly parsed `?tenant=` instead; handlers read it once (`api.StoreFromContext`) and pass it down as an argument, and nothing below a handler reads context. The stream hub and the ingest worker read each message's tenant off its `mq.Topic` and their getters take it; the sweeper folds over the tenants served (`longestGapWindow`); each served tenant has a schema registry of its own (story 6) ## Key Design Decisions @@ -67,8 +67,8 @@ The invariant index — what must stay true. Full narrative and rationale live i 14. **TypeScript SDK** — `@wavehouse/sdk`: typed query builder, real-time SSE over `fetch`, live queries (incrementable/decomposable/poll aggregation), codegen CLI. Exactly one runtime dependency — `eventsource-parser` (SSE framing, itself dependency-free); adding a second needs the same scrutiny the first got. The canonical client (see §SDK Sync). 15. **Observability invariants** — stdout always 100% (sampling is OTLP-push-only); WARN+ERROR always export at 100% (a non-configurable floor — don't expose it); gRPC OTel exporters dial lazily so an unreachable collector never blocks startup; the OTel Prometheus exporter uses a **private** `prometheus.Registry`. The OTLP endpoint/TLS/custom-CA/mTLS/headers are delegated to the OpenTelemetry SDK's standard `OTEL_EXPORTER_OTLP_*` env vars — `InitProvider` passes **no** endpoint/header options. Known gap, intentionally not patched in WaveHouse app code: the pinned gRPC logs exporter (`otlploggrpc` v0.19/v0.20) ignores the env TLS-cert vars, so a custom/private CA and mutual TLS apply to traces/metrics but **not** the logs signal (public-CA/system-roots TLS and plaintext still work for logs) — upstream bug open-telemetry/opentelemetry-go#6661. A malformed `OTEL_EXPORTER_OTLP_HEADERS` is logged and skipped by the SDK (fail-soft), not fatal. Preserve when touching the logger/sampler/provider. Detail: architecture.md § `observability/`. 16. **Bearer-token-only CORS posture (security)** — Bearer JWT on every request, no cookies/sessions; `corsMiddleware` deliberately **never** emits `Access-Control-Allow-Credentials` (not needed, and `*` + credentials is a spec violation browsers reject). `cors.allowed_origins` (settings directory, per tenant: a tenant route is decorated from the list of the tenant it names, everything else from tenant `0`'s — `corsOrigins`) controls who can *read* responses, not cookie scope; CSRF protection is structural. Don't reintroduce cookie auth or `Allow-Credentials` without a design discussion — answers GitHub #29/#30. Code: `internal/api/router.go`. -17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (backoff 2s → 60s). Bounds supervisor restart loops. -18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. +17. **Non-fatal boot** — schema-discovery failure on boot is non-fatal: `internal/app` records an `api.BootState`, binds `:8080`, serves 503 on `/livez`/`/readyz` with the diagnostic, and retries via `SchemaRegistry.RetryRefresh` (backoff 2s → 60s), per tenant over a nested directory: `/livez` is 503 while no tenant has completed a first discovery, then sticky 200, and a tenant's outage after that is its log line and counter, never a probe failure. Until a tenant's first discovery its table lookups are a 503 with `Retry-After`, not a 404. Bounds supervisor restart loops. +18. **Health endpoints** — liveness `/livez`, readiness `/readyz` (k8s convention; `/readyz` pings every open ClickHouse pool at once and is ready at the first answer, 503 naming each when none answers); `/healthz` is a permanent alias of `/livez`; `/health` + `/ready` are deprecated (removal v0.2.0, CHANGELOG #144). `/v1/health` is the SDK's content-free public ping (no ClickHouse check), a `/v1` route so it survives reverse-proxy probe-path filtering. Point k8s at `/livez`/`/readyz`, SDK/online-checks at `/v1/health`, never the deprecated aliases. 19. **Canonical timestamp wire form (fail-open at ingest)** — the HTTP ingest handler rewrites every top-level `DateTime`/`DateTime64` column value it can parse to RFC 3339 UTC (`discovery.CanonicalizeTimestamps`; per-column precision + zone precomputed at schema refresh) after validation + policy checks and **before** the NATS publish, so the one payload every consumer shares — SSE subscribers, the ClickHouse insert, the DLQ — carries the same spelling `/v1/query` renders: live and query reads can't drift on the instant (#372). Zone-less inputs are read in the column's declared zone, else the discovered server default — ClickHouse's own rule, so the spelling changes but never the instant. Deliberately **fail-open**: an unparseable value or unresolvable zone (no tzdata embedded — never a failed refresh, never a silent UTC reinterpretation, which would move instants) publishes verbatim; ingest must not reject a record over its timestamp spelling — fail-closed enforcement belongs to the stream row-filter (#381). Don't re-spell timestamps downstream. Preserve when touching `internal/discovery`, the ingest handler, or the SSE fan-out. Detail: architecture.md § `discovery/` + §Ingest Path; the exact spelling spec (truncation, zero-trimming, `Z`-only) lives in api.md §Timestamp canonicalization — keep it in sync with `canonicalTimestamp`. 20. **Sealed MQ boundary** — only `internal/mq` imports NATS/JetStream (`github.com/nats-io/…`), enforced by the `depguard` rule in `.golangci.yml`, so `make lint` fails on a leak in every package it builds (the `integration`-tagged files under `tests/` are outside lint's build context — keep them clean by convention, through `mq.Broker`). The boundary is semantic as well: everything else addresses events by `mq.Topic` and states intent through mq-owned interfaces (`Publisher`, `Consumer`, `DeadLetterer`, `Purger`, `Replayer`, …), and never builds a subject, names a stream, or reasons in sequences — so a subject, stream, or broker change lands in one package ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 4; story 5's tenant token landed there alone — `Topic.Tenant`, first in every subject). Don't add a raw accessor (`JetStream()`, `NatsConn()`, `GetServer()`) back, and don't hand-build `"ingest."`/`"dlq."` subjects outside `internal/mq` — widen the mq surface with an intent-level method instead. @@ -428,7 +428,7 @@ internal/api/ → HTTP layer (handlers, router, middleware, schema/DLQ internal/app/ → Process wiring (build every component, run them under one errgroup, release in reverse) internal/auth/ → JWT/JWKS authentication middleware (HMAC or JWKS, role extraction from claims) internal/cache/ → Query cache (interface, Ristretto L1, tenant-led version index) -internal/chconn/ → ClickHouse connection manager (driver.Conn swapped on settings reload) +internal/chconn/ → ClickHouse pools, one per connection tuple among the served tenants (reconciled on settings reload) internal/chsql/ → Shared ClickHouse SQL helpers (identifier quoting + bind-safety) internal/config/ → Configuration structs + loader internal/dedupe/ → Optional deduplication (interface + embedded/distributed) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5212e5e4..eacf2399 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,7 @@ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), ### Added +- **One ClickHouse pool per tuple and one schema registry per tenant** (`internal/chconn/chconn.go` (+ tests), `internal/discovery/discovery.go` (+ tests), `internal/app/discoveries.go` (new), `internal/app/{app,wire}.go` (+ tests), `internal/api/{schema,ingest,structured_query,pipes,query,health,errors,clickhouse_exec}.go` (+ tests), `internal/stream/hub.go`, `internal/ingest/worker.go`, `internal/cache/{cache,local,version_manager}.go` (+ tests), `internal/testutil/{testutil,mocks}.go`, `tests/integration/{setup,tenants,boot_resilience,query_limits}_test.go`, `clients/ts/src/{schema,table,sql,client,types}.ts` (+ tests), `tests/e2e/sdk/admin.test.ts`, `docs/src/content/docs/{api,deployment,architecture,ingest-pipeline}.md`, `docs/src/content/docs/{settings-directory,configuration,access-control,reverse-proxy}.mdx`, `docs/src/content/docs/sdk/{admin,reference,queries}.md`, `AGENTS.md`): the second slice of story 6 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)), with no behavior change for a settings directory that holds the four files beyond the three noted at the end. The process opens one native pool per distinct `clickhouse.addr` / `database` / `username` / password / `tls` tuple among the served tenants (`chconn.Identity`, `chconn.Pools`), shared by the tenants naming it and sized to their largest `max_open_conns` and `max_idle_conns`; `http_port`, `http_scheme`, `headers` and `query_timeout` stay each tenant's own. Every reload reconciles the pools: a new tuple opens (never dials), a tenant whose tuple changed is repointed, a tuple no tenant names closes after the longest `query_timeout` among the tenants it had, and a pool whose largest ask changed is resized with the same grace. The boot config's `clickhouse.max_total_conns` now bounds the open pools together: boot is refused naming the sum and the ceiling; at a reload a resize above it is refused with the pool kept at its size, and a tuple that cannot be opened — the ceiling, a certificate file that cannot be read, or options the driver refuses — leaves its tenants on the pool they had (the keep-previous-wiring rule of the connection ceiling) or on none when they had none; both are logged and the next reload retries. A tenant on no pool fails closed: `503` with `Retry-After: 30` on `POST /v1/query`, `GET/POST /v1/pipes/{name}`, `POST /v1/ops/query` and `POST /v1/ops/schema/refresh`, ahead of the cache. Each served tenant gets a `discovery.SchemaRegistry` of its own over its pool, kept fresh by its own loop — the boot retry until the first success, then `schema.refresh_interval` with the first refresh at a random point within the interval so tenants adopted together do not refresh together — created and stopped from the settings reload and stopped under `App.Close`; `wavehouse_schema_refresh_failures_total{tenant}` counts a loop's failed attempts. `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the strict `?tenant=` the pipe reads take (absent is tenant `0`, `400` malformed, `404` unknown, `503` rejected); the SDK sends it as the `tenant` option of `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()` and `wh.sql()`. Over a nested directory `/livez` (and `/v1/health`) is `503` with the latest discovery failure, naming its tenant, while no tenant has completed a first discovery, then `200` for the rest of the process lifetime; `/readyz` pings every open pool at once, is ready at the first answer, and names every pool that did not answer when none does — one tenant's ClickHouse outage is its log line and counter, never a probe failure. The ingest worker inserts each batch into its own tenant's ClickHouse — the tenant the message's topic names, through that tenant's HTTP wiring (`chconn.Pools.Target`); a tenant on no pool takes the failure path an unreachable ClickHouse takes — and its cache invalidation fans out to the tenants on the same address and database as the batch's tenant, whatever their user or `tls` block (they read the same tables), rather than to every known tenant, and a tenant adopted after an absence — rejected or removed, so out of that fan-out — or moved to another address or database has its cached structured-query results orphaned in one step (`Cache.InvalidateTenant`, a tenant generation in every version key), so a repaired folder never serves query rows cached before the inserts it missed (a pipe result names no table, so neither this nor any insert invalidates it: it stays until its TTL expires). Three changes reach the single-tenant directory too: a table lookup before the first discovery is a `503` with `Retry-After: 5` (`schema not loaded yet`) rather than a `404`, on `POST /v1/ingest`, `POST /v1/query` and `GET /v1/ops/schema` (the list included, where `[]` would read as no tables); the first periodic refresh fires at a random point within the interval rather than a full interval after boot; and a reload that moves `clickhouse.addr` or `clickhouse.database` now orphans the structured-query results cached before it, where they were served until their TTL. Until [#529](https://github.com/Wave-RF/WaveHouse/issues/529) every tenant's user authenticates with `WH_CH_PASSWORD`, so the tuple is in effect the address, database, username and `tls` block. - **One token verifier per tenant, built off the boot and reload paths** (`internal/auth/auth.go` (+ tests), `internal/api/router.go` (+ tests), `internal/app/{app,wire}.go` (+ tests), `go.mod`, `docs/src/content/docs/sdk/{reference,streaming}.md`, `docs/src/content/docs/{architecture,deployment,api}.md`, `docs/src/content/docs/{settings-directory,configuration}.mdx`, `SECURITY.md`): story 9 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). Over a nested settings directory each tenant's folder now wires that tenant's verifier (`auth.jwks_url`, `auth.role_claim`), so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its identity provider's key set — under any other tenant's header it is refused as invalid — where before one verifier, tenant `0`'s, accepted a token under any header. `auth.Config` is now the boot-config half alone (the HMAC secret and the operator key, shared by every tenant) and the new `auth.Wiring` a tenant's half; `NewAuthenticator` builds no verifier, `Reconfigure(id, wiring)` gives a tenant one — swapped atomically when its wiring changed, kept when it did not — `Prune` drops the verifiers of the tenants a reload stopped serving, rejected or removed alike (no work runs for a tenant that is not served; a folder adopted again gets a fresh verifier), and `Close` stops every JWKS refresh as a component of `App.Close`. This holds on every reload because the registry's `AfterAdopt` hooks run after every reload it applied, empty list included, so a per-tenant reload that rejects a folder drops its verifier then rather than at the next adoption. The middleware reads the request tenant's verifier through an injected `auth.TenantSource` — the store `api.TenantMW` resolved names its tenant with `settings.Store.Tenant()`, one read of the context — a tenant-exempt route (the ops tree) verifies as tenant `0`, and a tenant with no verifier fails closed. The operator key stamps the request tenant's `admin_role`, read from that tenant's policy, rather than tenant `0`'s. A JWKS key set is fetched on its own goroutine: the verifier is in place at once and *pending* until a set has been stored — the first fetch retried with backoff from one second to a minute, a stored set then kept fresh by the library hourly and, rate-limited, on an unknown key id — so neither boot nor a reload (which holds the registry's lock) waits on the endpoint, and **an unreachable JWKS no longer refuses boot** — it logs (`jwks refresh failed; no token validates until it succeeds`) and that tenant alone is affected. A token checked against a pending verifier is a new outcome, `auth.ErrVerifierPending`: every `/v1` route answers it `503 {"error": "token verifier not ready: …"}` with `Retry-After: 30` (`api.refuseUnverifiable`) rather than evaluate the request under the `default_role`, which could accept its data under a lesser role while another pod holding the keys would have served it as its own; a request without a token, and the operator key, are unaffected. Every fetch goes through one client that caps the response at 1 MiB, refusing a larger one as unreachable with `jwks response exceeds 1048576 bytes` as the logged cause. Nothing changes for a flat directory beyond that boot rule: its one tenant gets exactly the verifier it had. The boot warning for the secretless posture (no `auth.jwt_secret` and no `jwks_url`, whose token refusal landed in [#607](https://github.com/Wave-RF/WaveHouse/pull/607)) is now per tenant — each served tenant with no `jwks_url` while the boot secret is unset — where one line that any JWKS tenant silenced used to stand for the whole directory. - **ClickHouse TLS, HTTP-interface headers, pool sizes and a connection ceiling** (`internal/settings/{settings,validate,store}.go` + seed, `internal/chconn/chconn.go`, `internal/ingest/worker.go`, `internal/api/query.go`, `internal/config/config.go`, `internal/app/wire.go`, `deployments/compose/settings/config.json`, `deployments/compose/standalone.yaml`, `config.yaml`, `docs/src/content/docs/{settings-directory,configuration,reverse-proxy}.mdx`, `docs/src/content/docs/{architecture,deployment}.md`): the tenant-agnostic first slice of story 6 of the multi-tenant epic ([#583](https://github.com/Wave-RF/WaveHouse/issues/583)). `config.json`'s `clickhouse` block gains a `tls` block (`enabled`, `ca_file`, `cert_file`, `key_file`, `insecure_skip_verify`, `server_name`), a `headers` map for the HTTP interface, and `max_open_conns` / `max_idle_conns`. **Every key is required, so an existing `config.json` must add them**; the seed values (`tls.enabled: false` with the other `tls` keys empty, `headers: {}`, `10` / `5`) change nothing. `tls.enabled` switches the native hop to TLS, `http_scheme` stays the HTTP hop's switch, and the material applies to whichever hop uses TLS: the driver gets the TLS config and the pool sizes, the ingest worker and the raw-SQL proxy get the TLS config and the headers, set ahead of their own so the credentials win (naming `X-ClickHouse-User`, `X-ClickHouse-Key` or `Authorization` is a validation error, and so are two spellings of one name). Validation checks shape only — the paths are not opened, so `wavehouse validate` runs anywhere — warns when `insecure_skip_verify` is on or when only one of the two hops is on TLS (each carries the credentials in the clear without it), and a certificate file that cannot be read or parsed refuses boot or leaves a reload's connection unchanged; the files are read when the connection is built, so a file replaced in place needs a restart. Boot config gains the optional `clickhouse.max_total_conns` (`WH_CH_MAX_TOTAL_CONNS`, `0` = no ceiling): a settings pool above it refuses boot, naming both numbers in the error, and a reload that raises the pool above it is refused and logged (the reload still reports adopted), leaving the connection as it was. diff --git a/clients/ts/src/client.ts b/clients/ts/src/client.ts index 46b770e4..be6afc3b 100644 --- a/clients/ts/src/client.ts +++ b/clients/ts/src/client.ts @@ -7,7 +7,14 @@ import { StreamController } from "./stream/controller.js"; import { SSETransport } from "./stream/sse.js"; import { SysNamespace } from "./sys.js"; import { TableRef } from "./table.js"; -import type { ClientConfig, Database, HttpContext, Result, StreamOptions } from "./types.js"; +import type { + ClientConfig, + Database, + HttpContext, + OpsRequestOptions, + Result, + StreamOptions, +} from "./types.js"; type TableName = DB extends Database ? Extract : string; type RowType = DB extends Database @@ -71,11 +78,13 @@ export class WaveHouseClient { * `service` role). The endpoint proxies straight to ClickHouse's HTTP * interface so any ClickHouse-accepted SQL works; positional `?` param * binding is NOT supported — inline literals or use the structured query - * builder for safe binding. See sql.ts for details. + * builder for safe binding. See sql.ts for details. `opts.tenant` names the + * tenant whose ClickHouse the SQL runs against, the default tenant without + * it. */ sql>( query: string, - opts?: { signal?: AbortSignal }, + opts?: OpsRequestOptions, ): Promise> { // Migration guard: the second argument used to be a positional-`?` // params array. TS callers get a compile-time error from the type diff --git a/clients/ts/src/namespaces.test.ts b/clients/ts/src/namespaces.test.ts index 48df893f..d1c0db5d 100644 --- a/clients/ts/src/namespaces.test.ts +++ b/clients/ts/src/namespaces.test.ts @@ -24,6 +24,12 @@ describe("sql", () => { it("POSTs to /v1/ops/query with sql field", async () => { const result = await sql(makeCtx(), "SELECT count() FROM clicks"); + expect(new URL(fetchSpy.mock.calls[0][0]).search).toBe(""); + + await sql(makeCtx(), "SELECT 1", { tenant: "acme" }); + expect( + new URL(fetchSpy.mock.calls[1][0]).pathname + new URL(fetchSpy.mock.calls[1][0]).search, + ).toBe("/v1/ops/query?tenant=acme"); expect(result.data).toEqual([{ count: 10 }]); const [url, init] = fetchSpy.mock.calls[0]; @@ -79,6 +85,24 @@ describe("SchemaNamespace", () => { expect(fetchSpy.mock.calls[0][0]).toContain("/v1/ops/schema"); }); + it("list() and refresh() send opts.tenant as ?tenant=, and nothing without it", async () => { + fetchSpy.mockImplementation(async () => new Response("[]", { status: 200 })); + const ns = new SchemaNamespace(makeCtx()); + + await ns.list({ tenant: "acme" }); + await ns.refresh({ tenant: "acme" }); + await ns.list(); + // An empty id is the caller's bug: it is sent for the server to refuse, + // never dropped into a read of the default tenant. + await ns.refresh({ tenant: "" }); + + const urls = fetchSpy.mock.calls.map((call) => new URL(call[0])); + expect(urls[0].pathname + urls[0].search).toBe("/v1/ops/schema?tenant=acme"); + expect(urls[1].pathname + urls[1].search).toBe("/v1/ops/schema/refresh?tenant=acme"); + expect(urls[2].search).toBe(""); + expect(urls[3].search).toBe("?tenant="); + }); + it("refresh() POSTs to /v1/ops/schema/refresh", async () => { fetchSpy.mockResolvedValue(new Response(JSON.stringify({}), { status: 200 })); diff --git a/clients/ts/src/schema.ts b/clients/ts/src/schema.ts index 397d897b..9813495c 100644 --- a/clients/ts/src/schema.ts +++ b/clients/ts/src/schema.ts @@ -1,6 +1,6 @@ import { err, ok } from "./errors.js"; -import { request } from "./http.js"; -import type { HttpContext, Result, Schemas } from "./types.js"; +import { request, tenantParam } from "./http.js"; +import type { HttpContext, OpsRequestOptions, Result, Schemas } from "./types.js"; /** Namespace for schema introspection. */ export class SchemaNamespace { @@ -10,12 +10,17 @@ export class SchemaNamespace { this._ctx = ctx; } - /** List all table schemas discovered from ClickHouse. */ - async list(opts?: { signal?: AbortSignal }): Promise> { + /** + * List all table schemas discovered from ClickHouse — `opts.tenant`'s, the + * default tenant's without it. A `503` with `Retry-After` is a tenant whose + * first discovery has not succeeded yet. + */ + async list(opts?: OpsRequestOptions): Promise> { // The backend returns TableSchema[] — transform to Record. const { data, error } = await request(this._ctx, { method: "GET", path: "/v1/ops/schema", + params: tenantParam(opts), signal: opts?.signal, }); if (error) return err(error); @@ -34,11 +39,15 @@ export class SchemaNamespace { return ok(schemas); } - /** Force a schema refresh from ClickHouse system.columns. */ - async refresh(opts?: { signal?: AbortSignal }): Promise> { + /** + * Force a schema refresh from ClickHouse system.columns — of `opts.tenant`, + * the default tenant without it. + */ + async refresh(opts?: OpsRequestOptions): Promise> { const { error } = await request(this._ctx, { method: "POST", path: "/v1/ops/schema/refresh", + params: tenantParam(opts), signal: opts?.signal, }); if (error) return err(error); diff --git a/clients/ts/src/sql.ts b/clients/ts/src/sql.ts index 4df26dae..48a81501 100644 --- a/clients/ts/src/sql.ts +++ b/clients/ts/src/sql.ts @@ -1,6 +1,6 @@ import { err, ok } from "./errors.js"; -import { request } from "./http.js"; -import type { HttpContext, Result } from "./types.js"; +import { request, tenantParam } from "./http.js"; +import type { HttpContext, OpsRequestOptions, Result } from "./types.js"; /** * Execute a raw SQL query against ClickHouse. @@ -37,11 +37,12 @@ import type { HttpContext, Result } from "./types.js"; export async function sql>( ctx: HttpContext, query: string, - opts?: { signal?: AbortSignal }, + opts?: OpsRequestOptions, ): Promise> { const { data, error } = await request(ctx, { method: "POST", path: "/v1/ops/query", + params: tenantParam(opts), body: { sql: query }, signal: opts?.signal, }); diff --git a/clients/ts/src/table.test.ts b/clients/ts/src/table.test.ts index 35958cfd..6c0e2fe6 100644 --- a/clients/ts/src/table.test.ts +++ b/clients/ts/src/table.test.ts @@ -226,6 +226,13 @@ describe("TableRef", () => { expect(result.data).toEqual(schema); expect(fetchSpy.mock.calls[0][0]).toContain("/v1/ops/schema?table=clicks"); + + // The tenant rides beside the table, as on every admin route. + await table().schema({ tenant: "acme" }); + const url = new URL(fetchSpy.mock.calls[1][0]); + expect(url.pathname).toBe("/v1/ops/schema"); + expect(url.searchParams.get("table")).toBe("clicks"); + expect(url.searchParams.get("tenant")).toBe("acme"); }); // --- stream --- diff --git a/clients/ts/src/table.ts b/clients/ts/src/table.ts index 011d1ca2..37eb8da8 100644 --- a/clients/ts/src/table.ts +++ b/clients/ts/src/table.ts @@ -1,11 +1,12 @@ import { err, ok } from "./errors.js"; -import { request } from "./http.js"; +import { request, tenantParam } from "./http.js"; import { QueryBuilder } from "./query-builder.js"; import type { StreamController } from "./stream/controller.js"; import type { HttpContext, InsertRecordResult, InsertResult, + OpsRequestOptions, RequestOptions, Result, StreamOptions, @@ -190,11 +191,12 @@ export class TableRef> { return ok(result); } - /** Fetch the schema for this table. */ - async schema(opts?: { signal?: AbortSignal }): Promise> { + /** Fetch the schema for this table — under `opts.tenant`, the default tenant without it. */ + async schema(opts?: OpsRequestOptions): Promise> { const { data, error } = await request(this._ctx, { method: "GET", - path: `/v1/ops/schema?table=${encodeURIComponent(this._table)}`, + path: "/v1/ops/schema", + params: { table: this._table, ...tenantParam(opts) }, signal: opts?.signal, }); if (error) return err(error); diff --git a/clients/ts/src/types.ts b/clients/ts/src/types.ts index d4de4b8b..5feae56d 100644 --- a/clients/ts/src/types.ts +++ b/clients/ts/src/types.ts @@ -470,16 +470,19 @@ export interface PipeRequestOptions { /** * Options for a call to one of the admin routes that address a tenant: - * `wh.pipes.list()`, `wh.pipes.get()`, and `wh.settings.reload()`. + * `wh.pipes.list()`, `wh.pipes.get()`, `wh.settings.reload()`, + * `wh.schema.list()`, `wh.schema.refresh()`, `wh.from(t).schema()` and + * `wh.sql()`. */ export interface OpsRequestOptions { signal?: AbortSignal; /** * The tenant the call addresses, sent as `?tenant=`. The admin routes ignore * the `X-Tenant-ID` header, so `options.headers` cannot select one. Omitted, - * the reads serve the default tenant (`0`) and `reload()` reloads every - * tenant. An id the server does not accept — the empty string included — is - * a `400`, never a silent fallback to the default. + * the reads, the refresh and `sql()` serve the default tenant (`0`) and + * `reload()` reloads every tenant. An id the server does not accept — the + * empty string included — is a `400`, never a silent fallback to the + * default. */ tenant?: string; } diff --git a/docs/src/content/docs/access-control.mdx b/docs/src/content/docs/access-control.mdx index 38b31da7..92ca8003 100644 --- a/docs/src/content/docs/access-control.mdx +++ b/docs/src/content/docs/access-control.mdx @@ -213,7 +213,7 @@ On a structured query (`POST /v1/query?table={table}`) the allowlist is a **hard This produces `WHERE (tenant_id = ?)` with the caller's `app_metadata.tenant_id` claim bound as the parameter — so a `viewer` only ever sees rows for their own tenant, and the value comes from the signed token, not from anything the client sends. :::note[Two meanings of "tenant"] -On this page a *tenant* is a row-scoping value carried in the signed token. The [`X-Tenant-ID` header](/deployment#multi-tenant-deployments) is a different axis: it selects which set of settings files — and so which `policies.json` — serves the request. It is client-supplied and resolved before authentication, so it is no row-isolation boundary: queries under every tenant read the same tables, and the one place rows divide by tenant is the live stream, which carries only the events ingested under the tenant it names. A settings directory that holds the four files is one such tenant (`0`), so most deployments never send it. +On this page a *tenant* is a row-scoping value carried in the signed token. The [`X-Tenant-ID` header](/deployment#multi-tenant-deployments) is a different axis: it selects which set of settings files — and so which `policies.json` and which ClickHouse — serves the request. It is client-supplied and resolved before authentication, so on its own it is no row-isolation boundary: a caller that presents another tenant's header is evaluated under that tenant's policy against that tenant's ClickHouse, and tenants whose folders name the same ClickHouse address and database read the same tables. Scoping a caller to its own rows stays the policy's job, from a value in the signed token. A settings directory that holds the four files is one such tenant (`0`), so most deployments never send it. ::: Supported comparison operators: diff --git a/docs/src/content/docs/api.md b/docs/src/content/docs/api.md index 452f83ee..119be82e 100644 --- a/docs/src/content/docs/api.md +++ b/docs/src/content/docs/api.md @@ -111,13 +111,15 @@ Status code: `503 Service Unavailable` The boot-degraded response lets an operator `curl /livez` to learn why the gateway isn't ready to serve traffic yet, instead of grepping a restart-loop log. The binary is bound on `:8080` and serves diagnostics, but is not yet accepting ingest/query traffic. Schema discovery retries with exponential backoff (2s → 60s); once a Refresh succeeds, `/livez` flips to `200` and stays there for the rest of the process lifetime — transient ClickHouse blips after that point are reflected in `/readyz`, not `/livez`. +Over a [nested settings directory](/deployment#the-nested-settings-directory) the probe reads every tenant together: `/livez` is `503` while **no** tenant has completed a first discovery — the diagnostic names the tenant whose attempt it reports (`schema discovery: tenant acme: …`), and reads `no tenant has completed a first discovery yet` before any attempt or when the directory serves no tenant — and `200` from the first tenant's success on, for the rest of the process lifetime. A tenant whose ClickHouse is unreachable after that is a log line and the `wavehouse_schema_refresh_failures_total{tenant}` counter, never a probe failure. A tenant that has not completed its own first discovery answers `503` (`schema not loaded yet`) on its schema-aware routes until it does; one whose ClickHouse goes down after that answers query errors, as a single-tenant server does. + --- ### `GET /readyz` — Readiness Probe > Canonical name (current Kubernetes convention). Also served at **`/ready`** — a deprecated alias kept for v0.1.x and scheduled for removal in v0.2.0. -Returns `200 OK` if the process is fully booted (schema discovery complete) and ClickHouse is currently reachable. Returns `503 Service Unavailable` otherwise. No authentication required. +Returns `200 OK` if the process is fully booted (schema discovery complete) and ClickHouse is currently reachable. Returns `503 Service Unavailable` otherwise. No authentication required. Over a [nested settings directory](/deployment#the-nested-settings-directory) it pings every open ClickHouse pool at once and answers `200` at the first one that does, so a tenant whose ClickHouse does not answer does not make the process unready; the `503` names every pool that failed (one per line in `error`) when none answers — including when no pool is open at all, a directory serving no tenant. **Response (ready):** @@ -128,7 +130,7 @@ Returns `200 OK` if the process is fully booted (schema discovery complete) and **Response (not ready):** ```json -{"status": "not ready", "error": "connection refused"} +{"status": "not ready", "error": "localhost:9000 database default user default: dial tcp 127.0.0.1:9000: connect: connection refused"} ``` Status code: `503 Service Unavailable` @@ -144,7 +146,7 @@ Status code: `503 Service Unavailable` | Post-boot, ClickHouse dies | 200 ★ | 503 | | Post-boot, ClickHouse back | 200 | 200 | -★ Once boot completes, `/livez` no longer tracks ClickHouse state — a runtime ClickHouse outage surfaces in `/readyz` only. This is what keeps a Kubernetes `livenessProbe` from restart-looping the pod during a transient backend blip (see [Deployment → Boot-time degraded mode](/deployment#boot-time-degraded-mode)). +★ Once boot completes, `/livez` no longer tracks ClickHouse state — a runtime ClickHouse outage surfaces in `/readyz` only. This is what keeps a Kubernetes `livenessProbe` from restart-looping the pod during a transient backend blip (see [Deployment → Boot-time degraded mode](/deployment#boot-time-degraded-mode)). Over a nested directory the rows hold per process rather than per tenant: "ClickHouse up" means at least one tenant's pool answers, and "discovery complete" means one tenant's has. --- @@ -266,10 +268,11 @@ The body is a **flat JSON object** whose keys must match column names in the tar | 403 | `{"error":"column \"x\" not allowed for insert"}` | The record names a column the role's `allow_columns`/`deny_columns` forbids ([Access control → Column permissions](/access-control#column-permissions)) | | 403 | `{"error":"check failed for column \"x\""}` | The record's value for a checked column doesn't satisfy the policy `check` (`_eq`/`_in`), or an `_in`-checked column is omitted ([Access control → Insert checks](/access-control#insert-checks)). On the batch path both this and the column error above are per-record failures reported in `results`, not whole-request rejections | | 403 | `{"error":"policy check references column \"x\", which table \"t\" does not have"}` (also `… which is materialized and cannot be inserted`, the same for `alias`, and `… which is ephemeral and is never stored`) | A **policy misconfiguration**, not a bad request: the role's `check` names a column the table lacks, one ClickHouse computes, or an `EPHEMERAL` one. None can be enforced — the published row carries one slot per insertable column, and an ephemeral column is never stored — so the check would have passed silently while enforcing nothing. Like the rejections above this is decided per record, so the **status depends on the body shape**: a single-object request answers `403`, while a batch answers `200` and carries the same message against each record in `results`. It fires on **every** insert by that role until the policy or the table is corrected, and names every offending column rather than one of them. `wavehouse validate` cannot catch it: it never sees the ClickHouse schema | -| 404 | `{"error":"unknown table: ..."}` | Table not found in ClickHouse schema | +| 404 | `{"error":"unknown table: ..."}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 16777216 bytes"}` | Request body over the 16 MiB cap | | 415 | `{"error":"no Content-Type: ingest requires one of application/json, application/x-ndjson, …"}` (declared variant: `Content-Type "text/plain": ingest requires one of …` — see the note above on how declarations are echoed; conflicting variant: `conflicting Content-Type declarations "application/json", "application/x-ndjson": ingest reads one format per request, and requires one of …`) | The request declared no `Content-Type`, one whose media type is unsupported or does not parse, a comma-bearing value that does not parse as a single media type, or repeated header lines that disagree — different formats, or one supported and one not. Checked before the body is parsed | | 500 | `{"error":"dedupe failed"}` | Deduplication backend error | +| 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or [no pool for it](/settings-directory#clickhouse)), so whether the table exists is not known; `Retry-After: 5`. Decided before the body is read | | 500 | `{"error":"publish failed"}` | Message queue error | | 503 | `{"error":"service unavailable"}` | NATS JetStream stream full (backpressure). Response includes `Retry-After: 30` header. | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | @@ -427,6 +430,8 @@ The route is mounted under `/v1/ops/*`, behind the `RequireAdmin` gate: only a c `/v1/ops/query` is the only sanctioned surface for non-insert mutations (the ingest pipeline is insert-only). Granting raw-SQL access to a non-admin role via the policy engine is no longer supported: authenticate with the admin role (`admin_role`). +An optional `?tenant=` names the [tenant](/deployment#the-nested-settings-directory) whose ClickHouse the SQL runs against — its own database, credentials and HTTP wiring; without it the SQL runs against tenant `0`'s, which is the whole settings directory unless it is nested. The parameter is parsed as strictly as on the [schema routes](#get-v1opsschema--list-all-table-schemas): `400` for a query string that does not parse or an empty, repeated or malformed id, `404` for an unknown tenant, `503` for one whose settings folder was rejected — all decided before the body is read. A tenant on no ClickHouse pool ([no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused) answers `503` `{"error":"no ClickHouse connection is open for this tenant"}` with `Retry-After: 30`. + **Request:** ```json @@ -460,6 +465,9 @@ The earlier handler accepted a `params` array bound to `?` placeholders; the HTT | Status | Body | Cause | | ------ | ---- | ----- | +| 400 | `{"error":"invalid ?tenant: …"}` / `{"error":"invalid query string: …"}` | The query string does not parse (`?tenant=acme;x=1`, a bad `%` escape), or `tenant` is empty, repeated, or not a tenant id — parsed as strictly as on the [pipe reads](#get-v1opspipes--list-named-pipes) | +| 404 | `{"error":"unknown tenant: "}` | No such tenant | +| 503 | `{"error":"tenant settings are invalid"}` | The tenant's settings folder was rejected | | 400 | `{"error":"invalid json"}` | Malformed request body | | 400 | `{"error":"missing sql"}` | Missing `sql` field | | 400 | `{"error":""}` | ClickHouse rejected the statement with a 4xx (bad SQL, missing table, type error, …). The body carries ClickHouse's own error text verbatim, e.g. `Code: 60. DB::Exception: Table default.x doesn't exist.`. The proxy maps any ClickHouse 4xx to HTTP 400 — caller-fault, the request itself is what's wrong. | @@ -468,6 +476,7 @@ The earlier handler accepted a `params` array bound to `?` placeholders; the HTT | 502 | `{"error":""}` | ClickHouse returned a 5xx (internal error, overloaded, etc.). The proxy maps any ClickHouse 5xx to HTTP 502 — gateway-fault, the upstream service had a problem. Same body convention: ClickHouse's text is forwarded as-is. | | 502 | `{"error":"clickhouse request failed: ..."}` | Transport-level failure reaching ClickHouse (connection refused, timeout, the upstream went away mid-request) | | 502 | `{"error":"clickhouse response exceeded N bytes; ..."}` | Response body exceeded the 64 MiB memory-safety cap. Narrow the query, add a `LIMIT`, or use `FORMAT JSONEachRow` with a streaming client outside WaveHouse. | +| 503 | `{"error":"no ClickHouse connection is open for this tenant"}` | The tenant is on no ClickHouse pool — [no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused — so the SQL cannot run; `Retry-After: 30`, a settings reload retries the pool | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while tenant `0`'s JWKS has not been fetched yet (the ops tree verifies as tenant `0`); refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | **curl example:** @@ -541,8 +550,10 @@ The inbound request body is capped at 1 MiB; a body over the cap is rejected wit | 403 | `{"error":"forbidden"}` | Role lacks select permission on table | | 403 | `{"error":"column \"x\" not allowed"}` | Column denied by policy | | 403 | `{"error":"aggregation \"x\" not allowed"}` | Aggregation fn denied by policy | -| 404 | `{"error":"unknown table: x"}` | Table not found | +| 404 | `{"error":"unknown table: x"}` | Table not found in the tenant's discovered schema | | 413 | `{"error":"request body exceeded 1048576 bytes"}` | Request body over the 1 MiB cap | +| 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet, so whether the table exists is not known; `Retry-After: 5` | +| 503 | `{"error":"no ClickHouse connection is open for this tenant"}` | The tenant is on no ClickHouse pool — [no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused — so the query cannot run; decided ahead of the cache, so nothing cached before is served either; `Retry-After: 30`, a settings reload retries the pool | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while the tenant's JWKS has not been fetched yet; refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | --- @@ -573,6 +584,7 @@ The POST parameter body is capped at 1 MiB; a body over the cap is rejected with | Status | Body | Cause | | ------ | ---- | ----- | | 404 | `{"error":"pipe not found"}` | Pipe name not registered | +| 503 | `{"error":"no ClickHouse connection is open for this tenant"}` | The tenant is on no ClickHouse pool — [no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused; decided ahead of the cache; `Retry-After: 30` | | 403 | `{"error":"forbidden"}` | Role not in pipe's `allowed_roles` (and not the admin role). Fails closed: a request with no role (no token, or a JWT missing `auth.role_claim`) is denied unless a `default_role` resolves it into the list; a pipe with no `allowed_roles` denies everyone but the admin role. | | 400 | `{"error":"missing required parameter: x"}` | Required parameter not supplied | | 400 | `{"error":"parameter \"x\": unsupported parameter type object"}` | A non-scalar value with no SQL literal form — a JSON object, whether supplied directly or nested as an array element. A JSON **array** is valid and renders as an `IN`-style `(…)` list. | @@ -649,7 +661,7 @@ No admin endpoint in this section accepts a request body — they are reads and #### `GET /v1/ops/schema` — List All Table Schemas -Returns all discovered ClickHouse table schemas. +Returns all discovered ClickHouse table schemas of one tenant. All three schema routes take an optional `?tenant=` naming the [tenant](/deployment#the-nested-settings-directory) whose schema is read or refreshed; without it they address tenant `0`, which is the whole settings directory unless it is nested. The query string is parsed strictly, with the pipe reads' answers: `400` for a query that does not parse or an empty, repeated or malformed id, `404` for an unknown tenant, `503` for one whose settings folder was rejected. A tenant whose first discovery has not succeeded yet answers `503` `{"error":"schema not loaded yet"}` with `Retry-After: 5` rather than an empty list, which would read as "no tables". **Response:** @@ -693,20 +705,26 @@ Per-column fields: `name`, `type` and `is_nullable` describe the column; `positi | ------ | ---- | ----- | | 401 | `{"error":"invalid token"}` / `{"error":"token expired"}` | A present-but-invalid/expired token was supplied and denied (the gate surfaces the token reason) | | 403 | `{"error":"forbidden"}` | Caller's role is not the policy `admin_role` (`"admin"` by default) | -| 404 | `{"error":"table not found"}` | Table not in discovered schemas | +| 400 | `{"error":"invalid ?tenant: …"}` / `{"error":"invalid query string: …"}` | The query string does not parse (`?tenant=acme;x=1`, a bad `%` escape), or `tenant` is empty, repeated, or not a tenant id — parsed as strictly as on the [pipe reads](#get-v1opspipes--list-named-pipes) | +| 404 | `{"error":"unknown tenant: "}` | No such tenant | +| 503 | `{"error":"tenant settings are invalid"}` | The tenant's settings folder was rejected | +| 404 | `{"error":"table not found"}` | Table not in the tenant's discovered schema | +| 503 | `{"error":"schema not loaded yet"}` | The tenant's first schema discovery has not succeeded yet (its ClickHouse unreachable, or no pool for it); `Retry-After: 5` | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while tenant `0`'s JWKS has not been fetched yet (the ops tree verifies as tenant `0`); refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | --- #### `POST /v1/ops/schema/refresh` — Refresh Schemas -Triggers an immediate re-discovery of ClickHouse table schemas, then returns the refreshed schema list (same array shape as `GET /v1/ops/schema`). Admin-only, like the rest of this section. +Triggers an immediate re-discovery of the `?tenant=`'s ClickHouse table schemas (tenant `0`'s without it), then returns the refreshed schema list (same array shape as `GET /v1/ops/schema`). Admin-only, like the rest of this section. **Error responses:** | Status | Body | Cause | | ------ | ---- | ----- | | 401 / 403 | as above | Not the admin role | +| 400 / 404 / 503 | as on `GET /v1/ops/schema` | The `?tenant=` could not be resolved | +| 503 | `{"error":"no ClickHouse connection is open for this tenant"}` | The tenant is on no ClickHouse pool — [no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused — so nothing can be discovered; `Retry-After: 30`, a settings reload retries the pool | | 500 | `{"error":"refresh failed"}` | ClickHouse discovery query failed | | 503 | `{"error":"token verifier not ready: the tenant's JWKS has not been fetched yet"}` | A token was supplied, with no valid operator key, while tenant `0`'s JWKS has not been fetched yet (the ops tree verifies as tenant `0`); refused before any policy runs, with a `Retry-After: 30` header — see [Authentication](#authentication) | diff --git a/docs/src/content/docs/architecture.md b/docs/src/content/docs/architecture.md index 69171d5b..0e1c6476 100644 --- a/docs/src/content/docs/architecture.md +++ b/docs/src/content/docs/architecture.md @@ -55,7 +55,7 @@ internal/ ├── app/ Process wiring: build every component, run them under one errgroup, release in reverse ├── auth/ JWT/JWKS authentication middleware (HMAC or JWKS, role extraction) ├── cache/ Query cache: Ristretto L1 + the tenant-led version index -├── chconn/ The one ClickHouse driver.Conn every consumer holds; reload swaps the connection behind it +├── chconn/ One ClickHouse pool per connection tuple among the served tenants, reconciled on reload under the ceiling ├── chsql/ Shared ClickHouse SQL helpers (identifier quoting, bind-safety) ├── config/ YAML + env var configuration loading ├── dedupe/ Optional deduplication (Pebble) @@ -77,20 +77,20 @@ The API layer uses [Chi](https://github.com/go-chi/chi) for routing with Request - **router.go** — Route definitions. Public: `/livez`, `/readyz`, and the content-free `/v1/health` SDK ping (plus the permanent `/healthz` alias and the deprecated `/health`, `/ready` aliases). Policy-gated: `/v1/ingest?table={table}`, `/v1/query?table={table}` (structured), `/v1/pipes/{name}` (named pipes), `/v1/stream`. Admin-only (`RequireAdmin` — role == `policy.admin_role`, or a request bearing the operator key's operator bit, which passes even under a nil policy; over a nested settings directory `NewRouter` mounts the gate with no policy at all, whatever `Dependencies.PolicySource` was wired, so the operator key alone passes): `/v1/ops/schema/*`, `/v1/ops/dlq/stats`, `GET /v1/ops/pipes[/{name}]`, `/v1/ops/settings/reload`, `/v1/ops/query` (raw SQL — same gate as the rest of `/v1/ops/*`). - **auth middleware** — the JWT/JWKS authentication middleware is its own package, [`auth/`](#auth--authentication); the router runs it on every `/v1/*` route. -- **tenant.go** — `TenantMW` resolves the request's tenant ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: the [`X-Tenant-ID`](/deployment#multi-tenant-deployments) header (absent means `tenant.Default`), validated by `tenant.Parse` (`400`), looked up in the `settings.Registry` (`resolveStore`: `404` for an id it does not hold, a bare `503` for a tenant whose folder was rejected — the findings stay out of a body answered before authentication), and the resolved `*settings.Store` stored in the request context (`WithStore` / `StoreFromContext` — here rather than in `tenant/`, because `settings` names `tenant.ID`). A handler reads the store once and passes it down as an argument — the per-tenant getters it holds take it as a parameter (`(*settings.Store).Policy`, `.DedupeFor`, `.DefaultMaxRows`, … in production) — and nothing below a handler reads the context; a tenant route reached without a resolved store answers `500` rather than fall back to a tenant. The probes, `/version`, the metrics path, and `/v1/ops/*` are tenant-exempt; an ops route that addresses one tenant — the admin pipe reads and the settings reload — names it in `?tenant=` (`opsTenant`), parsed strictly so that a query `url.ParseQuery` would half-read is a `400` rather than a read of the default tenant, which is what it means when absent on the reads (on the reload, absent is the whole directory). +- **tenant.go** — `TenantMW` resolves the request's tenant ahead of the auth middleware on every `/v1` route outside `/v1/ops/*`: the [`X-Tenant-ID`](/deployment#multi-tenant-deployments) header (absent means `tenant.Default`), validated by `tenant.Parse` (`400`), looked up in the `settings.Registry` (`resolveStore`: `404` for an id it does not hold, a bare `503` for a tenant whose folder was rejected — the findings stay out of a body answered before authentication), and the resolved `*settings.Store` stored in the request context (`WithStore` / `StoreFromContext` — here rather than in `tenant/`, because `settings` names `tenant.ID`). A handler reads the store once and passes it down as an argument — the per-tenant getters it holds take it as a parameter (`(*settings.Store).Policy`, `.DedupeFor`, `.DefaultMaxRows`, … in production) — and nothing below a handler reads the context; a tenant route reached without a resolved store answers `500` rather than fall back to a tenant. The probes, `/version`, the metrics path, and `/v1/ops/*` are tenant-exempt; an ops route that addresses one tenant — the admin pipe reads, the schema routes, the raw-SQL proxy and the settings reload — names it in `?tenant=` (`opsTenant`, and `opsStore` over it for the routes that need the tenant's store), parsed strictly so that a query `url.ParseQuery` would half-read is a `400` rather than a read of the default tenant, which is what it means when absent on the reads (on the reload, absent is the whole directory). - **pipes.go** — Named query pipe handlers: admin listing (`GET /v1/ops/pipes[/{name}]`, read per request from its `pipes.Source`) and execution with parameter binding. `pipes.json` is the only write path. - **structured_query.go** — Handler for `POST /v1/query?table={table}`: validates query AST, enforces permissions, builds and executes SQL. - **ingest.go** — Accepts `POST /v1/ingest?table={table}` in three body shapes: one flat JSON object, a JSON array of them, or NDJSON. The **required** `Content-Type` chooses the format *family* — `application/json` versus the four NDJSON spellings — and within the JSON family the body's first non-whitespace byte picks array versus single object; the bytes never choose the family. Anything that is not exactly one readable media type is a `415`, decided before the body is read: the header is parsed per RFC 9110 §8.3, and because `Content-Type` is a singleton field, repeated header lines must all resolve to the same format and a value carrying a comma is refused unless the value as a whole parses as one media type — a comma inside a *quoted* parameter value is data, so `application/json; a=", application/x-ndjson; b="` is accepted. It then reads the whole (`MaxBytesReader`-capped) body into a pooled buffer and runs the per-format record readers over those bytes, so the `413` lands before any record is processed and peak memory per request is O(body) rather than O(record). Then it validates each record against the discovered schema, optional dedup, and publishes each row through `mq.Publisher` on `mq.Topic{Tenant, Table, Scope}` (the request's tenant, read off its resolved store — `store.Tenant()` — and raw names; the subject it becomes is `internal/mq`'s; a full queue comes back as `mq.ErrQueueFull`, which is the `503` + `Retry-After`). When dedup is on, a row missing the configured `id_field` can't be deduped: it is logged at `WARN` and counted by `wavehouse_ingest_dedupe_missing_id_total` (labeled by `table`), then published un-deduped — or rejected when `dedupe.require_id` is set ([#219](https://github.com/Wave-RF/WaveHouse/issues/219)). -- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to ClickHouse's HTTP interface. **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` (the Go-side type conversion lives in the structured-query / pipes path, not here). +- **query.go** — Proxies raw SQL for `POST /v1/ops/query` straight to the `?tenant=`'s ClickHouse HTTP interface (`chconn.Pools.Target` by the resolved store's tenant; the zero target — no pool — is a `503` with `Retry-After`). **Not cached** — sets `Cache-Control: no-store` so every request hits ClickHouse; DateTime is rendered ISO-8601 via `date_time_output_format=iso` (the Go-side type conversion lives in the structured-query / pipes path, not here). - **stream.go** — Real-time streaming via SSE. Callers select a table with the `?table=` query parameter. Each connection registers one `Subscriber` (the `stream/` package) with both the event `Hub` (under its `(topic, role)`) and the shared keepalive wheel, then drains both from a single byte-pump — so idle streams keep emitting `:` keepalive comments (surviving reverse-proxy idle timeouts) while live events arrive already projected and serialized. Per-event projection/serialization happens **once per role** in the `Hub`, not once per subscriber ([#294](https://github.com/Wave-RF/WaveHouse/issues/294)); the handler also snapshots the connection's JWT claims onto the `Subscriber`, which the `Hub` evaluates per subscriber when the role carries a row-level `filter` ([#319](https://github.com/Wave-RF/WaveHouse/issues/319)). Gap-fill replay (`mq.Replayer.ReplaySince` on the connection's `mq.Topic` — a `DeliverByStartTime` consumer inside `internal/mq`) stays per-connection (low-volume, one-time on connect). -- **schema.go** — Schema discovery API: list all schemas, get one table, trigger refresh. +- **schema.go** — Schema discovery API of one tenant, the `?tenant=` (`opsStore`): list all schemas, get one table, trigger refresh. `lookupSchema`, shared with the ingest and structured-query handlers, is the one reading of a `SchemaRegistry.Lookup` miss: `503` with `Retry-After` before the tenant's first discovery (`ErrNotLoaded`, or no registry built yet), `404` for a table the discovered schema lacks; the list answers the same `503` rather than `[]`. A refresh of a tenant on no pool (`discovery.ErrNoConnection`) is a `503` with `Retry-After` too. The handlers hold `RegistrySource`, `func(*settings.Store) *discovery.SchemaRegistry`, and the query paths a `func(*settings.Store) driver.Conn` beside it — each resolves the request's tenant per call, and a nil connection (a tenant no pool could be opened for, such as by the connection ceiling) is a `503` ahead of the cache, so nothing cached before is served. - **dlq.go** — DLQ stats endpoint (`GET /v1/ops/dlq/stats`): asks `mq.DeadLetterStats.DeadLetterCounts` for the per-table parked counts (optionally one table) and the total. A dead-letter queue that does not exist (`mq.ErrNoDeadLetterQueue`) reads as empty; any other failure to read it is a 500. The queue itself is `internal/mq`'s. -- **health.go** — Liveness (`/livez`), readiness (`/readyz`), and a content-free `Online` ping (`/v1/health`, the SDK's public liveness check); `/healthz` is a permanent alias of `/livez`, and `/health`/`/ready` are deprecated aliases. All three consult an optional `BootState` so they can return 503 while boot-time schema discovery is still failing in the retry loop (see `internal/app`); once `BootState.Set(nil)` fires, `/livez` returns 200 and stays there. `/readyz` additionally pings ClickHouse each call; `/v1/health` deliberately does not. +- **health.go** — Liveness (`/livez`), readiness (`/readyz`), and a content-free `Online` ping (`/v1/health`, the SDK's public liveness check); `/healthz` is a permanent alias of `/livez`, and `/health`/`/ready` are deprecated aliases. All three consult an optional `BootState` so they can return 503 while boot-time schema discovery is still failing in the retry loop (see `internal/app`; over a nested directory, while no tenant's has succeeded); once `BootState.Set(nil)` fires, `/livez` returns 200 and stays there. `/readyz` additionally runs a `Ping` each call — `chconn.Pools.Ping` in production: every open pool at once, ready at the first answer, every pool's error joined when none answers; `/v1/health` deliberately does not. ### `app/` — Process wiring -- **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse connection, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. -- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker and `tenant.Default` for the schema registry until story 6 — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant the registry knows (`Registry.Known`, a rejected tenant included, since it comes back into service with the entries it has), because the tenants share one ClickHouse until story 6 and an insert changes what each of them reads; a tenant removed and restored inside a TTL is the residual, story 3's. The resources one process still has one of (ClickHouse connection, MQ byte budget) follow the default tenant: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`; the zero value when a nested directory has never served a tenant `0`, warned about once at boot), and `onDefaultAdopt` runs their hooks only after a reload that adopted it, so another tenant's reload never moves them and a `0` folder that a reload rejects or removes leaves all of them as they were — the hook-reconciled ones and the one read per request (the ops gate's admin role) alike. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. Two settings are shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)); and the sweeper keeps the longest `stream.gap_window_minutes` (`longestGapWindow`, read every sweep), since the ingest queue is one stream and a purge is one bound over it — a stream per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 5b) gives each its own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over a factory that opens each tenant's embedded Pebble store at `data_dir//dedupe` whatever the directory's shape (the four files are tenant `0`; an earlier layout's `data_dir/pebble` is moved to tenant `0`'s once by `moveLegacyDedupeStore`, both present is left alone and warned about, and a failed move refuses boot) — the factory being the one place Pebble is named, so a shared backend behind the same switch is a wiring change — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its data left on disk when the tenant is switched off, rejected, or removed. A store that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for that tenant alone over a nested one. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. The `mq.max_bytes_gb` hook only hands the adopted budget to `mq.Broker.SetMaxBytes` under the App's stop context; how it is split across the streams, the time bounds, and the rollback are `internal/mq`'s. +- **app.go** — `New(ctx, Options)` builds every component from the boot config (`Options.Config`) and the settings directory it names, in dependency order: settings registry, observability, ClickHouse pools, schema discovery, the dedupe stores, embedded NATS (ingest + DLQ streams), cache, sweeper, streaming (hub, MQ→hub bridge, keepalive wheel), ingest worker, auth, reload triggers, HTTP. Each is one `component` value — what it opens, what it loops, what it releases — so a failure part-way releases what was already opened and returns the error. `Run(ctx)` drives every loop under one `errgroup` until `ctx` is canceled (a clean stop: every loop drains, the API server and the ingest worker within `server.shutdown_timeout`; open SSE streams are ended as the drain begins rather than waited on) or a component fails, which stops the rest and returns that error. `Close(ctx)` releases what `New` opened, newest first, under the caller's release budget (`ReleaseTimeout`, 5s), a real bound: a remote implementation's close gives up at the deadline itself, and a close that ignores the context (the local stores) is abandoned at it, with the components below it left unreleased rather than overlapping it, both named in the error — and then flushes telemetry under its own 3s budget, so the flush that reports on the stop is never handed a deadline a slow close already spent. The SIGHUP registration is released last of all. `Handler`, `Registry`, and `MQ` expose the pieces a harness needs; `Options.Listener` lets one serve the API on its own listener instead of `server.port`. +- **wire.go** — one `wire*` function per component, each handed the settings registry whole and deriving the per-call getters the internal packages take (`DLQFor`, `DedupeFor`, `GapWindow`, …) and registering its `AfterAdopt` hook there where it has one. Those wiring functions are where the per-tenant registry of [#583](https://github.com/Wave-RF/WaveHouse/issues/583) is injected, not `main`: `wireSettings` opens the `settings.Registry`, the HTTP handlers get store-keyed getters (method expressions such as `(*settings.Store).Policy`), and `perTenant` adapts a store accessor into the `func(tenant.ID) T` getter the async packages take, with the tenant each message's `mq.Topic` names for the stream hub and the ingest worker — a tenant the registry is not serving is logged and read as the zero value, except in `dlqFor`, the ingest worker's DLQ switch, where it reads as on so a message the worker cannot read is parked rather than dropped. The ClickHouse pools (`chconn.Pools`) and the per-tenant schema registries (`discoveries`, in `discoveries.go`) are reconciled from `AfterAdopt` after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): `wireClickHouse` builds each served tenant's `chconn.Member` from its store and logs what the reconcile refused; `wireDiscovery` builds a registry over `pools.For` for each newly served tenant — a flat directory's tenant `0` refreshed synchronously first, as before — runs its loop under the App's stop context, stops the loop of a tenant no longer served, and drives the `BootState` from the first tenant's first discovery, sticky from there. The handlers resolve both per request through store-keyed getters (`chConnFor`, `registryFor`, `chTargetFor`, `queryTimeout`), the hub and the ingest worker through tenant-keyed ones (`discoveries.For`, `pools.Target`) called with the tenant the message's topic names; a tenant on no pool is an untyped nil connection, the handlers' `503`. The ingest worker is handed the cache through `sharedTables`, which bumps each namespace the worker invalidates under every tenant on the same ClickHouse address and database (`pools.SharingTables`), and the pools hook orphans the table-keyed cache — the structured-query results — of a tenant back on a pool after an absence (`Cache.InvalidateTenant`), since it was out of that fan-out while away, and of a tenant moved to another address or database, since it now reads other tables (both returned by `Pools.Reconcile`). The one resource a process still has one of, the MQ byte budget, follows the default tenant: `defaultSetting` reads the store tenant `0` last adopted (`App.defaultStore`; the zero value when a nested directory has never served a tenant `0`, warned about once at boot), and `onDefaultAdopt` runs its hook only after a reload that adopted it, so another tenant's reload never moves it and a `0` folder that a reload rejects or removes leaves it as it was — like the one setting read per request that follows tenant `0`, the ops gate's admin role. The auth verifiers are per tenant: `wireAuth` builds one for each tenant being served, its `AfterAdopt` hook reconfigures the adopted tenants' (rebuilt only when their wiring changed) and prunes the ones no longer served, and the operator key's admin role is read from the request tenant's policy. Two settings are shared by folding over the tenants being served rather than by following tenant `0`: the keepalive wheel runs at the shortest `stream.keepalive_interval` among them (`shortestKeepalive`), re-derived after every reload the registry applies — an adoption, a rejection, or a removal — so a dropped tenant's interval leaves the wheel at once ([#597](https://github.com/Wave-RF/WaveHouse/issues/597)); and the sweeper keeps the longest `stream.gap_window_minutes` (`longestGapWindow`, read every sweep), since the ingest queue is one stream and a purge is one bound over it — a stream per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 5b) gives each its own. The dedupe stores are per tenant ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 7): `wireDedupe` builds a `dedupe.Stores` over a factory that opens each tenant's embedded Pebble store at `data_dir//dedupe` whatever the directory's shape (the four files are tenant `0`; an earlier layout's `data_dir/pebble` is moved to tenant `0`'s once by `moveLegacyDedupeStore`, both present is left alone and warned about, and a failed move refuses boot) — the factory being the one place Pebble is named, so a shared backend behind the same switch is a wiring change — and one reconcile closure, the boot apply and the `AfterAdopt` hook alike, sets every store to what the registry says: open exactly when its tenant is served with `dedupe.enabled` on, closed with its data left on disk when the tenant is switched off, rejected, or removed. A store that cannot open follows the registry's rule for the shape: fatal at boot over a flat directory, fail-closed for that tenant alone over a nested one. The ingest handler picks the tenant's store off the request's `settings.Store` (`Store.Tenant()`). The reload triggers only start in `Run`, after `New` has registered every hook, so the watcher's first reload already drives all of them: SIGHUP in both shapes, the directory watcher for a flat directory only. The `mq.max_bytes_gb` hook only hands the adopted budget to `mq.Broker.SetMaxBytes` under the App's stop context; how it is split across the streams, the time bounds, and the rollback are `internal/mq`'s. ### `stream/` — SSE keepalive & fan-out @@ -109,9 +109,9 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `cache/` — Query Cache -- **cache.go** — `Cache` interface: `Get`, `Set`, `Invalidate`, `Close`, plus `QueryTimeToTTL`, which sets a result's TTL from how long its query took (10 s floor, 1 h ceiling). Every entry is keyed by the caller's query key — `:query:`, built by the two cached handlers in `api/` (`queryCacheKey`, with the tenant read off the request's store — `settings.Store.Tenant`), which use it as their [singleflight](https://pkg.go.dev/golang.org/x/sync/singleflight) key too — folded with the `Namespace`s the result depends on, each naming its tenant, table and scope: one for a structured query, none yet for a pipe (a pipe's table dependencies are [#343](https://github.com/Wave-RF/WaveHouse/pull/343)). +- **cache.go** — `Cache` interface: `Get`, `Set`, `Invalidate`, `InvalidateTenant`, `Close`, plus `QueryTimeToTTL`, which sets a result's TTL from how long its query took (10 s floor, 1 h ceiling). Every entry is keyed by the caller's query key — `:query:`, built by the two cached handlers in `api/` (`queryCacheKey`, with the tenant read off the request's store — `settings.Store.Tenant`), which use it as their [singleflight](https://pkg.go.dev/golang.org/x/sync/singleflight) key too — folded with the `Namespace`s the result depends on, each naming its tenant, table and scope: one for a structured query, none yet for a pipe (a pipe's table dependencies are [#343](https://github.com/Wave-RF/WaveHouse/pull/343)). - **local.go** — `LocalCache`, the in-process L1 on [Ristretto](https://github.com/dgraph-io/ristretto): one pool shared by every tenant (a heavier tenant holds more of it), sized by the boot config's `cache.l1_max_cost`. -- **version_manager.go** — `VersionManager`, the invalidation index behind `Invalidate`: a namespace key is `.
.
.`, and a query key is folded with each dependency's namespace key and namespace version, so bumping a table (a scopeless write) or one scope — scope is reserved and empty today, so every write is the whole-table bump — orphans every dependent entry without touching the pool. The tenant leads every key ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8): the same table under two tenants is two namespaces, so a bump through `Invalidate` under one tenant never touches — and a read under one tenant is never served — the other's results, and the flat directory's single tenant simply carries the `0` prefix. The index stays per tenant while the tenants still share one ClickHouse (until story 6); the cross-tenant invalidation an insert needs until then is not the index's but the wiring's: `internal/app` hands the ingest worker a cache (`sharedTables`) that repeats each bump under every tenant the registry knows, a rejected one included. +- **version_manager.go** — `VersionManager`, the invalidation index behind `Invalidate` and `InvalidateTenant`: a namespace key is `..
.
.`, and a query key is folded with each dependency's namespace key and namespace version, so bumping a table (a scopeless write) or one scope — scope is reserved and empty today, so every write is the whole-table bump — orphans every dependent entry without touching the pool. The tenant leads every key ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 8): the same table under two tenants is two namespaces, so a bump through `Invalidate` under one tenant never touches — and a read under one tenant is never served — the other's results, and the flat directory's single tenant simply carries the `0` prefix. `BumpTenant` (behind `InvalidateTenant`) advances the tenant version that leads every namespace key of one tenant, orphaning its every namespace, and every cached query keyed by one, in one step (a pipe result names no table and keeps its TTL) — a table no bump ever keyed included, which an enumeration of the index would miss — for a tenant back on a pool after an absence from the fan-out, or moved to another address or database (story 6). The index is per tenant; the cross-tenant invalidation an insert into a shared table needs is not the index's but the wiring's: `internal/app` hands the ingest worker a cache (`sharedTables`) that repeats each bump under every tenant on the same ClickHouse address and database. ### `config/` — Configuration @@ -129,7 +129,7 @@ The SSE fan-out, factored out of `api/` so the delivery hot path ([#294](https:/ ### `discovery/` — Schema Discovery & Validation -- **discovery.go** — `SchemaRegistry` queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), reads each column's `default_expression` and 1-based `position` alongside its type, discovers the server's default time zone (`SELECT timezone()`) and bakes every `DateTime`/`DateTime64` column's canonicalization spec (precision + resolved zone) into the cached schema, so the per-record ingest path parses no type strings and loads no zones ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). Supports periodic auto-refresh, on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary). Thread-safe via `sync.RWMutex`. +- **discovery.go** — `SchemaRegistry`, one per tenant since [#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6, over a `Source` read once per refresh: the tenant's pool's connection and the database that pool was opened for, one snapshot (`App.discoverySource` over `chconn.Pools.For` in production, so a reload that repoints the tenant applies to the next refresh, and a refused move keeps discovering the database the tenant's queries and inserts still use; no pool is `ErrNoConnection`), queries `system.columns` to discover ClickHouse table schemas, keeping each column's `default_kind` so `IsInsertable` / `InsertableColumns` / `InsertableColumnNames` (memoized per table at refresh) can decide the insertable subset the ingest envelope and the SSE announcement are both built from. Each refresh also records the server version (`SELECT version()`), joins `system.tables` for each table's `create_table_query` (kept in-process as `TableSchema.DDL` and marked `json:"-"` — an external-engine table renders its wiring in that statement — endpoint, bucket/host, database, username, access key id — so it must never reach `/v1/ops/schema`; ClickHouse masks the password as `[HIDDEN]` from ~23.9, so what is withheld here is the topology), reads each column's `default_expression` and 1-based `position` alongside its type, discovers the server's default time zone (`SELECT timezone()`) and bakes every `DateTime`/`DateTime64` column's canonicalization spec (precision + resolved zone) into the cached schema, so the per-record ingest path parses no type strings and loads no zones ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)). `Lookup` tells the two misses apart — `ErrNotLoaded` before the first successful refresh, `ErrUnknownTable` after — where `Get` answers nil for both (the stream hub's fail-closed reading). Supports periodic auto-refresh (`StartAutoRefresh`, the first tick at a random point within the interval so tenants adopted together do not refresh together, the cadence re-read after every tick), on-demand refresh, and `RetryRefresh` (boot-time exponential backoff loop used by `internal/app` so a transiently unreachable ClickHouse doesn't crash-loop the binary); a loop's failed attempt counts in `wavehouse_schema_refresh_failures_total{tenant}`. Thread-safe via `sync.RWMutex`. - **timestamp.go** — `CanonicalizeTimestamps(schema, data)` rewrites every parseable value in a top-level `DateTime`/`DateTime64` column to the canonical RFC 3339 UTC wire form before the event is published ([#372](https://github.com/Wave-RF/WaveHouse/issues/372)): zone-less values are interpreted in the column's declared zone, else the discovered server default — ClickHouse's own rule, so the spelling changes but never the instant. Fail-open: an unparseable value or unresolvable zone passes through verbatim for ClickHouse's own parser to judge; ingest never rejects a record over its timestamp spelling. `Column.TimeParser()` exposes the same grammar as a value→instant parser (nil for a column with no resolved timestamp spec — a non-timestamp column, or one whose declared zone couldn't be loaded), which the stream row-filter uses so filter constants and canonicalized payloads can't disagree on the instant ([#381](https://github.com/Wave-RF/WaveHouse/issues/381)). - **validation.go** — `Validate(schema, data)` checks incoming JSON against the discovered schema: unknown fields, type compatibility, missing required columns, null handling. Also exports the type classifiers `IsNumericType` / `IsStringType` and the storage-model classifier `NumericStorageOf` (all unwrapping `Nullable`/`LowCardinality`; the latter yields a numeric column's float width, `Decimal` scale, or integer exactness), which — together with `Column.TimeParser` from timestamp.go — seed the stream row-filter's `policy.ColumnSpec` comparison. - **discovery_test.go** — Unit tests for validation logic. @@ -190,11 +190,11 @@ The hot-reloadable half of configuration: a directory of four JSON files (`confi ### `tenant/` — Tenant Identifier -- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes, and not `nats` or `pebble` in any letter case, the entries `data_dir` keeps for itself beside the tenants' own directories. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper folds over the tenants served; the schema registry is still constructed with `tenant.Default` (story 6). +- **tenant.go** — `ID`, a validated string (never a number: a 19-digit id already rounds as a float64), and `Parse`, the one grammar that makes an id safe both as a folder name and as a message-queue subject token: ASCII letters, digits, `_`, `-`, at most `MaxLen` (64) bytes, and not `nats` or `pebble` in any letter case, the entries `data_dir` keeps for itself beside the tenants' own directories. `Default` (`"0"`) is the tenant a request without the header resolves to; `Header` is `X-Tenant-ID`. The package imports nothing from the rest of the repository, so any package can name a tenant. HTTP handlers receive the tenant as its resolved `*settings.Store`, which knows its id (`Store.Tenant`) for the topics they publish and subscribe on; the stream hub and the ingest worker read each event's tenant off its `mq.Topic` — the leading subject token — and their settings getters take it as a parameter, which `internal/app` resolves through the registry; the sweeper folds over the tenants served; each served tenant has a schema registry of its own, built with its id (story 6). -### `chconn/` — ClickHouse Connection Manager +### `chconn/` — ClickHouse Connection Pools -- **chconn.go** — `Manager` is a `driver.Conn` whose backing connection is swapped by `Reconfigure(Params)` after a settings reload changes the ClickHouse wiring (`clickhouse.addr` / `http_port` / `http_scheme` / `database` / `username` / `query_timeout` / `tls` / `headers` / `max_open_conns` / `max_idle_conns`, combined with the boot-config password). Like `clickhouse.Open` it never dials, so boot tolerates an unreachable ClickHouse (schema discovery retries) and a bad address surfaces where reachability is already handled (`/readyz`, query errors); certificate files it cannot read or parse are the one thing it refuses, and a reload then keeps the previous connection. The `tls` block becomes one `tls.Config`, handed to the driver when `tls.enabled` and carried on `Target()` for the https hop; it is rebuilt only when the block changes, so the HTTP consumers' transports are not. The replaced connection closes after a `query_timeout` grace so in-flight queries finish. `Target()` / `Database()` / `QueryTimeout()` expose the current wiring — URL, credentials, database, TLS config, headers — for the HTTP-interface consumers (ingest INSERTs, raw-SQL proxy), which take their `http.Client` from an `HTTPClients` cache keyed on that TLS config. +- **chconn.go** — `Pools` holds one `Manager` per distinct connection tuple among the served tenants — `Identity{Addr, Database, Username, Password, TLS}`, a plain comparable value, the map key — reconciled from the settings registry's `AfterAdopt` hook after every reload ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6): tenants naming one tuple share its pool, sized to the largest `max_open_conns` and `max_idle_conns` among them (`Sizes`); a tenant whose tuple changed is repointed; a tuple no tenant names is released after the longest `query_timeout` among the tenants it had; a changed largest ask is a `Resize` with the same grace. The walk keeps the boot config's `clickhouse.max_total_conns` — the ceiling on the open pools' `max_open_conns` together — at every step, tenants no longer served leaving first and what was refused placed once more at the end: a refused resize keeps the pool's size, and a tuple that cannot be opened (the ceiling, a certificate file that cannot be read, or options the driver refuses — the pool opens as the walk places its first tenant, at the largest ask among the tenants naming it when that fits the ceiling and otherwise at that tenant's own, so each of these is undone in place) leaves its tenants on the pool they had — `Params` and all, so their `Target` stays whole — or on none; `NewPools` refuses boot on any refusal, `Reconcile` returns them joined for the wiring to log, and the next reload retries. `Manager` is a `driver.Conn` over one tuple's pool whose backing connection `Resize` swaps; like `clickhouse.Open` it never dials, so boot tolerates an unreachable ClickHouse (schema discovery retries) and a bad address surfaces where reachability is already handled (`/readyz`, query errors). The `tls` block is the tuple's, read once into one `tls.Config` handed to the driver when `tls.enabled` and carried on each tenant's `Target` for the https hop. Resolution is per tenant: `For` (the `driver.Conn`, nil for a tenant on no pool — the wiring returns an untyped nil), `Target` (the tenant's own `http_port`, `http_scheme` and `headers` over its pool's host, credentials, database and TLS config, from the `Params` last applied for it), `SharingTables` (the tenants on the same address and database, whatever their user — the cache fan-out's rule) and `Ping` (every pool at once, nil at the first answer). The HTTP-interface consumers (ingest INSERTs, raw-SQL proxy) take their `http.Client` from an `HTTPClients` cache, one client per TLS config ever handed to it, since the proxy serves tenants on different configs in alternation. ### `chsql/` — ClickHouse SQL Helpers diff --git a/docs/src/content/docs/configuration.mdx b/docs/src/content/docs/configuration.mdx index e9f4ca12..44bc8022 100644 --- a/docs/src/content/docs/configuration.mdx +++ b/docs/src/content/docs/configuration.mdx @@ -53,7 +53,7 @@ Only the secret and the connection ceiling are boot config. The wiring — nativ | YAML Key | Env Var | Default | Description | | --- | --- | ------- | ----------- | | `clickhouse.password` | `WH_CH_PASSWORD` | *(empty)* | Authentication password, combined with the settings directory's `clickhouse.username` on every (re)connect. A secret, so it never lives in a tracked JSON file; rotating it is a restart. | -| `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the settings directory's `clickhouse.max_open_conns` must not exceed it. A settings pool above the ceiling refuses boot, naming both numbers in the error; a reload that raises the pool above it is refused and logged (the reload itself still reports `adopted`), and the connection keeps its previous wiring until the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. | +| `clickhouse.max_total_conns` | `WH_CH_MAX_TOTAL_CONNS` | `0` | Ceiling on the native ClickHouse connections the process holds open: the `max_open_conns` of the open pools — one per distinct connection tuple among the served tenants, see [the settings directory](/settings-directory#clickhouse) — must not add up to more. Pools above it at boot refuse to start, naming the sum and the ceiling; on a reload a pool resized above it is refused and keeps its size, and a new pool that would cross it is not opened — its tenants keep the pool they had, or have none — both logged (the reload itself still reports `adopted`) and retried by the next reload. `0` is no ceiling. Capacity is sized once per process, which is why it is boot config rather than a settings key. | ### Server-side resource limits diff --git a/docs/src/content/docs/deployment.md b/docs/src/content/docs/deployment.md index cf183d18..b92c333c 100644 --- a/docs/src/content/docs/deployment.md +++ b/docs/src/content/docs/deployment.md @@ -233,14 +233,14 @@ UID 65532 is the canonical distroless `nonroot` user; the same number works rega API servers in standalone mode expose liveness and readiness endpoints under the Kubernetes-convention names `/livez` and `/readyz`: -- `GET /livez` — Liveness probe. Returns 200 once the gateway has discovered ClickHouse table schemas at least once. Returns 503 with a diagnostic body while the boot-time schema discovery retry loop is still running (e.g. ClickHouse unreachable, target database missing). After successful boot, `/livez` stays 200 — transient ClickHouse blips at runtime are reflected in `/readyz`, not `/livez`. -- `GET /readyz` — Readiness probe. Returns 200 if the gateway is fully booted and ClickHouse is currently reachable, 503 otherwise. +- `GET /livez` — Liveness probe. Returns 200 once the gateway has discovered ClickHouse table schemas at least once. Returns 503 with a diagnostic body while the boot-time schema discovery retry loop is still running (e.g. ClickHouse unreachable, target database missing). After successful boot, `/livez` stays 200 — transient ClickHouse blips at runtime are reflected in `/readyz`, not `/livez`. Over a [nested settings directory](#the-nested-settings-directory) it is 503 while no tenant has completed a first discovery, and 200 from the first tenant's success on. +- `GET /readyz` — Readiness probe. Returns 200 if the gateway is fully booted and ClickHouse is currently reachable, 503 otherwise. Over a nested directory every open ClickHouse pool is pinged at once and one that answers is enough; the 503 names every pool that did not. `/healthz` remains registered as a **permanent alias** of `/livez` (it's the most widely-recognized name); `/health` and `/ready` are **deprecated aliases** for the v0.1.x line and will be removed in v0.2.0. Point new deployments at the `/livez` / `/readyz` names. Configure your load balancer or orchestrator to use these endpoints. -**Exposure.** Probes share the API server's port (`:8080`) — kubelet probes the container internally, so there's no separate-port convention for them (metrics are the signal that optionally gets its own `prometheus.port`). If you forward `:8080` to the public internet the probe paths become reachable. The **recommended** posture is to keep `/livez`/`/readyz`/`/healthz` to internal callers and expose only **`/v1/health`** publicly (the SDK's content-free liveness ping, which never touches ClickHouse). `/readyz` issues a ClickHouse `Ping` on every call, so a public `/readyz` lets an unauthenticated flood become per-request backend pings, and the bare probes leak boot/readiness state — keeping them internal is a [reverse-proxy/ingress concern](/reverse-proxy#health-probes), and your orchestrator reaches them the internal way (kubelet on the container, LB on the backend) regardless. +**Exposure.** Probes share the API server's port (`:8080`) — kubelet probes the container internally, so there's no separate-port convention for them (metrics are the signal that optionally gets its own `prometheus.port`). If you forward `:8080` to the public internet the probe paths become reachable. The **recommended** posture is to keep `/livez`/`/readyz`/`/healthz` to internal callers and expose only **`/v1/health`** publicly (the SDK's content-free liveness ping, which never touches ClickHouse). `/readyz` pings every open ClickHouse pool on every call, so a public `/readyz` lets an unauthenticated flood become per-request backend pings, and the bare probes disclose boot and readiness state — each unanswering pool's ClickHouse address, database and user, and over a nested directory the failing tenant's id — keeping them internal is a [reverse-proxy/ingress concern](/reverse-proxy#health-probes), and your orchestrator reaches them the internal way (kubelet on the container, LB on the backend) regardless. ### Boot-time degraded mode @@ -250,7 +250,7 @@ This means: - The binary itself no longer exits and crash-loops every ~10s under a supervisor. Process state is preserved across CH outages. - An operator can `curl /livez` and read the exact failure mode instead of grepping a restart-loop log. -- `/v1/ingest?table={table}` and other schema-aware endpoints will reject requests with a 4xx until discovery succeeds, since the schema registry is empty. +- `/v1/ingest?table={table}` and the other schema-aware endpoints answer `503` with `Retry-After: 5` (`schema not loaded yet`) until discovery succeeds — not a `404`, since the table may well exist. **Important — orchestrator restart semantics.** `/livez` returning 503 during the retry window is what most LB / `depends_on` setups want (route around the unready instance, hold dependents), but a Kubernetes `livenessProbe` pointed at `/livez` will still mark the pod unhealthy and restart it after `failureThreshold × periodSeconds` elapses (default ~30s) — effectively re-creating the restart loop at a slower cadence. Use a `startupProbe` to gate liveness/readiness until the first successful schema discovery (see the K8s example below). Docker `HEALTHCHECK` marks the container `(unhealthy)` but does not restart it by default, so docker-compose deployments don't need a separate startupProbe-equivalent — the `HEALTHCHECK`'s `--start-period=15s` plus `service_healthy` dependency wait covers the same idea at a smaller scale. @@ -379,19 +379,19 @@ settings/ └── roles.json ``` -That is the layout a control plane writes, and it does not answer queries on its own: tenant `0`'s folder supplies the wiring the whole process shares — the ClickHouse address included — so a directory that serves no tenant `0` boots with none. "What a tenant's folder decides", below, lists what comes from where. +That is the layout a control plane writes. Each folder's `clickhouse` block is its tenant's own ClickHouse, so a tenant answers queries once its first schema discovery against that ClickHouse succeeds (until then its schema-aware routes answer `503`, `schema not loaded yet`); what tenant `0`'s folder still supplies for the whole process — the message queue's budget, the token verifier of the routes that name no tenant, their CORS list — is listed under "What a tenant's folder decides", below. The folder name is the tenant id, and each folder is a complete settings directory: everything on the [Settings Directory](/settings-directory) page applies to it as written, except where the rules below say otherwise. The two shapes don't mix — a folder beside the four files, or a loose file beside the folders, is a validation error — and a running server keeps the shape it booted with, so switching is stop, restructure, start. The dedupe stores need no restructuring: the four files' store already lives at `/0/dedupe` (see [Persistent Storage](#persistent-storage-required-for-containers)), which is where a `0` folder's store goes. Dot-prefixed entries are ignored in either shape. `wavehouse validate` checks either shape with the same exit codes; a finding in a nested directory names its folder (`acme/policies.json`), and a folder whose name is not a tenant id (`nats` and `pebble` are not, in any letter case: `data_dir` keeps those names for itself) is a finding of its own — that folder is skipped, and the rest of the directory still loads. -**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process draws its shared wiring from that folder, so rejecting it costs every tenant ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. +**A rejected folder fails closed, for that tenant alone — tenant `0`'s excepted.** A folder that fails validation stops its tenant being served — its requests answer `503` — while every other tenant carries on, at boot and on a reload alike. Tenant `0` is the exception: the process still draws some shared wiring from that folder, so rejecting it costs every tenant something ("What a lost tenant `0` costs", below, says what). There is no fall back to the tenant's previous settings, unlike [the single-tenant directory](/settings-directory#loading-and-hot-reload): the recovery is fixing the folder and reloading it. A request already in flight finishes on the settings it started with. The findings go to the log and to the reload response, never into the `503`. A finding about the directory itself — a loose file, an entry or a directory that can't be read, a changed shape — is another matter: it refuses boot, and on a reload it rejects the reload whole and leaves every tenant as it was. **Reloading is the writer's call.** A nested directory is not watched, because a watcher would validate a folder halfway through being written and drop its tenant. Whoever writes a tenant's folder reloads it once it is complete: `POST /v1/ops/settings/reload?tenant=acme` re-validates that folder and reads nothing else. It must name a tenant the server already holds (`404` otherwise), so a folder the server does not hold yet — one added since the last whole-directory reload — is picked up by a whole-directory reload, not by naming it; a tenant it holds but rejected is reloaded by name like any other. Without the parameter — and on `SIGHUP` — the whole directory is reloaded and mirrors its folders: a new folder becomes a tenant, and a removed one becomes unknown. A whole-directory reload re-validates every folder, so it carries the exposure the watcher would: a folder caught halfway through being written can fail validation, and its tenant then stops being served until a later reload adopts it. The response is the [single-tenant one](/api#post-v1opssettingsreload--reload-settings-directory). After a whole-directory reload, `adopted: false` with a `422` can mean adopted in part: the folders with an error among their `findings` were rejected and the rest were adopted — warnings included, since `findings` carries every folder's. -**The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes` and `GET /v1/ops/pipes/{name}` take the same `?tenant=`, and read tenant `0` without it. The other `/v1/ops/*` routes — schema, DLQ stats, raw SQL — act on the resources the whole process shares and ignore the parameter. On the three routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). +**The admin routes take the operator key only.** `/v1/ops/*` reaches every tenant, so over a nested directory no tenant's admin role opens it: the [operator key](/api#authentication) alone does, and a token carrying an admin role gets `403`. Boot a nested directory without `auth.operator_key` and no caller can reach these routes at all, which leaves `SIGHUP` as the only reload; the server warns about it at boot. `GET /v1/ops/pipes`, `GET /v1/ops/pipes/{name}`, `GET /v1/ops/schema`, `POST /v1/ops/schema/refresh` and `POST /v1/ops/query` take the same `?tenant=`, and address tenant `0` without it; `GET /v1/ops/dlq/stats` reads the queue the whole process shares and ignores the parameter. On the routes that take it the parameter is parsed strictly — a query string that does not parse, an empty or repeated `tenant`, or a malformed id is a `400`, never a silent read of the default tenant or, on the reload route, a reload of every tenant. The SDK sends it as the [`tenant` option](/sdk/admin#settings--whsettings). -**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store at `//dedupe`, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does — so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. The process still has one ClickHouse connection and one message queue, and those follow tenant `0`'s folder: `clickhouse.*`, `mq.max_bytes_gb`, and `schema.refresh_interval`. So every tenant reads and writes the same ClickHouse. The message queue is shared but addressed per tenant: an event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool too, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. Every tenant reads the same ClickHouse tables until each gets its own ([#583](https://github.com/Wave-RF/WaveHouse/issues/583) story 6), so an insert invalidates the table's cached results under every tenant the directory holds, a rejected one included (it comes back into service with the entries it has), not only under the tenant it was ingested for. The one gap is a folder removed and restored inside a result's TTL (the query's duration × 1000, floored at 10 s and capped at 1 h): a removed tenant is forgotten, so its entries are not bumped while it is gone, and what becomes of a removed tenant's cache is story 3's to decide. Two settings weigh every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`; and the sweeper, which keeps the longest `stream.gap_window_minutes` among them, since every tenant's events share one message-queue stream and a purge is one bound over it. +**What a tenant's folder decides, and what tenant `0`'s does.** A request is evaluated against its own tenant's `policies.json` and `pipes.json` (ingest, structured queries, pipes), its `query.*` keys, its `cors.allowed_origins`, and its `dedupe` block: whether its records are deduplicated, by which id, against the tenant's own store at `//dedupe`, which that folder's `dedupe.enabled` opens and closes on reload exactly as [the single-tenant one](/settings-directory#deduplication) does — so `wavehouse_ingest_dedupe_disabled_total` ticks only across a tenant's own reload, whatever the other tenants' switches say. A tenant's seen ids are its own: the same event id is first seen under each tenant that sends it. Its `auth` block is its own too: each tenant's folder wires that tenant's token verifier (`jwks_url`, `role_claim`), built when the folder is adopted and rebuilt when its wiring changes, so a JWKS-issued token verifies only under the tenants whose `jwks_url` names its provider's key set. Under another tenant's header a token is treated as invalid, and the request falls back to that tenant's `default_role` like any other unverifiable token, possibly after a rate-limited key refetch (see [Authentication](/settings-directory#authentication)). Keep `X-Tenant-ID` pinned at the proxy so a token is never presented under the wrong tenant. Tenants can still accept each other's tokens: those that leave `jwks_url` empty share the boot HMAC secret when `auth.jwt_secret` is set, so a token verifies under any of them (with no secret they validate no token at all), and those whose `jwks_url` names the same key set accept each other's tokens; isolate them by provider, or scope rows by a signed claim ([row-level security](/access-control#row-level-security)). A tenant whose `jwks_url` has not been fetched yet answers `503` with `Retry-After` to its token-bearing requests alone. A tenant that stops being served — its folder rejected or removed — loses its verifier and the JWKS refresh with it, and gets a fresh one when its folder is adopted again. The HMAC secret and the operator key stay boot config, shared by every tenant; the operator key is stamped with the request tenant's `admin_role`. A tenant's `clickhouse` and `schema` blocks are its own as well: each tenant reads and writes its own ClickHouse — one native pool per distinct address, database, user, password and `tls` tuple, shared by the tenants naming it, under the process-wide [connection ceiling](/settings-directory#clickhouse) — and discovers its own tables from its own database on its own `schema.refresh_interval`. The process still has one message queue, and its budget, `mq.max_bytes_gb`, follows tenant `0`'s folder. The queue is shared but addressed per tenant: an event is published on its tenant's subject (`ingest.{tenant}.{table}`), so a `GET /v1/stream` connection is authorized by its own tenant's `policies.json` and receives its own tenant's rows alone, the ingest worker inserts a row into its own tenant's ClickHouse, a failed row is parked under its own tenant's `dlq.enabled` and subject (`dlq.{tenant}.{table}`), and two tenants' tables of one name never share a batch. The query cache is one pool too, but its entries are keyed by tenant: identical `POST /v1/query` and pipe requests from two tenants are two entries and two queries to ClickHouse, and a tenant is never served another's cached rows. An insert invalidates the table's cached results under every tenant on the same ClickHouse address and database as the tenant it was ingested for, whatever their user or `tls` block, since they read the same tables; a tenant on no pool — its folder rejected or removed, or no pool could be opened for it, such as by the ceiling — is out of that fan-out while it is, and has its cached `POST /v1/query` results dropped the moment it is back on one, so a repaired or restored folder never serves query rows cached before the inserts it missed, and so does a tenant whose folder moves it to another address or database, whose cached rows came from other tables; a cached pipe result is left alone by all of this — no insert invalidates one, since it names no table — and stays until its TTL expires. Two settings weigh every tenant: the SSE keepalive, where the wheel runs at the shortest `stream.keepalive_interval` among the tenants being served, with that tenant's `stream.keepalive_buckets`; and the sweeper, which keeps the longest `stream.gap_window_minutes` among them, since every tenant's events share one message-queue stream and a purge is one bound over it. -**What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. `clickhouse.*` and `mq.max_bytes_gb` stay as tenant `0` last adopted them; tenant `0`'s verifier is dropped with the folder, rejected or removed, like any tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its data staying on disk for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: its failed rows are parked on the DLQ whatever its switch said, and its `GET /v1/stream` subscribers have every row withheld, since the hub reads no policy for it — the other tenants' events are untouched. The sweeper keeps the longest gap window among the tenants still served, so tenant `0`'s history is purged at theirs, and with no tenant left being served the window is zero, which purges the acknowledged history gap-fill replays; schema discovery keeps the refresh cadence it had. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — has no ClickHouse address: it boots, reports degraded on `/livez`, and answers no query. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. +**What a lost tenant `0` costs.** A `0` folder that a reload rejects or removes stops tenant `0` being served like any other, and what becomes of the shared settings depends on how they are read. `mq.max_bytes_gb` stays as tenant `0` last adopted it; tenant `0` leaves its ClickHouse pool (closed only once no served tenant names its tuple), and its schema registry and verifier are released with the folder, like any other tenant's; the `/v1/ops/*` routes, which resolve no tenant, verify against it, so a token there reads as invalid (`401`) rather than merely non-admin (`403`) until tenant `0` is served again — the operator key, which never consults a verifier, is unaffected. CORS does not stay either: the responses that read tenant `0`'s list — the tenant-exempt routes, the refusals, a preflight naming no tenant — carry no CORS headers until the folder is served again, while every other tenant's routes keep their own list. Tenant `0`'s own dedupe store closes, as any rejected or removed tenant's does, its data staying on disk for the folder that restores it. What is read per event follows the event's tenant, so tenant `0`'s events are the ones affected: with no ClickHouse to insert into, its rows fail and are parked on the DLQ whatever its switch said, and its `GET /v1/stream` subscribers have every row withheld, since the hub reads no policy for it — the other tenants' events are untouched. The sweeper keeps the longest gap window among the tenants still served, so tenant `0`'s history is purged at theirs, and with no tenant left being served the window is zero, which purges the acknowledged history gap-fill replays. A nested directory that has never served a tenant `0` — no `0` folder, or one rejected at boot — serves every other tenant from its own ClickHouse. Outside `/v1/ops/*`, a `/v1` request that sends no `X-Tenant-ID` resolves to tenant `0`, so with no `0` folder it answers `404 unknown tenant: 0` (`503` with a rejected one) — the SDK's `/v1/health` reachability ping included. ### Upgrading behind a proxy that already sends `X-Tenant-ID` diff --git a/docs/src/content/docs/development.md b/docs/src/content/docs/development.md index 3437fc39..01f82e73 100644 --- a/docs/src/content/docs/development.md +++ b/docs/src/content/docs/development.md @@ -455,7 +455,7 @@ WaveHouse/ │ ├── app/ # Process wiring (build every component, run under one errgroup, release in reverse) │ ├── auth/ # JWT/JWKS authentication middleware │ ├── cache/ # L1 (Ristretto) + L2 caching -│ ├── chconn/ # ClickHouse connection manager (swapped on settings reload) +│ ├── chconn/ # ClickHouse pools, one per connection tuple (reconciled on settings reload) │ ├── chsql/ # Shared ClickHouse SQL helpers (quoting + bind-safety) │ ├── config/ # YAML + env var configuration │ ├── dedupe/ # Optional deduplication (Pebble) diff --git a/docs/src/content/docs/ingest-pipeline.md b/docs/src/content/docs/ingest-pipeline.md index d64829a0..dd90bd37 100644 --- a/docs/src/content/docs/ingest-pipeline.md +++ b/docs/src/content/docs/ingest-pipeline.md @@ -13,7 +13,7 @@ It is deliberately detailed: this is a hot, concurrency-heavy path, and the goro | File | Contents | | --- | --- | -| `worker.go` | `StartIngestWorker`, the `dispatchLoop`, `parseMsg` (+ `rejectPoison` for an envelope it cannot read), the per-tenant-table `tableBatcher`/`tableLoop`, `flushTable` (splits a batch per column list via `groupByColumns`) and `flushGroup` (bulk insert with a row-by-row poison-isolation fallback), `insertToClickHouse`, `handleSuccess` (acks, after `invalidate` bumps the tenant's cache namespaces — under every tenant the registry knows, through the cache `internal/app` hands the worker, while the tenants share one ClickHouse), `sendToDLQ`/`parkOnDLQ` | +| `worker.go` | `StartIngestWorker`, the `dispatchLoop`, `parseMsg` (+ `rejectPoison` for an envelope it cannot read), the per-tenant-table `tableBatcher`/`tableLoop`, `flushTable` (splits a batch per column list via `groupByColumns`) and `flushGroup` (bulk insert with a row-by-row poison-isolation fallback), `insertToClickHouse` (into the batch's tenant's ClickHouse, `chconn.Pools.Target`), `handleSuccess` (acks, after `invalidate` bumps the tenant's cache namespaces — under every tenant on the same ClickHouse address and database, through the cache `internal/app` hands the worker, since they read the same tables), `sendToDLQ`/`parkOnDLQ` | | `compact.go` | `EncodeCompactRow` — renders one record as a `JSONCompactEachRow` line over the table's **insertable** columns, in declaration order. Serialization only: it validates nothing and judges no value | | `sweeper.go` | The **Active Sweeper** — every minute, asks the MQ to purge the events that are both written to ClickHouse and past the SSE gap window (the purge arithmetic below lives in `internal/mq/purge.go`) | | `types.go` | `EventMessage` wire format and the `BufferConsumerName` constant | diff --git a/docs/src/content/docs/reverse-proxy.mdx b/docs/src/content/docs/reverse-proxy.mdx index e644cd7d..408bc8d2 100644 --- a/docs/src/content/docs/reverse-proxy.mdx +++ b/docs/src/content/docs/reverse-proxy.mdx @@ -216,12 +216,12 @@ Every admin-gated endpoint — raw SQL, pipe inspection, settings reload, schema WaveHouse serves Kubernetes-convention probes on `:8080` (full behavior in [Deployment → Health Checks](/deployment#health-checks)): - **`/livez`** — liveness; sticky-200 after first successful boot. Does not touch ClickHouse. -- **`/readyz`** — readiness; issues a ClickHouse `Ping` on **every** call. Point your load balancer's (internal) health check here so it routes around an instance whose ClickHouse is unreachable. +- **`/readyz`** — readiness; pings every open ClickHouse pool on **every** call. Point your load balancer's (internal) health check here so it routes around an instance whose ClickHouse is unreachable. - **`/healthz`** — permanent alias of `/livez`. - **`/v1/health`** — the SDK's content-free liveness ping; mirrors `/livez` (200 once booted) and never touches ClickHouse. :::caution[Recommended: keep the bare probe paths off the public vhost] -Expose only **`/v1/health`** (and your API) to the internet — route `/livez`, `/readyz`, and `/healthz` on a private/internal listener, not the public proxy. Your orchestrator still reaches them the internal way regardless: kubelet probes the container directly on `:8080`, and a load balancer health-checks the backend — neither goes through the public proxy. Two reasons to keep them internal: `/readyz` pings ClickHouse on every call, so a public `/readyz` lets an unauthenticated flood turn into a per-request backend ping; and the probes leak boot/readiness state. `/v1/health` is the safe public liveness endpoint because it answers the same "is this server up" question without touching ClickHouse — it's what the SDK's `wh.sys.health()` calls. +Expose only **`/v1/health`** (and your API) to the internet — route `/livez`, `/readyz`, and `/healthz` on a private/internal listener, not the public proxy. Your orchestrator still reaches them the internal way regardless: kubelet probes the container directly on `:8080`, and a load balancer health-checks the backend — neither goes through the public proxy. Two reasons to keep them internal: `/readyz` pings every open ClickHouse pool on every call, so a public `/readyz` lets an unauthenticated flood turn into per-request backend pings; and the probe bodies disclose boot and readiness state — each unanswering pool's ClickHouse address, database and user, and over a nested directory the failing tenant's id. `/v1/health` is the safe public liveness endpoint because it answers the same "is this server up" question without touching ClickHouse — it's what the SDK's `wh.sys.health()` calls. ::: ## Timeouts and slow links diff --git a/docs/src/content/docs/sdk/admin.md b/docs/src/content/docs/sdk/admin.md index b1e7a750..ae1635f7 100644 --- a/docs/src/content/docs/sdk/admin.md +++ b/docs/src/content/docs/sdk/admin.md @@ -20,6 +20,14 @@ await wh.schema.refresh(); Individual table schema is also available via `wh.from('clicks').schema()`. +Over [a nested settings directory](/deployment#the-nested-settings-directory), pass `tenant` to read or refresh that tenant's schema, authenticating with the [operator key](/api#authentication) (sent as `X-Operator-Key` via [`options.headers`](/sdk#custom-headers)): the nested `/v1/ops/*` routes admit it alone, and a token carrying an admin role gets `403`; without it the calls address tenant `0`. A `503` with `Retry-After` is a tenant whose first discovery has not succeeded yet (`Retry-After: 5`), or one on no ClickHouse pool — [no pool could be opened for it](/settings-directory#clickhouse), such as one the connection ceiling refused — on the refresh (`Retry-After: 30`); the SDK retries both, and `wh.sql()` takes the same option to run against that tenant's ClickHouse: + +```ts +const { data } = await wh.schema.list({ tenant: 'acme' }); +await wh.schema.refresh({ tenant: 'acme' }); +const { data: rows } = await wh.sql('SELECT count() FROM clicks', { tenant: 'acme' }); +``` + > `wh.schema.list()`, `wh.schema.refresh()`, and `wh.from(t).schema()` hit `/v1/ops/schema*`, which are **admin-only** endpoints: the caller must pass the admin gate — resolve to the policy admin role (`admin_role`, `"admin"` by default) or present the non-JWT [operator key](/api#authentication). Unless the deployment deliberately sets `default_role` to the admin role (the loudly-warned dev-only setting), construct the client with an admin-role token — or send the operator key via [`options.headers`](/sdk#custom-headers) — or these calls return `403`. --- diff --git a/docs/src/content/docs/sdk/queries.md b/docs/src/content/docs/sdk/queries.md index aded70c1..38de8890 100644 --- a/docs/src/content/docs/sdk/queries.md +++ b/docs/src/content/docs/sdk/queries.md @@ -62,7 +62,7 @@ await clicks.insertNDJSON(await openAsBlob('events.ndjson')); ### `.schema(opts?)` -Fetch the table's column definitions from ClickHouse. `.schema()` hits `/v1/ops/schema`, an **admin-only** endpoint: the caller must pass the admin gate — resolve to the policy admin role or present the non-JWT [operator key](/api#authentication) via [`options.headers`](/sdk#custom-headers) — or this returns `403`. +Fetch the table's column definitions from ClickHouse. `.schema()` hits `/v1/ops/schema`, an **admin-only** endpoint: the caller must pass the admin gate — resolve to the policy admin role or present the non-JWT [operator key](/api#authentication) via [`options.headers`](/sdk#custom-headers) — or this returns `403`. Over [a nested settings directory](/deployment#the-nested-settings-directory), pass `{ tenant: 'acme' }` to read that tenant's schema (tenant `0` without it) — see [`wh.schema`](/sdk/admin#schema--whschema). ```ts const { data } = await clicks.schema(); @@ -252,7 +252,7 @@ while (result.hasMore && result.next) { ## Raw SQL — `wh.sql(query, opts?)` -Execute a raw SQL query. `/v1/ops/query` is admin-only: the caller must resolve to the policy admin role (`admin_role`, `"admin"` by default). A tokenless request falls back to the `default_role`, so it is rejected with `403` on any policy that doesn't deliberately set `default_role` to the admin role (a loudly-warned dev-only setting); an invalid or expired token is rejected with `401`. The SDK has no first-class option for the server's non-JWT [operator key](/api#authentication) — an operator can send its `X-Operator-Key` header via [`options.headers`](/sdk#custom-headers). +Execute a raw SQL query. `/v1/ops/query` is admin-only: over a settings directory of the four files the caller must resolve to the policy admin role (`admin_role`, `"admin"` by default); over [a nested one](/deployment#the-nested-settings-directory) only the operator key is admitted, and a token carrying an admin role gets `403`. A tokenless request falls back to the `default_role`, so it is rejected with `403` on any policy that doesn't deliberately set `default_role` to the admin role (a loudly-warned dev-only setting); an invalid or expired token is rejected with `401`. The SDK has no first-class option for the server's non-JWT [operator key](/api#authentication) — an operator can send its `X-Operator-Key` header via [`options.headers`](/sdk#custom-headers). Over [a nested settings directory](/deployment#the-nested-settings-directory), pass `{ tenant: 'acme' }` to run the SQL against that tenant's ClickHouse (tenant `0`'s without it) — see [`wh.schema`](/sdk/admin#schema--whschema). ```ts const { data, error } = await wh.sql('SELECT page, count() FROM clicks GROUP BY page LIMIT 10'); diff --git a/docs/src/content/docs/sdk/reference.md b/docs/src/content/docs/sdk/reference.md index dc5fd43c..56fc09d9 100644 --- a/docs/src/content/docs/sdk/reference.md +++ b/docs/src/content/docs/sdk/reference.md @@ -32,7 +32,7 @@ The SDK **never throws** for anything the server returns — all API errors come | 403 | `HTTP_403` | No | Insufficient permissions | | 404 | `HTTP_404` | No | Table, pipe, or tenant not found | | 500 | `HTTP_500` | Yes | Server error (retried per `maxRetries`) | -| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, or a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on that last cause waits the 30 s; a stream re-dials on its own jittered backoff instead | +| 503 | `HTTP_503` | Yes | Service unavailable, a tenant whose settings folder was rejected, a schema not discovered yet, a tenant on no ClickHouse pool, or a token sent while that tenant's JWKS has not been fetched yet (`token verifier not ready`, `Retry-After: 30`). REST calls auto-retry, honoring `Retry-After` when the response carries one — so each attempt on that last cause waits the 30 s; a stream re-dials on its own jittered backoff instead | | 0 | `NETWORK_ERROR` | Yes | Network failure (retried with exponential backoff) | | 0 | `ABORTED` | No | Request canceled via `AbortSignal` | | 0 | `SSE_CONNECT_ERROR` | No | Stream could not be started (e.g. a non-absolute `baseURL`) | @@ -89,7 +89,7 @@ createClient(config) → WaveHouseClient │ ├── .selectAll() → QueryBuilder (PromiseLike) │ ├── .insert(data) → Promise> │ ├── .insertNDJSON(source) → Promise> -│ ├── .schema() → Promise> (admin) +│ ├── .schema(opts?) → Promise> (admin) │ └── .stream(opts?) → StreamController ├── .pipe(name, params?) → PipeRef (PromiseLike) │ ├── .fetch(opts?) → Promise> // { signal } only — no limit @@ -99,8 +99,8 @@ createClient(config) → WaveHouseClient │ └── .get(name, opts?) → Promise> ├── .sql(query, opts?) → Promise> (admin) ├── .schema (admin) -│ ├── .list() → Promise> -│ └── .refresh() → Promise> +│ ├── .list(opts?) → Promise> +│ └── .refresh(opts?) → Promise> ├── .settings (admin) │ └── .reload(opts?) → Promise> ├── .dlq (admin) @@ -129,7 +129,7 @@ npx wavehouse-codegen --url http://localhost:8080 --out ./src/db.d.ts pnpm codegen --url http://localhost:8080 --out ./src/db.d.ts ``` -Codegen reads `/v1/ops/schema`, which is **admin-only**. Against a non-dev server, pass an admin-role token with `--auth ` or the request is denied with `403`. +Codegen reads `/v1/ops/schema`, which is **admin-only**. Against a non-dev server of the four settings files, pass an admin-role token with `--auth ` or the request is denied with `403`. Codegen does not support [a nested settings directory](/deployment#the-nested-settings-directory): its `/v1/ops/*` routes admit the operator key alone, which codegen has no option to send, and it would read tenant `0`'s schema only. **Options:** diff --git a/docs/src/content/docs/settings-directory.mdx b/docs/src/content/docs/settings-directory.mdx index 44c724e9..2a034bfa 100644 --- a/docs/src/content/docs/settings-directory.mdx +++ b/docs/src/content/docs/settings-directory.mdx @@ -116,7 +116,7 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `clickhouse.tls.insecure_skip_verify` | `false` | Skips server certificate verification. `true` validates with a warning: both hops then accept any certificate. | | `clickhouse.tls.server_name` | `""` | Name the server certificate is verified against; empty derives it from the host in `addr` (native) or the URL (HTTP). | | `clickhouse.headers` | `{}` | Extra request headers for the HTTP interface (ingest `INSERT`s, `POST /v1/ops/query`). WaveHouse's own `Content-Type` and credential headers are set after them and win; naming `X-ClickHouse-User`, `X-ClickHouse-Key` or `Authorization` is a validation error, and so are two spellings of one name (names are case-insensitive). See [ClickHouse](#clickhouse). | -| `clickhouse.max_open_conns` | `10` | Native connection pool size (`>= max_idle_conns`). Must also not exceed the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) when one is set. | +| `clickhouse.max_open_conns` | `10` | Native connection pool size (`>= max_idle_conns`); tenants sharing a pool size it to the largest ask among them. The open pools together must not exceed the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) when one is set. See [ClickHouse](#clickhouse). | | `clickhouse.max_idle_conns` | `5` | Idle native connections kept open (`>= 1`). | | `auth.jwks_url` | `""` | JWKS endpoint (absolute `http(s)` URL). When set, JWKS is the **sole** verifier and `jwt_secret` is ignored. See [Authentication](#authentication). | | `auth.role_claim` | `role` | Dot-separated JWT claim path the role is read from (e.g. `app_metadata.role`). | @@ -128,7 +128,7 @@ The tenant tunables. Every key is required (a missing one is a validation error) | `dlq.tables.
.enabled` | `{}` | Optional per-table override of the switch. | | `query.timestamp_bucket_seconds` | `60` | Bucket (seconds, `>= 0`) that a structured query's relative time range is truncated to, so near-identical queries share a cache entry; `0` disables bucketing. Read per query. | | `query.default_max_rows` | `10000` | Fallback result `LIMIT` (`>= 1`) applied to a structured query when the caller and policy specify none. A result-**shaping** default, not a resource limit — server-wide limits (memory, rows scanned, execution time) belong in ClickHouse, see [Server-side resource limits](/configuration#server-side-resource-limits). | -| `schema.refresh_interval` | `60` | Seconds between ClickHouse table-schema re-discoveries (`>= 1`); a reloaded value takes effect from the next refresh cycle. Schemas are also refreshable on demand via `POST /v1/ops/schema/refresh` (admin-only). | +| `schema.refresh_interval` | `60` | Seconds between ClickHouse table-schema re-discoveries (`>= 1`), per tenant; a reloaded value takes effect from the next refresh cycle, and the first periodic refresh lands at a random point within the interval. Schemas are also refreshable on demand via `POST /v1/ops/schema/refresh` (admin-only). | | `stream.keepalive_interval` | `30` | Seconds (`>= 1`) a quiet `GET /v1/stream` connection may go without a write before the server sends a `:` keepalive comment — keep it under your proxy's idle timeout; see [Streaming](#streaming). | | `stream.keepalive_buckets` | `3` | Load-spreading (`>= 1`): connections are spread across N buckets so each tick nudges ~1/N of live streams. Most deployments leave it. | | `stream.gap_window_minutes` | `15` | Minutes (`>= 0`) of written-to-ClickHouse history the Active Sweeper keeps in NATS for `Last-Event-ID` gap-fill; applies from the next sweep. | @@ -192,14 +192,16 @@ Every dedupe knob lives here — there are no boot-config keys for it. The switc ## ClickHouse -The `clickhouse` block is the connection wiring, minus the password. A reload that changes any of it swaps the connection behind every consumer (schema discovery, structured queries, pipes, `/readyz`, the ingest worker's HTTP `INSERT`s, and the raw-SQL proxy — the HTTP-side ones re-read the target per request). The replaced connection stays open for one `query_timeout` so in-flight queries finish. The swap is **unconditional**: the adopted settings are the authority, so an address that isn't reachable is applied all the same and shows up where reachability already does — schema discovery retries and logs, `/readyz` fails, queries return errors — until the next reload fixes it. Two things are refused instead, and the connection then keeps its previous wiring until the next reload: a `tls` block whose certificate files cannot be loaded (unreadable, not PEM, or a `cert_file` and `key_file` that do not pair), and a `max_open_conns` above the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse). The reload itself still reports `adopted` — the refusal is an `ERROR` log line, not a finding — and the next reload retries it; at boot either refuses to start. Validation checks shape only (`host:port`, port range, scheme, non-empty database and user, timeout `>= 1`, the `tls` block's shape without opening its paths, header names and values, pool sizes); reachability is a runtime concern, so `wavehouse validate` needs no ClickHouse and no certificate files. At boot the address is dialed lazily, as before: an unreachable ClickHouse degrades `/livez` and retries rather than refusing to start. +The `clickhouse` block is the connection wiring, minus the password. A reload applies it to every consumer (schema discovery, structured queries, pipes, `/readyz`, the ingest worker's HTTP `INSERT`s, and the raw-SQL proxy — the HTTP-side ones re-read the target per request). A change to `addr`, `database`, `username` or the `tls` block moves the tenant to the pool of its new tuple, opened for it when no served tenant has that tuple; a change to `max_open_conns` or `max_idle_conns` resizes its pool; a change to `http_port`, `http_scheme`, `headers` or `query_timeout` needs no new connection. A replaced connection stays open for the longest `query_timeout` among the tenants that were on it, so in-flight queries finish. The change is **unconditional**: the adopted settings are the authority, so an address that isn't reachable is applied all the same and shows up where reachability already does — schema discovery retries and logs, `/readyz` fails, queries return errors — until the next reload fixes it. Two things are refused instead: a `tls` block whose certificate files cannot be loaded (unreadable, not PEM, or a `cert_file` and `key_file` that do not pair), and a pool that would put the process over the boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) — a `max_open_conns` raised above it keeps the pool at its size, and a new connection tuple that would cross it is not opened. An option the ClickHouse driver refuses to open is refused the same way, though validation already excludes every one it knows of. A refused tenant keeps the pool and wiring it had until the next reload, or, newly served with no pool to keep, stays on none and its queries answer `503`. The reload itself still reports `adopted` — the refusal is an `ERROR` log line, not a finding — and the next reload retries it; at boot either refuses to start. Validation checks shape only (`host:port`, port range, scheme, non-empty database and user, timeout `>= 1`, the `tls` block's shape without opening its paths, header names and values, pool sizes); reachability is a runtime concern, so `wavehouse validate` needs no ClickHouse and no certificate files. At boot the address is dialed lazily, as before: an unreachable ClickHouse degrades `/livez` and retries rather than refusing to start. -**TLS.** There are two hops and two switches: `tls.enabled` puts the native-protocol connection (`addr`) on TLS, and `http_scheme: "https"` does the same for the HTTP interface (`http_port`). The rest of the `tls` block — the authority bundle, a client certificate, `insecure_skip_verify`, `server_name` — applies to whichever hop uses TLS, so a ClickHouse behind a private authority needs `ca_file` once for both. The certificate files are read when the connection is built and re-read only when the `tls` block itself changes: at boot, and on a reload that changes `tls`; a reload that changes another `clickhouse` key reuses the material already loaded. A file replaced in place with the `tls` block unchanged is picked up by a restart, like a rotated secret. Both switches move the hop to ClickHouse's TLS listeners, so the ports move too: `addr` to the secure native port (`9440` by default) and `http_port` to the HTTPS one (`8443`), as the example above does. Validation warns when only one hop is on TLS: a plaintext HTTP hop carries the credentials on every insert and raw-SQL query, and a plaintext native hop sends the password in its handshake. In the container images these are container paths: bind-mount the bundle (`-v /srv/clickhouse-ca.pem:/etc/wavehouse/clickhouse-ca.pem:ro`) and make it readable by UID 65532, like the settings directory. +**TLS.** There are two hops and two switches: `tls.enabled` puts the native-protocol connection (`addr`) on TLS, and `http_scheme: "https"` does the same for the HTTP interface (`http_port`). The rest of the `tls` block — the authority bundle, a client certificate, `insecure_skip_verify`, `server_name` — applies to whichever hop uses TLS, so a ClickHouse behind a private authority needs `ca_file` once for both. The certificate files are read when a pool opens: at boot, and on a reload that names a tuple no open pool has — a changed `tls` block, `addr`, `database` or `username`. A reload that changes only `http_port`, `http_scheme`, `headers`, `query_timeout` or the pool sizes reuses the material already loaded, so a file replaced in place is picked up the next time a pool for its tuple opens, or by a restart, like a rotated secret. Both switches move the hop to ClickHouse's TLS listeners, so the ports move too: `addr` to the secure native port (`9440` by default) and `http_port` to the HTTPS one (`8443`), as the example above does. Validation warns when only one hop is on TLS: a plaintext HTTP hop carries the credentials on every insert and raw-SQL query, and a plaintext native hop sends the password in its handshake. In the container images these are container paths: bind-mount the bundle (`-v /srv/clickhouse-ca.pem:/etc/wavehouse/clickhouse-ca.pem:ro`) and make it readable by UID 65532, like the settings directory. **Headers.** `clickhouse.headers` rides on every HTTP-interface request: routing or identification metadata for a proxy or gateway in front of ClickHouse. Values are stored in the file as written, so a credential placed there is only as protected as the settings directory itself; secrets stay in boot config. WaveHouse's own headers — `Content-Type` and the `X-ClickHouse-User` / `X-ClickHouse-Key` credentials — are set after them and win. Naming `X-ClickHouse-User` or `X-ClickHouse-Key` is a validation error, and so is `Authorization`: ClickHouse's HTTP interface reads it as Basic credentials, a second and conflicting credential path, so a gateway that authenticates that way needs a header name of its own. The native protocol carries no headers. **Existing directories.** Every key is required, so a `config.json` written before these keys existed fails validation until `tls`, `headers`, `max_open_conns` and `max_idle_conns` are added. The seed values — `tls.enabled: false` with the other `tls` keys empty, `headers: {}`, `10` and `5` — change nothing; `wavehouse bootstrap` in an empty directory writes them for reference. +**Per tenant.** Over a [nested settings directory](/deployment#the-nested-settings-directory) each folder's `clickhouse` block is that tenant's own. The process opens one native connection pool per distinct `addr`, `database`, `username`, password and `tls` tuple among the tenants being served — ClickHouse authenticates per connection, so tenants with different credentials never share one — and tenants naming the same tuple share one pool, sized to the largest `max_open_conns` and the largest `max_idle_conns` among them. `http_port`, `http_scheme`, `headers` and `query_timeout` stay each tenant's own whatever it shares: the HTTP hop is addressed per request and the deadline is per query. A reload reconciles the pools: a tenant whose tuple changed moves to the pool that tuple names, opened if it is new; a pool no tenant names any more closes once the longest `query_timeout` among the tenants it had has passed; a pool whose largest ask changed is resized the same way. The boot config's [`clickhouse.max_total_conns`](/configuration#clickhouse) bounds the open pools **together**: at boot, pools that would add up to more refuse to start, naming the sum and the ceiling; on a reload, a resize above it is refused and the pool keeps its size, and a tuple that cannot be opened — the ceiling, a `tls` block whose files cannot be loaded, or options the driver refuses — leaves its tenants on the pool they had, or on none for a tenant that had none. A tenant on no pool answers `503` with `Retry-After` on every route that reaches its ClickHouse until a reload opens one. The refusals are `ERROR` log lines, not findings, and the next reload retries. Schema discovery is per tenant too: each tenant's tables are discovered from its own `database` over its own pool, on its own `schema.refresh_interval`, with the first periodic refresh at a random point within the interval so tenants adopted together do not refresh together; until a tenant's first discovery, its table lookups answer `503` with `Retry-After` rather than `404` — the table may well exist. Until [#529](https://github.com/Wave-RF/WaveHouse/issues/529) every tenant's user authenticates with the one `WH_CH_PASSWORD`, so the tuple is in effect the address, database, user and `tls` block. + Moving to a different ClickHouse leaves ingest traffic unaffected: events land in the embedded queue regardless, and the worker's next flush uses the new target. ## Authentication diff --git a/internal/api/boot_chain_test.go b/internal/api/boot_chain_test.go index 213d7e5a..e623a9b4 100644 --- a/internal/api/boot_chain_test.go +++ b/internal/api/boot_chain_test.go @@ -87,7 +87,7 @@ func TestBoot_Chain_DegradedThenRecovers(t *testing.T) { // comes up partway through the retry backoff. conn := &errsThenSuccessConn{errs: []error{connRefused, connRefused, dbMissing}} - registry := discovery.NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + registry := discovery.NewSchemaRegistry(func() (driver.Conn, string) { return conn, "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) // Phase 0 — synchronous boot Refresh fails. internal/app records the // diagnostic in BootState and proceeds with the retry loop in a diff --git a/internal/api/cache_key.go b/internal/api/cache_key.go index ebe080b4..028bb778 100644 --- a/internal/api/cache_key.go +++ b/internal/api/cache_key.go @@ -22,7 +22,7 @@ import ( // a key is tenant-keyed by construction and readable as such (#583 story 8): // identical SQL and params under two tenants are two entries and two flights, // and a pipe — whose result carries no dependency namespaces — can never -// answer one tenant with another's rows once each tenant has its own +// answer one tenant with another's rows now that each tenant reads its own // ClickHouse (story 6). // // Every section is framed with a 1-byte type marker (0x01 for sql, 0x00 for diff --git a/internal/api/cache_tenant_test.go b/internal/api/cache_tenant_test.go index dcb99826..3c86f8b0 100644 --- a/internal/api/cache_tenant_test.go +++ b/internal/api/cache_tenant_test.go @@ -68,17 +68,17 @@ func cachedRouter(t *testing.T, tenants *settings.Registry, conn driver.Conn, c DefaultRole: "viewer", Tables: map[string]policy.TablePolicy{"clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}}, }) - timeout := func() time.Duration { return 5 * time.Second } + timeout := func(*settings.Store) time.Duration { return 5 * time.Second } return NewRouter(Dependencies{ Tenants: tenants, - Ingest: NewIngestHandler(reg, &testutil.MockPublisher{}), - StructuredQuery: NewStructuredQueryHandler(conn, c, reg, viewer, func(*settings.Store) int { return 60 }, timeout, nil), - Pipes: NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, conn, c, timeout), + Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), + StructuredQuery: NewStructuredQueryHandler(fixedConn(conn), c, fixedRegistry(reg), viewer, func(*settings.Store) int { return 60 }, timeout, nil), + Pipes: NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, fixedConn(conn), c, timeout), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, Version: NewVersionHandler("test", "test", "test"), - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, tenants), AuthMW: func(next http.Handler) http.Handler { return next }, PolicySource: policy.Static(&policy.Policy{}), }) @@ -103,9 +103,9 @@ func serveAs(t *testing.T, router http.Handler, path, body string, id tenant.ID) // behind them — each miss once and hit once, and ClickHouse is queried once // per tenant: the second tenant is never served the first one's rows. The // query key, the singleflight key, and the dependency namespaces all lead -// with the tenant (#583 story 8), which is what keeps this true once each -// tenant has its own ClickHouse (story 6) — a tenant-blind key would then be -// a silent cross-tenant read. +// with the tenant (#583 story 8), which is what keeps this true now that each +// tenant reads its own ClickHouse (story 6) — a tenant-blind key would be a +// silent cross-tenant read. // // The subtests share one router, one pool and one query counter, so they run // in order: neither the parent nor the subtests are parallel. diff --git a/internal/api/clickhouse_exec.go b/internal/api/clickhouse_exec.go index 83258993..6a296fc7 100644 --- a/internal/api/clickhouse_exec.go +++ b/internal/api/clickhouse_exec.go @@ -8,9 +8,28 @@ import ( "time" "github.com/ClickHouse/clickhouse-go/v2/lib/driver" + "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/google/uuid" ) +// connOf is conn's answer for store — the tenant's pool — and nil for an +// unwired source or a tenant on no pool. The nil is untyped: a nil *Manager +// inside a non-nil driver.Conn would pass a nil check and panic on use. +func connOf(conn func(*settings.Store) driver.Conn, store *settings.Store) driver.Conn { + if conn == nil { + return nil + } + return conn(store) +} + +// timeoutOf is timeout's answer for store, zero for an unwired source. +func timeoutOf(timeout func(*settings.Store) time.Duration, store *settings.Store) time.Duration { + if timeout == nil { + return 0 + } + return timeout(store) +} + // executeCHQuery runs sql against the native-protocol driver conn, // classifying by leading SQL verb to pick the Exec-vs-Query path — // clickhouse-go's driver.Query() errors on statements that return no diff --git a/internal/api/errors.go b/internal/api/errors.go index ab71601a..d03ccf2a 100644 --- a/internal/api/errors.go +++ b/internal/api/errors.go @@ -24,6 +24,26 @@ func writeJSONError(w http.ResponseWriter, status int, message string) { _ = json.NewEncoder(w).Encode(map[string]string{"error": message}) } +// The Retry-After hints of the two 503s a tenant's ClickHouse side answers +// with: a schema not discovered yet, which discovery retries on a 2s → 60s +// backoff, and no pool — one that could not be opened, such as one the +// connection ceiling refused — which the next settings reload retries (the +// ingest backpressure hint). +const ( + retryAfterSchema = "5" + retryAfterPool = "30" + + schemaNotLoadedMessage = "schema not loaded yet" + noConnectionMessage = "no ClickHouse connection is open for this tenant" +) + +// writeUnavailable writes a 503 with a Retry-After hint: what the request +// needs is not there yet, and a retry is the right response. +func writeUnavailable(w http.ResponseWriter, message, retryAfter string) { + w.Header().Set("Retry-After", retryAfter) + writeJSONError(w, http.StatusServiceUnavailable, message) +} + // writeAuthzDenied writes the response for an authorization denial and emits a // structured WARN (see logAuthzDenied) so a misconfigured role or policy shows // up in the logs without having to reproduce the request. When the request diff --git a/internal/api/errors_test.go b/internal/api/errors_test.go index 79035d19..68456611 100644 --- a/internal/api/errors_test.go +++ b/internal/api/errors_test.go @@ -154,7 +154,7 @@ func TestPipesHandler_Execute_DenialLogsAllowedRoles(t *testing.T) { // /v1/ingest route pattern alone doesn't say which check failed, or on what. func TestIngest_DenialLogsPolicyGate(t *testing.T) { buf := captureWarns(t) - h := NewIngestHandler(testRegistry(t), &testutil.MockPublisher{}) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": {"viewer": {Select: &policy.SelectPermissions{}}}, // no insert for viewer @@ -185,11 +185,11 @@ func TestAuthzDenied_LogsChiRoutePattern(t *testing.T) { reg := testutil.NewTestSchemaRegistry(t, nil) router := NewRouter(Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, &testutil.MockPublisher{}), + Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return next }, PolicySource: policy.Static(&policy.Policy{}), }) diff --git a/internal/api/health.go b/internal/api/health.go index 0f850f30..eb14ddf4 100644 --- a/internal/api/health.go +++ b/internal/api/health.go @@ -1,19 +1,18 @@ package api import ( + "context" "encoding/json" "net/http" "sync" - - "github.com/ClickHouse/clickhouse-go/v2/lib/driver" ) // BootState tracks a one-shot startup diagnostic surfaced by /livez. While // Err() returns non-nil the binary is considered to be in degraded-boot mode: // /livez responds 503 with the diagnostic message instead of 200, so an // operator can curl the endpoint to learn why the gateway isn't accepting -// traffic yet. Once boot work (today: ClickHouse schema discovery) succeeds, -// Set(nil) flips /livez back to 200. +// traffic yet. Once boot work (today: the first tenant's ClickHouse schema +// discovery) succeeds, Set(nil) flips /livez back to 200. // // BootState is safe for concurrent use. type BootState struct { @@ -45,7 +44,10 @@ func (b *BootState) Err() error { // HealthHandler provides liveness and readiness probes. type HealthHandler struct { - CHConn driver.Conn + // Ping is Readiness's ClickHouse check (chconn.Pools.Ping in production: + // every open pool at once, ready at the first answer, 503 with every + // pool's error when none answers). Nil skips the check. + Ping func(context.Context) error // Boot is consulted by both Liveness and Readiness. When non-nil and // its Err() is non-nil, both endpoints report 503 with the diagnostic // — used while boot-time schema discovery is still failing in the @@ -54,8 +56,8 @@ type HealthHandler struct { Boot *BootState } -func NewHealthHandler(chConn driver.Conn) *HealthHandler { - return &HealthHandler{CHConn: chConn} +func NewHealthHandler(ping func(context.Context) error) *HealthHandler { + return &HealthHandler{Ping: ping} } func (h *HealthHandler) Liveness(w http.ResponseWriter, _ *http.Request) { @@ -83,8 +85,8 @@ func (h *HealthHandler) Readiness(w http.ResponseWriter, r *http.Request) { return } } - if h.CHConn != nil { - if err := h.CHConn.Ping(r.Context()); err != nil { + if h.Ping != nil { + if err := h.Ping(r.Context()); err != nil { w.WriteHeader(http.StatusServiceUnavailable) _ = json.NewEncoder(w).Encode(map[string]string{"status": "not ready", "error": err.Error()}) return diff --git a/internal/api/health_test.go b/internal/api/health_test.go index 41ef55ee..b97d769f 100644 --- a/internal/api/health_test.go +++ b/internal/api/health_test.go @@ -65,7 +65,7 @@ func TestHealth_Readiness_PingFails(t *testing.T) { // without a test for the failure path a future refactor that moves // header setup into the success branch would silently drop them on // 503 responses. - h := NewHealthHandler(pingFailConn{err: errors.New("ch ping failed")}) + h := NewHealthHandler(pingFailConn{err: errors.New("ch ping failed")}.Ping) w := httptest.NewRecorder() r := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/readyz", nil) diff --git a/internal/api/ingest.go b/internal/api/ingest.go index 90905c01..c429aaa4 100644 --- a/internal/api/ingest.go +++ b/internal/api/ingest.go @@ -37,7 +37,8 @@ const maxReportedResults = 10000 // IngestHandler handles POST /v1/ingest?table={table} type IngestHandler struct { - Registry *discovery.SchemaRegistry + // Registry yields the request tenant's schema registry. + Registry RegistrySource // Dedup resolves the request tenant's deduplicator — the tenant's own // store, picked off the store the handler already holds (#583 story 7; // dedupe.Stores in production). nil when no dedupe store is wired (tests). @@ -64,7 +65,7 @@ type IngestHandler struct { maxRequestBytes int64 } -func NewIngestHandler(registry *discovery.SchemaRegistry, pub mq.Publisher) *IngestHandler { +func NewIngestHandler(registry RegistrySource, pub mq.Publisher) *IngestHandler { return &IngestHandler{Registry: registry, Publisher: pub} } @@ -167,10 +168,11 @@ func (h *IngestHandler) Handle(w http.ResponseWriter, r *http.Request) { } // TODO: prevent table-enumeration... - schema := h.Registry.Get(table) - if schema == nil { - slog.WarnContext(ctx, "unknown table requested", "table", table) - writeJSONError(w, http.StatusNotFound, "unknown table: "+table) + schema, err := lookupSchema(w, h.Registry, store, table, "unknown table: "+table) + if err != nil { + if errors.Is(err, discovery.ErrUnknownTable) { + slog.WarnContext(ctx, "unknown table requested", "table", table) + } return } diff --git a/internal/api/ingest_seams_test.go b/internal/api/ingest_seams_test.go index dc7de285..06335b92 100644 --- a/internal/api/ingest_seams_test.go +++ b/internal/api/ingest_seams_test.go @@ -48,7 +48,7 @@ func TestIngest_RecordValidatorSeam_IsUsed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} v := &recordingValidator{validateErr: errors.New("seam says no")} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Validator = v w := httptest.NewRecorder() @@ -66,7 +66,7 @@ func TestIngest_RecordValidatorSeam_IsUsed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} v := &recordingValidator{canonicalizeAs: "/rewritten"} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Validator = v w := httptest.NewRecorder() @@ -85,7 +85,7 @@ func TestIngest_RecordValidatorSeam_IsUsed(t *testing.T) { func TestIngest_DefaultValidator_WhenUnwired(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) require.Nil(t, h.Validator) assert.IsType(t, discoveryValidator{}, h.validator()) @@ -129,7 +129,7 @@ func TestIngest_InsertCheckerSeam_IsUsed(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(p) h.Checker = alwaysChecker{matches: tt.matches} @@ -168,7 +168,7 @@ func TestIngest_InsertCheckerSeam_InSet_IsUsed(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = checkInStore() h.Checker = alwaysChecker{inSet: tt.inSet} @@ -198,7 +198,7 @@ func TestIngest_DefaultChecker_WhenUnwired(t *testing.T) { }}}, }} pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(p) require.Nil(t, h.Checker) assert.IsType(t, canonicalChecker{}, h.checker()) @@ -224,7 +224,7 @@ func TestIngest_SeamOrdering_ChecksSitBetweenValidateAndCanonicalize(t *testing. }} pub := &testutil.MockPublisher{} v := &recordingValidator{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(p) h.Validator = v diff --git a/internal/api/ingest_test.go b/internal/api/ingest_test.go index 59e2d89d..fde3e46d 100644 --- a/internal/api/ingest_test.go +++ b/internal/api/ingest_test.go @@ -60,7 +60,7 @@ func ingestRequest(t *testing.T, table string, body any) *http.Request { func TestIngest_ValidPayload(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "count": 1}) w := httptest.NewRecorder() @@ -81,7 +81,7 @@ func TestIngest_ValidPayload(t *testing.T) { func TestIngest_PublishesOnTheRequestTenantsTopic(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) tenants := nestedTenants(t, map[string]string{"acme": fullConfig(100), "globex": fullConfig(200)}) for _, id := range []tenant.ID{"acme", "globex"} { @@ -134,7 +134,7 @@ func TestIngest_MissingTable(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := httptest.NewRequestWithContext( context.Background(), @@ -157,7 +157,7 @@ func TestIngest_MissingTable(t *testing.T) { func TestIngest_UnknownTable(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ingestRequest(t, "nonexistent", map[string]any{"x": 1}) w := httptest.NewRecorder() @@ -171,7 +171,7 @@ func TestIngest_UnknownTable(t *testing.T) { func TestIngest_InvalidJSON(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) r := rawIngestRequest(t, "clicks", "application/json", "not json") @@ -185,7 +185,7 @@ func TestIngest_InvalidJSON(t *testing.T) { func TestIngest_SchemaValidation_UnknownField(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "nonexistent_field": 42}) w := httptest.NewRecorder() @@ -199,7 +199,7 @@ func TestIngest_Dedup_FirstTime(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } @@ -215,7 +215,7 @@ func TestIngest_Dedup_Duplicate(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } @@ -241,7 +241,7 @@ func TestIngest_Dedup_Duplicate(t *testing.T) { func TestIngest_PublishError_503(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull)} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home"}) w := httptest.NewRecorder() @@ -255,7 +255,7 @@ func TestIngest_PublishError_503(t *testing.T) { func TestIngest_PublishError_500(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: errors.New("some other error")} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ingestRequest(t, "clicks", map[string]any{"page": "/home"}) w := httptest.NewRecorder() @@ -269,7 +269,7 @@ func TestIngest_PublishError_500(t *testing.T) { func TestIngest_Policy_Forbidden(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -294,7 +294,7 @@ func TestIngest_Policy_Forbidden(t *testing.T) { func TestIngest_Policy_ColumnDenied(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -318,7 +318,7 @@ func TestIngest_Policy_ColumnDenied(t *testing.T) { func TestIngest_Policy_CheckClause_Mismatch(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -346,7 +346,7 @@ func TestIngest_Policy_CheckClause_Mismatch(t *testing.T) { func TestIngest_Policy_CheckClause_Match(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -380,7 +380,7 @@ func TestIngest_Policy_CheckClause_Match(t *testing.T) { func TestIngest_Policy_CheckClause_NumericSpellingMatch(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) countTemplate := "{{ jwt.max_count }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -424,7 +424,7 @@ func TestIngest_Policy_CheckClause_StaticNumericSpelling(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) staticCount := "1.0" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -469,7 +469,7 @@ func TestIngest_Policy_CheckClause_StringClaimStrictEquality(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -502,7 +502,7 @@ func TestIngest_Policy_CheckClause_StringClaimStrictEquality(t *testing.T) { func TestIngest_Policy_CheckClause_NullValue_FailsClosed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -531,7 +531,7 @@ func TestIngest_Policy_CheckClause_NullValue_FailsClosed(t *testing.T) { func TestIngest_Policy_CheckClause_AutoInject(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -577,7 +577,7 @@ func checkInStore() PolicySource { func TestIngest_Policy_CheckIn_InSet(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = checkInStore() // org_id is one of the token's allowed orgs — should pass. @@ -596,7 +596,7 @@ func TestIngest_Policy_CheckIn_InSet(t *testing.T) { func TestIngest_Policy_CheckIn_NotInSet(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = checkInStore() // org_id is NOT one of the token's allowed orgs — forging another tenant's row. @@ -621,7 +621,7 @@ func TestIngest_Policy_CheckIn_NotInSet(t *testing.T) { func TestIngest_Policy_CheckIn_NullValue_FailsClosed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = checkInStore() req := ingestRequest(t, "clicks", map[string]any{"page": "/home", "org_id": nil}) @@ -640,7 +640,7 @@ func TestIngest_Policy_CheckIn_NullValue_FailsClosed(t *testing.T) { func TestIngest_Policy_CheckIn_Absent_FailsClosed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = checkInStore() // org_id omitted — unlike _eq there's no single value to auto-inject, so the @@ -667,7 +667,7 @@ func TestIngest_Policy_CheckIn_Absent_FailsClosed(t *testing.T) { func TestIngest_Policy_CheckIn_AbsentClaim_FailsClosed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = checkInStore() // The `orgs` claim is absent entirely, so the _in set resolves to a typed-nil @@ -703,7 +703,7 @@ func TestIngest_DedupIsTheTenants(t *testing.T) { require.NoError(t, stores.For(id).Apply(true)) } pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = func(s *settings.Store) dedupe.Deduplicator { return stores.For(s.Tenant()) } h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } @@ -726,7 +726,7 @@ func TestIngest_Dedup_MissingIDField(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } @@ -745,7 +745,7 @@ func TestIngest_Dedup_MissingIDField(t *testing.T) { func TestIngest_Dedup_RequireID_Rejects(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } @@ -767,7 +767,7 @@ func TestIngest_Dedup_RequireID_Rejects(t *testing.T) { func TestIngest_NDJSON_RequireID_Rejects(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(testutil.NewMockDeduplicator()) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } @@ -794,7 +794,7 @@ func TestIngest_NDJSON_RequireID_Rejects(t *testing.T) { func TestIngest_Policy_DenyColumns(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -817,7 +817,7 @@ func TestIngest_Policy_DenyColumns(t *testing.T) { func TestIngest_AdminRole_NoPolicy(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": {}, @@ -881,7 +881,7 @@ func resultAt(t *testing.T, resp batchResult, index int) recordResult { func TestIngest_NDJSON_AllValid(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a", "count": 1}), @@ -909,7 +909,7 @@ func TestIngest_NDJSON_AllValid(t *testing.T) { func TestIngest_NDJSON_PartialFailure_Validation(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"}), @@ -935,7 +935,7 @@ func TestIngest_NDJSON_PartialFailure_Validation(t *testing.T) { func TestIngest_NDJSON_MalformedLine(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"}), @@ -960,7 +960,7 @@ func TestIngest_NDJSON_MalformedLine(t *testing.T) { func TestIngest_NDJSON_BlankLinesSkipped(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // Leading, interior, and whitespace-only lines are all skipped; only real // records are counted. @@ -997,7 +997,7 @@ func TestIngest_NDJSON_EmptyBody(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", tt.lines...) w := httptest.NewRecorder() @@ -1015,7 +1015,7 @@ func TestIngest_NDJSON_Dedup(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", false } @@ -1044,7 +1044,7 @@ func TestIngest_NDJSON_Backpressure_503(t *testing.T) { // Publisher rejects every publish with the backpressure sentinel; the first // valid record aborts the whole batch with 503 + Retry-After. pub := &testutil.MockPublisher{Err: fmt.Errorf("%w: maximum bytes exceeded", mq.ErrQueueFull)} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"}), @@ -1061,7 +1061,7 @@ func TestIngest_NDJSON_Backpressure_503(t *testing.T) { func TestIngest_NDJSON_PublishError_500(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{Err: errors.New("some other error")} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"})) w := httptest.NewRecorder() @@ -1075,7 +1075,7 @@ func TestIngest_NDJSON_PublishError_500(t *testing.T) { func TestIngest_NDJSON_Policy_ColumnDenied_PerLine(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -1108,7 +1108,7 @@ func TestIngest_NDJSON_Policy_ColumnDenied_PerLine(t *testing.T) { func TestIngest_NDJSON_Policy_TableForbidden(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ "clicks": { @@ -1135,7 +1135,7 @@ func TestIngest_NDJSON_Policy_TableForbidden(t *testing.T) { func TestIngest_NDJSON_Policy_CheckClause_PerLineAndAutoInject(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) orgTemplate := "{{ jwt.org_id }}" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -1175,7 +1175,7 @@ func TestIngest_NDJSON_Policy_CheckClause_PerLineAndAutoInject(t *testing.T) { func TestIngest_NDJSON_ContentTypeWithCharset(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ndjsonRequest(t, "clicks", jsonLine(t, map[string]any{"page": "/a"})) req.Header.Set("Content-Type", "application/x-ndjson; charset=utf-8") @@ -1192,7 +1192,7 @@ func TestIngest_NDJSON_ContentTypeWithCharset(t *testing.T) { func TestIngest_NDJSON_ErrorsTruncated(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) const total = maxReportedResults + 50 lines := make([]string, total) @@ -1376,7 +1376,7 @@ func TestIngest_UndeclaredOrUnsupportedContentType_415(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", tt.ct, `{"page":"/a"}`))) @@ -1426,7 +1426,7 @@ func TestIngest_ContentTypeRefusalBeatsEmptyBody(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", ct, ""))) @@ -1445,7 +1445,7 @@ func TestIngest_ContentTypeRefusalBeatsEmptyBody(t *testing.T) { func TestIngest_DeclaredNDJSON_ArrayBodyIsNotReframed(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", "application/x-ndjson", `[{"page":"/a"},{"page":"/b"}]`))) @@ -1492,7 +1492,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("disagreeing declarations are refused", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/json", ndjson) req.Header.Add("Content-Type", "application/x-ndjson") @@ -1517,7 +1517,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("a supported and an unsupported declaration are refused", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/json", ndjson) req.Header.Add("Content-Type", "text/csv") @@ -1536,7 +1536,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("different spellings of the same format are accepted", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/x-ndjson", ndjson) req.Header.Add("Content-Type", "application/ndjson; charset=utf-8") @@ -1566,7 +1566,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", ct, ndjson))) @@ -1611,7 +1611,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() wJ := httptest.NewRecorder() - NewIngestHandler(testRegistry(t), &testutil.MockPublisher{}). + NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}). Handle(wJ, withTenant(rawIngestRequest(t, "clicks", tc.joined, `{"page":"/a"}`))) assert.Equal(t, tc.wJoined, wJ.Code, "joined") @@ -1620,7 +1620,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { req.Header.Add("Content-Type", v) } wR := httptest.NewRecorder() - NewIngestHandler(testRegistry(t), &testutil.MockPublisher{}). + NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}). Handle(wR, withTenant(req)) assert.Equal(t, tc.wRepeat, wR.Code, "repeated") }) @@ -1630,7 +1630,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("a quoted comma does not split a declaration", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(rawIngestRequest(t, "clicks", `application/json; profile="a,b"`, `{"page":"/a"}`))) @@ -1641,7 +1641,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("a third line that disagrees is refused", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/json", ndjson) req.Header.Add("Content-Type", "application/json") req.Header.Add("Content-Type", "application/x-ndjson") @@ -1663,7 +1663,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Parallel() for _, first := range []bool{false, true} { pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "", ndjson) if first { req.Header.Add("Content-Type", empty) @@ -1689,7 +1689,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("two unsupported lines name both", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "text/csv", ndjson) req.Header.Add("Content-Type", "text/plain") @@ -1708,7 +1708,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { t.Run("an identical declaration repeated is not ambiguous", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/x-ndjson", ndjson) req.Header.Add("Content-Type", "application/x-ndjson") @@ -1723,7 +1723,7 @@ func TestIngest_DuplicateContentTypeHeaders(t *testing.T) { func TestIngest_JSONArray_AllValid(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // A JSON array declared as application/json is read as a batch — the body's // first byte picks arity within the family ingestRequest declares. @@ -1748,7 +1748,7 @@ func TestIngest_JSONArray_AllValid(t *testing.T) { func TestIngest_JSONArray_SingleElement(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // A one-element array is still a batch (returns the results envelope, not // the single-object {"ok":true}). @@ -1768,7 +1768,7 @@ func TestIngest_JSONArray_SingleElement(t *testing.T) { func TestIngest_JSONArray_PartialValidationFailure(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := ingestRequest(t, "clicks", []map[string]any{ {"page": "/a"}, @@ -1793,7 +1793,7 @@ func TestIngest_JSONArray_PartialValidationFailure(t *testing.T) { func TestIngest_JSONArray_ScalarElements(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // Non-object elements (number, string, nested array) are wrong-typed: the // decoder stays in sync, so each is a per-record error and the objects @@ -1824,7 +1824,7 @@ func TestIngest_JSONArray_ScalarElements(t *testing.T) { func TestIngest_JSONArray_SyntaxError_Fatal(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // A structural syntax error desyncs the decoder — the whole request fails // (400), unlike a per-element type error. The leading good element may have @@ -1858,7 +1858,7 @@ func TestIngest_JSONArray_Truncated_Fatal(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/json", tt.body) w := httptest.NewRecorder() @@ -1874,7 +1874,7 @@ func TestIngest_JSONArray_Truncated_Fatal(t *testing.T) { func TestIngest_JSONArray_Empty(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // An explicit empty array is a valid, record-less batch → 200 with no rows. req := rawIngestRequest(t, "clicks", "application/json", `[]`) @@ -1891,7 +1891,7 @@ func TestIngest_JSONArray_Empty(t *testing.T) { func TestIngest_SingleObject_PrettyPrinted(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // A multi-line (pretty-printed) single object must not be mistaken for // NDJSON — it's one record on the single-object path. @@ -1909,7 +1909,7 @@ func TestIngest_SingleObject_PrettyPrinted(t *testing.T) { func TestIngest_DeclaredJSON_ConcatenatedObjects_FirstOnly(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) // Two concatenated objects declared as application/json take the // single-object path and ingest only the first (matching the historical @@ -1942,7 +1942,7 @@ func TestIngest_LeadingWhitespace_Sniff(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "application/json", tt.body) w := httptest.NewRecorder() @@ -1980,7 +1980,7 @@ func TestIngest_EmptyBody(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", tt.contentType, tt.body) w := httptest.NewRecorder() @@ -2011,7 +2011,7 @@ func TestIngest_BodyReadFailure_400(t *testing.T) { t.Run(ct, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := httptest.NewRequestWithContext(context.Background(), http.MethodPost, "/v1/ingest?table=clicks", iotest.ErrReader(errors.New("connection reset by peer"))) @@ -2063,7 +2063,7 @@ func TestIngest_BodyCap_413(t *testing.T) { t.Run(tt.name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.maxRequestBytes = tt.cap // below the body req := rawIngestRequest(t, "clicks", tt.ct, tt.body) @@ -2097,7 +2097,7 @@ func TestIngest_ContentTypeResolvesBeforeTheBodyIsRead(t *testing.T) { t.Run("a body that cannot be read at all", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := httptest.NewRequestWithContext(context.Background(), http.MethodPost, "/v1/ingest?table=clicks", iotest.ErrReader(errors.New("connection reset by peer"))) @@ -2115,7 +2115,7 @@ func TestIngest_ContentTypeResolvesBeforeTheBodyIsRead(t *testing.T) { t.Run("a body over the cap", func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.maxRequestBytes = 50 // below the body req := rawIngestRequest(t, "clicks", "text/csv", @@ -2179,7 +2179,7 @@ func publishedRow(t *testing.T, payload []byte) map[string]any { func TestIngest_TimestampsCanonicalized(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(tsRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) req := ingestRequest(t, "events", map[string]any{ "name": "e", @@ -2207,7 +2207,7 @@ func TestIngest_TimestampsCanonicalized(t *testing.T) { func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(tsRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) staticTS := "2026-06-21 04:00:00" h.PolicySource = staticPolicy(&policy.Policy{ Tables: map[string]policy.TablePolicy{ @@ -2239,7 +2239,7 @@ func TestIngest_AutoInjectedLiteralTimestampCanonicalized(t *testing.T) { func TestIngest_TimestampGarbage_PassesThrough(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(tsRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) req := ingestRequest(t, "events", map[string]any{"name": "e", "ts": "banana"}) w := httptest.NewRecorder() @@ -2254,7 +2254,7 @@ func TestIngest_TimestampGarbage_PassesThrough(t *testing.T) { func TestIngest_Batch_MixedTimestampSpellings(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(tsRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(tsRegistry(t)), pub) req := ingestRequest(t, "events", []map[string]any{ {"name": "a", "ts": "2026-06-21T04:00:00Z"}, @@ -2306,7 +2306,7 @@ func TestIngest_Dedup_DisabledBySettings(t *testing.T) { pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() dedup.Err = errors.New("must not be called while disabled") - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return false, "event_id", true } @@ -2326,7 +2326,7 @@ func TestIngest_Dedup_DisabledMidReload(t *testing.T) { pub := &testutil.MockPublisher{} dedup := testutil.NewMockDeduplicator() dedup.Err = dedupe.ErrDisabled - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.Dedup = staticDedup(dedup) h.DedupeSettings = func(*settings.Store, string) (bool, string, bool) { return true, "event_id", true } @@ -2359,7 +2359,7 @@ func TestProcessRecord_UnresolvedInsertSideAborts(t *testing.T) { }, } reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{schema}) - h := NewIngestHandler(reg, &testutil.MockPublisher{}) + h := NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}) // A grant resolved for SELECT, reaching the insert path. selectResolved := policy.Evaluate(&policy.Policy{ @@ -2417,7 +2417,7 @@ func TestIngest_ContentTypeEchoIsBounded(t *testing.T) { t.Run(name, func(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) w := httptest.NewRecorder() h.Handle(w, withTenant(build(t))) @@ -2445,7 +2445,7 @@ func TestIngest_ContentTypeEchoIsBounded(t *testing.T) { req.Header.Add("Content-Type", fmt.Sprintf("application/%04d", i)+strings.Repeat("\xff", 112)) } w := httptest.NewRecorder() - NewIngestHandler(testRegistry(t), &testutil.MockPublisher{}).Handle(w, withTenant(req)) + NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}).Handle(w, withTenant(req)) require.Equal(t, http.StatusUnsupportedMediaType, w.Code) return w.Body.Len() } @@ -2469,7 +2469,7 @@ func TestIngest_ContentTypeEchoIsBounded(t *testing.T) { func TestIngest_ConflictMessageNamesTheDisagreement(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "", "{\"page\":\"/a\"}\n{\"page\":\"/b\"}") for range 4 { req.Header.Add("Content-Type", "application/json") @@ -2499,7 +2499,7 @@ func TestIngest_ConflictMessageNamesTheDisagreement(t *testing.T) { func TestIngest_ConflictMessageNamesADifferentSpelling(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) req := rawIngestRequest(t, "clicks", "", `{"page":"/a"}`) for _, ct := range []string{ "application/json", @@ -2529,7 +2529,7 @@ func TestIngest_ConflictMessageNamesADifferentSpelling(t *testing.T) { // declaration buried. Nothing covered log CONTENT, because the tests discarded it. func TestIngest_ConflictLogNamesTheDisagreement(t *testing.T) { buf := logtest.Capture(t, slog.LevelInfo) - h := NewIngestHandler(testRegistry(t), &testutil.MockPublisher{}) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), &testutil.MockPublisher{}) req := rawIngestRequest(t, "clicks", "", `{"page":"/a"}`) for _, ct := range []string{ @@ -2557,7 +2557,7 @@ func TestIngest_ConflictLogNamesTheDisagreement(t *testing.T) { func TestIngest_CheckColumnNotInSchema_Rejected(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "tenant_id", "acme")) w := httptest.NewRecorder() @@ -2590,7 +2590,7 @@ func TestIngest_CheckOnComputedColumn_Rejected(t *testing.T) { }}}, }} pub := &testutil.MockPublisher{} - h := NewIngestHandler(computedRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(computedRegistry(t)), pub) h.PolicySource = staticPolicy(p) w := httptest.NewRecorder() @@ -2613,7 +2613,7 @@ func TestIngest_CheckOnComputedColumn_Rejected(t *testing.T) { func TestIngest_CheckColumnInSchema_StillInjects(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "org_id", "org-42")) w := httptest.NewRecorder() @@ -2638,7 +2638,7 @@ func TestIngest_CheckColumnInSchema_StillInjects(t *testing.T) { func TestIngest_CheckGuardLogsOncePerRequest(t *testing.T) { buf := logtest.Capture(t, slog.LevelInfo) pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "tenant_id", "acme")) const n = 25 @@ -2670,7 +2670,7 @@ func TestIngest_CheckGuardLogsOncePerRequest(t *testing.T) { func TestIngest_CheckColumnNotInSchema_BatchRejectsPerRecord(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(testRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(testRegistry(t)), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "tenant_id", "acme")) req := rawIngestRequest(t, "clicks", "application/json", @@ -2729,7 +2729,7 @@ func computedRegistry(t testing.TB) *discovery.SchemaRegistry { func TestIngest_CheckOnEphemeralColumn_Rejected(t *testing.T) { t.Parallel() pub := &testutil.MockPublisher{} - h := NewIngestHandler(computedRegistry(t), pub) + h := NewIngestHandler(fixedRegistry(computedRegistry(t)), pub) h.PolicySource = staticPolicy(checkColumnPolicy(t, "raw", "anything")) w := httptest.NewRecorder() diff --git a/internal/api/pipes.go b/internal/api/pipes.go index 1951d3b7..66c2a249 100644 --- a/internal/api/pipes.go +++ b/internal/api/pipes.go @@ -28,13 +28,15 @@ type PipesHandler struct { // is tenant-exempt, so they carry no request tenant and read the one // ?tenant= names, the default one without it (opsStore). Tenants *settings.Registry - CHConn driver.Conn - Cache cache.Cache - sf singleflight.Group - // queryTimeout bounds each pipe execution, read per request - // (chconn.Manager.QueryTimeout in production) so a settings reload - // applies without a restart. - queryTimeout func() time.Duration + // CHConn yields the request tenant's connection (chconn.Pools.For in + // production); nil is a tenant on no pool, a 503. + CHConn func(*settings.Store) driver.Conn + Cache cache.Cache + sf singleflight.Group + // queryTimeout bounds each pipe execution, read per request off the + // tenant's settings ((*settings.Store).ClickHouse().QueryTimeout in + // production) so a settings reload applies without a restart. + queryTimeout func(*settings.Store) time.Duration // maxRequestBytes optionally overrides the default inbound request body // cap (maxControlBodyBytes) for the body-decoding path (Execute). @@ -44,7 +46,7 @@ type PipesHandler struct { maxRequestBytes int64 } -func NewPipesHandler(source func(*settings.Store) pipes.Source, policySource PolicySource, conn driver.Conn, c cache.Cache, queryTimeout func() time.Duration) *PipesHandler { +func NewPipesHandler(source func(*settings.Store) pipes.Source, policySource PolicySource, conn func(*settings.Store) driver.Conn, c cache.Cache, queryTimeout func(*settings.Store) time.Duration) *PipesHandler { return &PipesHandler{Source: source, PolicySource: policySource, CHConn: conn, Cache: c, queryTimeout: queryTimeout} } @@ -150,6 +152,15 @@ func (h *PipesHandler) Execute(w http.ResponseWriter, r *http.Request) { return } + // The tenant's pool, ahead of the cache: a tenant on none — its tuple + // could not be opened, such as by the connection ceiling — fails + // closed rather than serve what it cached before (#583 story 6). + conn := connOf(h.CHConn, store) + if conn == nil { + writeUnavailable(w, noConnectionMessage, retryAfterPool) + return + } + // Cache. A pipe can read several tables, but the current pipe impl doesn't // expose its table/scope dependencies, so we pass no deps: the result is keyed // by the tenant and sha alone (TTL-only) and the ingest worker cannot @@ -169,12 +180,12 @@ func (h *PipesHandler) Execute(w http.ResponseWriter, r *http.Request) { // Execute with singleflight. v, err, _ := h.sf.Do(cacheKey, func() (interface{}, error) { - queryCtx, cancel := context.WithTimeout(r.Context(), h.queryTimeout()) + queryCtx, cancel := context.WithTimeout(r.Context(), timeoutOf(h.queryTimeout, store)) defer cancel() start := time.Now() - rows, err := executeCHQuery(queryCtx, h.CHConn, sql, params) + rows, err := executeCHQuery(queryCtx, conn, sql, params) queryDuration := time.Since(start) if err != nil { // TODO: depending on the error, we may actually want to cache it diff --git a/internal/api/pipes_test.go b/internal/api/pipes_test.go index 374386de..db7ff2c4 100644 --- a/internal/api/pipes_test.go +++ b/internal/api/pipes_test.go @@ -41,7 +41,7 @@ func pipesRequest(t *testing.T, method, path, name string, body any) *http.Reque // noTimeout is the pipe-execution deadline source for handler tests that // never reach ClickHouse. -func noTimeout() time.Duration { return 0 } +func noTimeout(*settings.Store) time.Duration { return 0 } func TestPipesHandler_List(t *testing.T) { t.Parallel() diff --git a/internal/api/query.go b/internal/api/query.go index 8d1d0e9d..72f25246 100644 --- a/internal/api/query.go +++ b/internal/api/query.go @@ -13,6 +13,7 @@ import ( "time" "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/settings" ) // QueryHandler handles POST /v1/ops/query. @@ -43,16 +44,21 @@ import ( // - The cache (TieredCache + singleflight) was already removed in an // earlier commit; this completes the simplification. type QueryHandler struct { - // clients follows the target's TLS config (chconn.HTTPClients). + // clients follows each target's TLS config (chconn.HTTPClients). clients *chconn.HTTPClients - // target resolves the ClickHouse HTTP wiring per request (chconn.Manager - // in production): the base URL (e.g. `http://localhost:8123`) the handler - // appends query-string params (`default_format`, `database`, - // `date_time_output_format`) to and POSTs the SQL against, plus the - // credentials and database. queryTimeout bounds each proxied query. + // target resolves the tenant's ClickHouse HTTP wiring per request + // (chconn.Pools.Target in production): the base URL (e.g. + // `http://localhost:8123`) the handler appends query-string params + // (`default_format`, `database`, `date_time_output_format`) to and POSTs + // the SQL against, plus the credentials and database; the zero Target is + // a tenant on no pool, a 503. queryTimeout bounds each proxied query. // Funcs, not values, so a settings reload applies to the next request. - target func() chconn.Target - queryTimeout func() time.Duration + target func(*settings.Store) chconn.Target + queryTimeout func(*settings.Store) time.Duration + // Tenants resolves the tenant the proxy queries: /v1/ops is + // tenant-exempt, so the request carries no tenant and the query runs + // against the one ?tenant= names, the default one without it (opsStore). + Tenants *settings.Registry // maxResponseBytes optionally overrides the default upstream response // buffer cap (maxCHResponseBytes). When 0, the default applies. Exists // so same-package tests can pin the cap-overflow path without @@ -112,7 +118,7 @@ const ( // bounds the whole exchange including body read. Setting `Timeout` here // too would just duplicate that bound (and silently truncate any // inbound context longer than queryTimeout). -func NewQueryHandler(target func() chconn.Target, queryTimeout func() time.Duration) *QueryHandler { +func NewQueryHandler(target func(*settings.Store) chconn.Target, queryTimeout func(*settings.Store) time.Duration) *QueryHandler { return &QueryHandler{ clients: chconn.NewHTTPClients(proxyHTTPClient), target: target, @@ -155,6 +161,11 @@ func (h *QueryHandler) Handle(w http.ResponseWriter, r *http.Request) { // matches writeJSONError's posture on the error path. w.Header().Set("X-Content-Type-Options", "nosniff") + store, ok := opsStore(w, r, h.Tenants) + if !ok { + return + } + reqCap := int64(maxRequestBodyBytes) if h.maxRequestBytes > 0 { reqCap = h.maxRequestBytes @@ -197,7 +208,11 @@ func (h *QueryHandler) Handle(w http.ResponseWriter, r *http.Request) { return } - target, timeout := h.target(), h.queryTimeout() + target, timeout := h.target(store), timeoutOf(h.queryTimeout, store) + if target.URL == "" { + writeUnavailable(w, noConnectionMessage, retryAfterPool) + return + } u, err := url.Parse(target.URL) if err != nil { writeJSONError(w, http.StatusInternalServerError, "invalid clickhouse endpoint: "+err.Error()) diff --git a/internal/api/query_test.go b/internal/api/query_test.go index 8fe241d6..49ded65e 100644 --- a/internal/api/query_test.go +++ b/internal/api/query_test.go @@ -17,6 +17,7 @@ import ( "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/Wave-RF/WaveHouse/internal/testutil" ) @@ -29,10 +30,18 @@ func safeHandle(handler http.HandlerFunc, w *httptest.ResponseRecorder, r *http. handler(w, r) } +// newTestQueryHandler is NewQueryHandler over the test tenant registry, so +// a handler called without the router still resolves ?tenant=. +func newTestQueryHandler(target func(*settings.Store) chconn.Target, timeout func(*settings.Store) time.Duration) *QueryHandler { + h := NewQueryHandler(target, timeout) + h.Tenants = testTenants() + return h +} + // staticTarget is a fixed-wiring source for tests — production hands the -// handler chconn.Manager.Target. -func staticTarget(url, username, password, database string) func() chconn.Target { - return func() chconn.Target { +// handler chconn.Pools.Target by the request's tenant. +func staticTarget(url, username, password, database string) func(*settings.Store) chconn.Target { + return func(*settings.Store) chconn.Target { return chconn.Target{URL: url, Username: username, Password: password, Database: database} } } @@ -41,7 +50,7 @@ func newProxyHandler(t *testing.T, fakeCH http.Handler) *QueryHandler { t.Helper() srv := httptest.NewServer(fakeCH) t.Cleanup(srv.Close) - return NewQueryHandler(staticTarget(srv.URL, "default", "secret", "default"), func() time.Duration { return 30 * time.Second }) + return newTestQueryHandler(staticTarget(srv.URL, "default", "secret", "default"), func(*settings.Store) time.Duration { return 30 * time.Second }) } func postQuery(h *QueryHandler, body []byte) *httptest.ResponseRecorder { @@ -118,7 +127,7 @@ func TestQueryHandler_RejectsMalformedRequests(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { t.Parallel() - h := NewQueryHandler(staticTarget("http://unused.invalid", "", "", ""), func() time.Duration { return 30 * time.Second }) + h := newTestQueryHandler(staticTarget("http://unused.invalid", "", "", ""), func(*settings.Store) time.Duration { return 30 * time.Second }) w := httptest.NewRecorder() r := httptest.NewRequestWithContext(context.Background(), http.MethodPost, "/v1/ops/query", bytes.NewReader([]byte(tt.body))) h.Handle(w, r) @@ -139,7 +148,7 @@ func TestQueryHandler_RejectsMalformedRequests(t *testing.T) { // that surfaces via the chi recoverer. func TestQueryHandler_NilHTTPClientReturnsError(t *testing.T) { t.Parallel() - h := &QueryHandler{target: staticTarget("http://unused.invalid", "", "", ""), queryTimeout: func() time.Duration { return time.Second }} // no HTTPClient + h := &QueryHandler{target: staticTarget("http://unused.invalid", "", "", ""), queryTimeout: func(*settings.Store) time.Duration { return time.Second }, Tenants: testTenants()} // no HTTPClient body, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) w := postQuery(h, body) @@ -362,7 +371,7 @@ func TestQueryHandler_NoAuthHeadersWhenBlank(t *testing.T) { }) srv := httptest.NewServer(fake) defer srv.Close() - h := NewQueryHandler(staticTarget(srv.URL, "", "", ""), func() time.Duration { return 30 * time.Second }) + h := newTestQueryHandler(staticTarget(srv.URL, "", "", ""), func(*settings.Store) time.Duration { return 30 * time.Second }) body, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) w := postQuery(h, body) @@ -442,7 +451,7 @@ func TestQueryHandler_RequestBodyCap(t *testing.T) { t.Parallel() const testCap = 64 - h := NewQueryHandler(staticTarget("http://unused.invalid", "", "", ""), func() time.Duration { return 30 * time.Second }) + h := newTestQueryHandler(staticTarget("http://unused.invalid", "", "", ""), func(*settings.Store) time.Duration { return 30 * time.Second }) h.maxRequestBytes = testCap body, _ := json.Marshal(queryRequest{SQL: strings.Repeat("x", 200)}) @@ -482,7 +491,7 @@ func TestQueryHandler_ContextCancelPropagates(t *testing.T) { defer srv.Close() defer close(allowReturn) - h := NewQueryHandler(staticTarget(srv.URL, "", "", ""), func() time.Duration { return 30 * time.Second }) + h := newTestQueryHandler(staticTarget(srv.URL, "", "", ""), func(*settings.Store) time.Duration { return 30 * time.Second }) body, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) w := httptest.NewRecorder() @@ -532,13 +541,13 @@ func TestQueryHandler_SetsConfiguredHeaders(t *testing.T) { w.WriteHeader(http.StatusOK) })) t.Cleanup(srv.Close) - target := func() chconn.Target { + target := func(*settings.Store) chconn.Target { return chconn.Target{ URL: srv.URL, Username: "default", Password: "secret", Database: "default", Headers: map[string]string{"X-Proxy-Token": "abc", "Content-Type": "application/json", "X-ClickHouse-Key": "someone-else"}, } } - h := NewQueryHandler(target, func() time.Duration { return 30 * time.Second }) + h := newTestQueryHandler(target, func(*settings.Store) time.Duration { return 30 * time.Second }) body, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) w := postQuery(h, body) require.Equal(t, http.StatusOK, w.Code, w.Body.String()) @@ -558,9 +567,9 @@ func TestQueryHandler_UsesTargetTLS(t *testing.T) { pool := x509.NewCertPool() pool.AddCert(srv.Certificate()) handler := func(cfg *tls.Config) *QueryHandler { - return NewQueryHandler(func() chconn.Target { + return newTestQueryHandler(func(*settings.Store) chconn.Target { return chconn.Target{URL: srv.URL, Username: "default", Password: "secret", Database: "default", TLS: cfg} - }, func() time.Duration { return 30 * time.Second }) + }, func(*settings.Store) time.Duration { return 30 * time.Second }) } body, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) diff --git a/internal/api/router_test.go b/internal/api/router_test.go index 1c371801..e0240ba6 100644 --- a/internal/api/router_test.go +++ b/internal/api/router_test.go @@ -338,12 +338,12 @@ func TestNewRouter_RoutesRegistered(t *testing.T) { deps := Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, pub), + Ingest: NewIngestHandler(fixedRegistry(reg), pub), Query: &QueryHandler{}, SSE: NewStreamHandler(hub, nil), Health: &HealthHandler{}, Version: NewVersionHandler("test", "test", "test"), - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), DLQ: NewDLQHandler(emb), Pipes: &PipesHandler{Source: staticPipes(), PolicySource: staticPolicy(&policy.Policy{}), Tenants: testTenants()}, AuthMW: func(next http.Handler) http.Handler { return next }, @@ -520,11 +520,11 @@ func TestNewRouter_RawSQLAdminGate(t *testing.T) { router := NewRouter(Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, pub), + Ingest: NewIngestHandler(fixedRegistry(reg), pub), Query: &QueryHandler{}, SSE: NewStreamHandler(hub, nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return next }, PolicySource: policy.Static(&policy.Policy{}), }) @@ -586,11 +586,11 @@ func TestNewRouter_NestedOpsGateAdmitsTheOperatorKeyAlone(t *testing.T) { pipesHandler.Tenants = tenants return NewRouter(Dependencies{ Tenants: tenants, - Ingest: NewIngestHandler(reg, &testutil.MockPublisher{}), + Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, tenants), Pipes: pipesHandler, Settings: NewSettingsHandler(tenants), AuthMW: func(next http.Handler) http.Handler { return next }, @@ -663,11 +663,11 @@ func TestNewRouter_MalformedTenantSurvivesTheTokenStrip(t *testing.T) { pipesHandler.Tenants = tenants router := NewRouter(Dependencies{ Tenants: tenants, - Ingest: NewIngestHandler(reg, &testutil.MockPublisher{}), + Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, tenants), Pipes: pipesHandler, Settings: NewSettingsHandler(tenants), AuthMW: authn.Middleware(), @@ -731,11 +731,11 @@ func TestNewRouter_TokenUnderPendingJWKSIs503(t *testing.T) { pipesHandler.Tenants = tenants router := NewRouter(Dependencies{ Tenants: tenants, - Ingest: NewIngestHandler(reg, &testutil.MockPublisher{}), + Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: NewSchemaHandler(fixedRegistry(reg)), Pipes: pipesHandler, AuthMW: authn.Middleware(), }) @@ -775,11 +775,11 @@ func TestNewRouter_OptionalDepsNil(t *testing.T) { deps := Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, pub), + Ingest: NewIngestHandler(fixedRegistry(reg), pub), Query: &QueryHandler{}, SSE: NewStreamHandler(hub, nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return next }, PolicySource: policy.Static(&policy.Policy{}), } @@ -853,11 +853,11 @@ func TestNewRouter_NotFoundEmitsJSON(t *testing.T) { hub := stream.NewHub(nil, nil, nil) deps := Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, pub), + Ingest: NewIngestHandler(fixedRegistry(reg), pub), Query: &QueryHandler{}, SSE: NewStreamHandler(hub, nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return next }, } router := NewRouter(deps) @@ -878,11 +878,11 @@ func TestNewRouter_MethodNotAllowedEmitsJSON(t *testing.T) { hub := stream.NewHub(nil, nil, nil) deps := Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, pub), + Ingest: NewIngestHandler(fixedRegistry(reg), pub), Query: &QueryHandler{}, SSE: NewStreamHandler(hub, nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return next }, } router := NewRouter(deps) @@ -978,11 +978,11 @@ func TestNewRouter_SchemaAdminOnly(t *testing.T) { router := NewRouter(Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, pub), + Ingest: NewIngestHandler(fixedRegistry(reg), pub), Query: &QueryHandler{}, SSE: NewStreamHandler(hub, nil), Health: &HealthHandler{}, - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return next }, PolicySource: policy.Static(&policy.Policy{}), }) diff --git a/internal/api/schema.go b/internal/api/schema.go index 3ec7f206..5e8cb5f4 100644 --- a/internal/api/schema.go +++ b/internal/api/schema.go @@ -2,50 +2,124 @@ package api import ( "encoding/json" + "errors" "net/http" "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/settings" ) +// RegistrySource yields a tenant's schema registry, its own since #583 +// story 6 (the per-tenant discovery map in internal/app). Nil for a tenant +// whose registry is not built yet — the beat between its adoption and the +// hook that builds it — which reads as a schema not loaded yet. +type RegistrySource func(*settings.Store) *discovery.SchemaRegistry + +// lookupSchema resolves table's schema for store's tenant, answering the two +// misses itself: a 503 with Retry-After while the tenant's schema is not +// discovered yet (the table may well exist), and a 404 with notFound for a +// table the discovered schema lacks. The error is nil once a schema is +// returned, and names the miss otherwise, after the response was written. +func lookupSchema(w http.ResponseWriter, registry RegistrySource, store *settings.Store, table, notFound string) (*discovery.TableSchema, error) { + reg := registryOf(registry, store) + if reg == nil { + writeUnavailable(w, schemaNotLoadedMessage, retryAfterSchema) + return nil, discovery.ErrNotLoaded + } + schema, err := reg.Lookup(table) + switch { + case errors.Is(err, discovery.ErrNotLoaded): + writeUnavailable(w, schemaNotLoadedMessage, retryAfterSchema) + case err != nil: + writeJSONError(w, http.StatusNotFound, notFound) + } + return schema, err +} + +// registryOf is registry's answer for store, nil for an unwired source. +func registryOf(registry RegistrySource, store *settings.Store) *discovery.SchemaRegistry { + if registry == nil { + return nil + } + return registry(store) +} + // SchemaHandler exposes the discovered ClickHouse table schemas. type SchemaHandler struct { - Registry *discovery.SchemaRegistry + Registry RegistrySource + // Tenants resolves the tenant the reads and the refresh serve: /v1/ops + // is tenant-exempt, so they carry no request tenant and read the one + // ?tenant= names, the default one without it (opsStore). + Tenants *settings.Registry } -func NewSchemaHandler(registry *discovery.SchemaRegistry) *SchemaHandler { +func NewSchemaHandler(registry RegistrySource) *SchemaHandler { return &SchemaHandler{Registry: registry} } -// List returns all discovered table schemas. -func (h *SchemaHandler) List(w http.ResponseWriter, _ *http.Request) { +// List returns all discovered table schemas of the ?tenant=. +func (h *SchemaHandler) List(w http.ResponseWriter, r *http.Request) { + store, ok := opsStore(w, r, h.Tenants) + if !ok { + return + } + h.list(w, store) +} + +// list writes store's tenant's schemas, or the 503 of a schema not +// discovered yet: an empty list would read as "no tables". +func (h *SchemaHandler) list(w http.ResponseWriter, store *settings.Store) { + reg := registryOf(h.Registry, store) + if reg == nil || !reg.Loaded() { + writeUnavailable(w, schemaNotLoadedMessage, retryAfterSchema) + return + } w.Header().Set("Content-Type", "application/json") - _ = json.NewEncoder(w).Encode(h.Registry.List()) + _ = json.NewEncoder(w).Encode(reg.List()) } -// Get returns the schema for a single table. +// Get returns the schema for a single table of the ?tenant=, every table's +// without ?table=. func (h *SchemaHandler) Get(w http.ResponseWriter, r *http.Request) { + store, ok := opsStore(w, r, h.Tenants) + if !ok { + return + } table := r.URL.Query().Get("table") - if table == "" { - h.List(w, r) + h.list(w, store) return } - - schema := h.Registry.Get(table) - if schema == nil { - writeJSONError(w, http.StatusNotFound, "table not found") + schema, err := lookupSchema(w, h.Registry, store, table, "table not found") + if err != nil { return } w.Header().Set("Content-Type", "application/json") _ = json.NewEncoder(w).Encode(schema) } -// Refresh forces an immediate schema refresh from ClickHouse. +// Refresh forces an immediate schema refresh of the ?tenant= from its +// ClickHouse, then returns its schemas. A tenant with no open pool — one +// could not be opened for it, such as by the connection ceiling — is a 503 +// with Retry-After, like the reads it would answer. func (h *SchemaHandler) Refresh(w http.ResponseWriter, r *http.Request) { - if err := h.Registry.Refresh(r.Context()); err != nil { + store, ok := opsStore(w, r, h.Tenants) + if !ok { + return + } + reg := registryOf(h.Registry, store) + if reg == nil { + writeUnavailable(w, schemaNotLoadedMessage, retryAfterSchema) + return + } + if err := reg.Refresh(r.Context()); err != nil { + if errors.Is(err, discovery.ErrNoConnection) { + writeUnavailable(w, noConnectionMessage, retryAfterPool) + return + } writeJSONError(w, http.StatusInternalServerError, "refresh failed") return } w.Header().Set("Content-Type", "application/json") - _ = json.NewEncoder(w).Encode(h.Registry.List()) + _ = json.NewEncoder(w).Encode(reg.List()) } diff --git a/internal/api/schema_test.go b/internal/api/schema_test.go index 39e65ca8..b601646e 100644 --- a/internal/api/schema_test.go +++ b/internal/api/schema_test.go @@ -19,7 +19,8 @@ func TestSchema_List(t *testing.T) { {Name: "clicks", Columns: []discovery.Column{{Name: "page", Type: "String"}}}, {Name: "users", Columns: []discovery.Column{{Name: "name", Type: "String"}}}, }) - h := NewSchemaHandler(reg) + h := NewSchemaHandler(fixedRegistry(reg)) + h.Tenants = testTenants() w := httptest.NewRecorder() r := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/schema", nil) @@ -40,7 +41,8 @@ func TestSchema_Get_Exists(t *testing.T) { {Name: "count", Type: "UInt64"}, }}, }) - h := NewSchemaHandler(reg) + h := NewSchemaHandler(fixedRegistry(reg)) + h.Tenants = testTenants() w := httptest.NewRecorder() r := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/schema?table=clicks", nil) @@ -58,7 +60,8 @@ func TestSchema_Get_Exists(t *testing.T) { func TestSchema_Get_NotFound(t *testing.T) { t.Parallel() reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{}) - h := NewSchemaHandler(reg) + h := NewSchemaHandler(fixedRegistry(reg)) + h.Tenants = testTenants() w := httptest.NewRecorder() r := httptest.NewRequestWithContext(context.Background(), http.MethodGet, "/v1/ops/schema?table=nonexistent", nil) diff --git a/internal/api/structured_query.go b/internal/api/structured_query.go index c13d3cc1..55fd8f35 100644 --- a/internal/api/structured_query.go +++ b/internal/api/structured_query.go @@ -12,7 +12,6 @@ import ( "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/cache" - "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/query" "github.com/Wave-RF/WaveHouse/internal/settings" @@ -21,15 +20,17 @@ import ( // StructuredQueryHandler handles POST /v1/query?table={table} type StructuredQueryHandler struct { - CHConn driver.Conn + // CHConn yields the request tenant's connection (chconn.Pools.For in + // production); nil is a tenant on no pool, a 503. + CHConn func(*settings.Store) driver.Conn Cache cache.Cache - Registry *discovery.SchemaRegistry + Registry RegistrySource PolicySource PolicySource sf singleflight.Group - // queryTimeout bounds each query, read per request - // (chconn.Manager.QueryTimeout in production) so a settings reload - // applies without a restart. - queryTimeout func() time.Duration + // queryTimeout bounds each query, read per request off the tenant's + // settings ((*settings.Store).ClickHouse().QueryTimeout in production) + // so a settings reload applies without a restart. + queryTimeout func(*settings.Store) time.Duration // bucketSecs returns the request tenant's current time-range bucket // ((*settings.Store).TimestampBucketSeconds in production) and @@ -50,12 +51,12 @@ type StructuredQueryHandler struct { } func NewStructuredQueryHandler( - conn driver.Conn, + conn func(*settings.Store) driver.Conn, c cache.Cache, - registry *discovery.SchemaRegistry, + registry RegistrySource, policyStore PolicySource, bucketSecs func(*settings.Store) int, - queryTimeout func() time.Duration, + queryTimeout func(*settings.Store) time.Duration, defaultMaxRows func(*settings.Store) int, ) *StructuredQueryHandler { return &StructuredQueryHandler{ @@ -80,9 +81,8 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) return } - schema := h.Registry.Get(table) - if schema == nil { - writeJSONError(w, http.StatusNotFound, "unknown table: "+table) + schema, err := lookupSchema(w, h.Registry, store, table, "unknown table: "+table) + if err != nil { return } @@ -159,6 +159,15 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) return } + // The tenant's pool, ahead of the cache: a tenant on none — its tuple + // could not be opened, such as by the connection ceiling — fails + // closed rather than serve what it cached before (#583 story 6). + conn := connOf(h.CHConn, store) + if conn == nil { + writeUnavailable(w, noConnectionMessage, retryAfterPool) + return + } + // Cache key, led by the tenant the store was resolved for (#583 story 8); // the singleflight key too. cacheKey := queryCacheKey(store.Tenant(), result.SQL, result.Params) @@ -185,7 +194,7 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) // Execute with singleflight. v, err, _ := h.sf.Do(cacheKey, func() (interface{}, error) { - timeout := h.queryTimeout() + timeout := timeoutOf(h.queryTimeout, store) // Bare Select reads: this handler resolved the grant for "select" (above), // so Select is non-nil, and query.Build has already rejected a mis-resolved // grant before this closure runs. If that changed, these would panic rather @@ -219,7 +228,7 @@ func (h *StructuredQueryHandler) Handle(w http.ResponseWriter, r *http.Request) start := time.Now() - rows, err := executeCHQuery(queryCtx, h.CHConn, result.SQL, result.Params) + rows, err := executeCHQuery(queryCtx, conn, result.SQL, result.Params) queryDuration := time.Since(start) if err != nil { // TODO: depending on the error, we may actually want to cache it diff --git a/internal/api/structured_query_test.go b/internal/api/structured_query_test.go index ada8d5d9..c715502b 100644 --- a/internal/api/structured_query_test.go +++ b/internal/api/structured_query_test.go @@ -42,7 +42,7 @@ func newStructuredQueryHandler(t testing.TB) *StructuredQueryHandler { }, }, }) - return NewStructuredQueryHandler(nil, nil, reg, nil, func(*settings.Store) int { return 60 }, func() time.Duration { return 5 * time.Second }, nil) + return NewStructuredQueryHandler(nil, nil, fixedRegistry(reg), nil, func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 5 * time.Second }, nil) } func TestStructuredQuery_MissingTable(t *testing.T) { @@ -291,7 +291,7 @@ func newCapturingHandler(t *testing.T, conn driver.Conn, p *policy.Policy) *Stru }, }, }) - return NewStructuredQueryHandler(conn, nil, reg, staticPolicy(p), func(*settings.Store) int { return 60 }, func() time.Duration { return 5 * time.Second }, nil) + return NewStructuredQueryHandler(fixedConn(conn), nil, fixedRegistry(reg), staticPolicy(p), func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 5 * time.Second }, nil) } func viewerRequest(t *testing.T, sq query.StructuredQuery) *http.Request { diff --git a/internal/api/tenant_clickhouse_test.go b/internal/api/tenant_clickhouse_test.go new file mode 100644 index 00000000..b152ad78 --- /dev/null +++ b/internal/api/tenant_clickhouse_test.go @@ -0,0 +1,299 @@ +package api + +import ( + "bytes" + "context" + "encoding/json" + "net" + "net/http" + "net/http/httptest" + "strings" + "testing" + "time" + + "github.com/ClickHouse/clickhouse-go/v2/lib/driver" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/auth" + "github.com/Wave-RF/WaveHouse/internal/chconn" + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/pipes" + "github.com/Wave-RF/WaveHouse/internal/policy" + "github.com/Wave-RF/WaveHouse/internal/query" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/stream" + "github.com/Wave-RF/WaveHouse/internal/tenant" + "github.com/Wave-RF/WaveHouse/internal/testutil" +) + +// closedAddr returns a 127.0.0.1 address nothing listens on: bind an +// ephemeral port, then release it. +func closedAddr(t *testing.T) string { + t.Helper() + var lc net.ListenConfig + ln, err := lc.Listen(t.Context(), "tcp", "127.0.0.1:0") + require.NoError(t, err) + addr := ln.Addr().String() + require.NoError(t, ln.Close()) + return addr +} + +// unloadedRegistry is a registry no refresh has ever succeeded on: a tenant +// whose ClickHouse has not answered yet, or which has no pool. +func unloadedRegistry() *discovery.SchemaRegistry { + return discovery.NewSchemaRegistry(func() (driver.Conn, string) { return nil, "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) +} + +// assertUnavailable pins the 503 a tenant's ClickHouse side answers with: the +// message, the Retry-After hint, and the JSON error contract. +func assertUnavailable(t *testing.T, w *httptest.ResponseRecorder, message, retryAfter string) { + t.Helper() + require.Equal(t, http.StatusServiceUnavailable, w.Code, "body: %s", w.Body.String()) + assert.Equal(t, retryAfter, w.Header().Get("Retry-After")) + assert.Contains(t, w.Body.String(), message) + testutil.AssertJSONErrorResponse(t, w) +} + +// A table lookup before the tenant's first discovery is a 503 with +// Retry-After, not the 404 of a table the schema lacks: the table may well +// exist. Every route that answered 404 answers it, and the schema list too, +// where an empty list would read as "no tables". A tenant whose registry is +// not built yet — the beat after its adoption — reads the same way. +func TestClickHouseRoutes_SchemaNotLoadedIs503(t *testing.T) { + t.Parallel() + sources := map[string]RegistrySource{ + "unloaded registry": fixedRegistry(unloadedRegistry()), + "no registry yet": fixedRegistry(nil), + "no registry source": nil, + } + for name, source := range sources { + t.Run(name, func(t *testing.T) { + t.Parallel() + schema := NewSchemaHandler(source) + schema.Tenants = testTenants() + structured := NewStructuredQueryHandler(nil, nil, source, nil, nil, noTimeout, nil) + routes := map[string]func() (*httptest.ResponseRecorder, string){ + "ingest": func() (*httptest.ResponseRecorder, string) { + w := httptest.NewRecorder() + NewIngestHandler(source, &testutil.MockPublisher{}).Handle(w, withTenant(ingestRequest(t, "clicks", map[string]any{"page": "/"}))) + return w, schemaNotLoadedMessage + }, + "structured query": func() (*httptest.ResponseRecorder, string) { + w := httptest.NewRecorder() + structured.Handle(w, withTenant(structuredQueryRequest(t, "clicks", selectAllQuery()))) + return w, schemaNotLoadedMessage + }, + "schema get": func() (*httptest.ResponseRecorder, string) { + w := httptest.NewRecorder() + schema.Get(w, httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/ops/schema?table=clicks", nil)) + return w, schemaNotLoadedMessage + }, + "schema list": func() (*httptest.ResponseRecorder, string) { + w := httptest.NewRecorder() + schema.Get(w, httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/ops/schema", nil)) + return w, schemaNotLoadedMessage + }, + } + for route, call := range routes { + w, message := call() + assertUnavailable(t, w, message, retryAfterSchema) + assert.NotContains(t, w.Body.String(), "unknown table", route) + } + }) + } +} + +// selectAllQuery is the structured query a permissive role may run. +func selectAllQuery() query.StructuredQuery { return query.StructuredQuery{SelectAll: true} } + +// A tenant on no pool — its tuple could not be opened, such as by the +// connection ceiling — fails closed on every route that reaches its +// ClickHouse: a 503 with Retry-After ahead of the cache, so nothing it +// cached before is served either, and on the refresh, which cannot run. +func TestClickHouseRoutes_NoPoolIs503(t *testing.T) { + t.Parallel() + reg := testRegistry(t) + allowAll := staticPolicy(&policy.Policy{ + DefaultRole: "viewer", + Tables: map[string]policy.TablePolicy{"clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}}, + }) + noConn := func(*settings.Store) driver.Conn { return nil } + + t.Run("structured query", func(t *testing.T) { + t.Parallel() + h := NewStructuredQueryHandler(noConn, nil, fixedRegistry(reg), allowAll, nil, noTimeout, nil) + w := httptest.NewRecorder() + h.Handle(w, withTenant(structuredQueryRequest(t, "clicks", selectAllQuery()))) + assertUnavailable(t, w, noConnectionMessage, retryAfterPool) + }) + t.Run("pipe execute", func(t *testing.T) { + t.Parallel() + h := NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), allowAll, noConn, nil, noTimeout) + w := httptest.NewRecorder() + h.Execute(w, withTenant(pipesRequest(t, http.MethodGet, "/v1/pipes/top_pages", "top_pages", nil))) + assertUnavailable(t, w, noConnectionMessage, retryAfterPool) + }) + t.Run("raw-SQL proxy", func(t *testing.T) { + t.Parallel() + h := newTestQueryHandler(func(*settings.Store) chconn.Target { return chconn.Target{} }, noTimeout) + body, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) + w := postQuery(h, body) + assertUnavailable(t, w, noConnectionMessage, retryAfterPool) + assertSecurityHeaders(t, w) + }) + t.Run("schema refresh", func(t *testing.T) { + t.Parallel() + h := NewSchemaHandler(fixedRegistry(unloadedRegistry())) + h.Tenants = testTenants() + w := httptest.NewRecorder() + h.Refresh(w, httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ops/schema/refresh", nil)) + assertUnavailable(t, w, noConnectionMessage, retryAfterPool) + }) +} + +// The three ops routes that reach a tenant's ClickHouse name it in ?tenant=, +// parsed as strictly as the pipe reads: the store the getters receive is +// that tenant's, absent is the default tenant, a query that does not parse +// is a 400, and a tenant that cannot be served gets the tenant routes' 404 +// or 503. +func TestClickHouseOpsRoutes_TenantParam(t *testing.T) { + t.Parallel() + tenants := nestedTenants(t, map[string]string{"0": fullConfig(100), "acme": fullConfig(100), "globex": `{"unknown_key": true}`}) + reg := testutil.NewTestSchemaRegistry(t, []*discovery.TableSchema{{Name: "clicks", Columns: []discovery.Column{{Name: "page", Type: "String"}}}}) + tests := []struct { + name, query string + wantStatus int + wantTenant tenant.ID // on 200, whose store the getters were handed + wantBody string + }{ + {name: "no parameter is the default tenant", query: "", wantStatus: http.StatusOK, wantTenant: tenant.Default}, + {name: "named tenant", query: "tenant=acme", wantStatus: http.StatusOK, wantTenant: "acme"}, + {name: "explicit default tenant", query: "tenant=0", wantStatus: http.StatusOK, wantTenant: tenant.Default}, + {name: "rejected tenant", query: "tenant=globex", wantStatus: http.StatusServiceUnavailable, wantBody: "tenant settings are invalid"}, + {name: "unknown tenant", query: "tenant=initech", wantStatus: http.StatusNotFound, wantBody: "unknown tenant: initech"}, + {name: "empty value is not absent", query: "tenant=", wantStatus: http.StatusBadRequest, wantBody: "invalid ?tenant: tenant id is empty"}, + {name: "repeated, even agreeing", query: "tenant=acme&tenant=acme", wantStatus: http.StatusBadRequest, wantBody: "sent more than once"}, + {name: "malformed id", query: "tenant=a.b", wantStatus: http.StatusBadRequest, wantBody: "invalid ?tenant"}, + {name: "semicolon pair", query: "tenant=acme;x=1", wantStatus: http.StatusBadRequest, wantBody: "invalid query string"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + t.Parallel() + var handed []*settings.Store + registry := func(s *settings.Store) *discovery.SchemaRegistry { + handed = append(handed, s) + return reg + } + target := func(s *settings.Store) chconn.Target { + handed = append(handed, s) + return chconn.Target{URL: "http://" + closedAddr(t)} + } + schema := NewSchemaHandler(registry) + schema.Tenants = tenants + proxy := NewQueryHandler(target, noTimeout) + proxy.Tenants = tenants + sql, _ := json.Marshal(queryRequest{SQL: "SELECT 1"}) + + routes := []struct { + name string + call func(w http.ResponseWriter, r *http.Request) + req *http.Request + // ok is the status a served tenant answers: the registry + // getter was handed its store, and the proxy reached its target. + ok int + }{ + {name: "schema get", call: schema.Get, req: httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/ops/schema?table=clicks", nil), ok: http.StatusOK}, + {name: "schema list", call: schema.List, req: httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/ops/schema", nil), ok: http.StatusOK}, + {name: "schema refresh", call: schema.Refresh, req: httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ops/schema/refresh", nil), ok: http.StatusOK}, + // The proxy's target is a closed port: a served tenant is the + // 502 of an unreachable ClickHouse, past every tenant check. + {name: "ops query", call: proxy.Handle, req: httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ops/query", bytes.NewReader(sql)), ok: http.StatusBadGateway}, + } + for _, route := range routes { + handed = nil + if tt.query != "" { + route.req.URL.RawQuery = strings.TrimSuffix(route.req.URL.RawQuery+"&"+tt.query, "&") + if route.req.URL.RawQuery[0] == '&' { + route.req.URL.RawQuery = route.req.URL.RawQuery[1:] + } + } + w := httptest.NewRecorder() + route.call(w, route.req) + if tt.wantStatus != http.StatusOK { + require.Equal(t, tt.wantStatus, w.Code, "%s: %s", route.name, w.Body.String()) + assert.Empty(t, handed, "%s: a refused request must not reach the getters", route.name) + assert.Contains(t, w.Body.String(), tt.wantBody, route.name) + testutil.AssertJSONErrorResponse(t, w) + continue + } + require.Equal(t, route.ok, w.Code, "%s: %s", route.name, w.Body.String()) + want, ok := tenants.For(tt.wantTenant) + require.True(t, ok) + require.NotEmpty(t, handed, route.name) + for _, got := range handed { + assert.Same(t, want, got, route.name) + } + } + }) + } +} + +// The ClickHouse-side getters — the connection, the registry, the HTTP +// target and the query deadline — receive the request's own tenant store +// through the real router, on the routes that reach ClickHouse: two tenants +// alternating never hand one the other's. +func TestNewRouter_ClickHouseGettersReceiveTheRequestTenantsStore(t *testing.T) { + tenants := nestedTenants(t, map[string]string{"acme": fullConfig(100), "globex": fullConfig(200)}) + var handed []*settings.Store + record := func(s *settings.Store) { handed = append(handed, s) } + reg := testRegistry(t) + viewer := staticPolicy(&policy.Policy{ + DefaultRole: "viewer", + Tables: map[string]policy.TablePolicy{"clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}}, + }) + conn := func(s *settings.Store) driver.Conn { record(s); return &countingConn{} } + registry := func(s *settings.Store) *discovery.SchemaRegistry { record(s); return reg } + timeout := func(s *settings.Store) time.Duration { record(s); return time.Second } + router := NewRouter(Dependencies{ + Tenants: tenants, + Ingest: NewIngestHandler(registry, &testutil.MockPublisher{}), + StructuredQuery: NewStructuredQueryHandler(conn, nil, registry, viewer, func(*settings.Store) int { return 60 }, timeout, nil), + Pipes: NewPipesHandler(staticPipes(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1", AllowedRoles: []string{"viewer"}}), viewer, conn, nil, timeout), + Query: &QueryHandler{}, + SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), + Health: &HealthHandler{}, + Version: NewVersionHandler("test", "test", "test"), + Schema: schemaHandlerOver(reg, tenants), + AuthMW: func(next http.Handler) http.Handler { return next }, + PolicySource: policy.Static(&policy.Policy{}), + }) + + routes := []struct { + name, path, body string + getters int // store-keyed ClickHouse getters the route consults + }{ + {name: "structured query", path: "/v1/query?table=clicks", body: `{"select_all": true}`, getters: 3}, + {name: "pipe execute", path: "/v1/pipes/top_pages", body: `{}`, getters: 2}, + } + for _, route := range routes { + for _, id := range []tenant.ID{"acme", "globex", "acme"} { + t.Run(route.name+" as "+id.String(), func(t *testing.T) { + want, ok := tenants.For(id) + require.True(t, ok) + handed = nil + req := httptest.NewRequestWithContext(auth.WithRole(context.Background(), "viewer"), http.MethodPost, route.path, strings.NewReader(route.body)) + req.Header.Set("Content-Type", "application/json") + req.Header.Set(tenant.Header, id.String()) + w := httptest.NewRecorder() + router.ServeHTTP(w, req) + + require.Equal(t, http.StatusOK, w.Code, "body: %s", w.Body.String()) + require.Len(t, handed, route.getters) + for _, got := range handed { + assert.Same(t, want, got) + } + }) + } + } +} diff --git a/internal/api/tenant_helpers_test.go b/internal/api/tenant_helpers_test.go index b10df720..d48cd2e5 100644 --- a/internal/api/tenant_helpers_test.go +++ b/internal/api/tenant_helpers_test.go @@ -7,9 +7,11 @@ import ( "path/filepath" "testing" + "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/dedupe" + "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/pipes" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" @@ -50,6 +52,24 @@ func nestedTenants(t *testing.T, configs map[string]string) *settings.Registry { return tenants } +// fixedRegistry is a RegistrySource fixed to reg, whatever the tenant. +func fixedRegistry(reg *discovery.SchemaRegistry) RegistrySource { + return func(*settings.Store) *discovery.SchemaRegistry { return reg } +} + +// schemaHandlerOver is a SchemaHandler serving reg for every tenant of +// tenants, which the ops reads resolve ?tenant= against. +func schemaHandlerOver(reg *discovery.SchemaRegistry, tenants *settings.Registry) *SchemaHandler { + h := NewSchemaHandler(fixedRegistry(reg)) + h.Tenants = tenants + return h +} + +// fixedConn is a connection source fixed to conn, whatever the tenant. +func fixedConn(conn driver.Conn) func(*settings.Store) driver.Conn { + return func(*settings.Store) driver.Conn { return conn } +} + // staticPolicy is a PolicySource fixed to p, whatever the tenant. func staticPolicy(p *policy.Policy) PolicySource { return func(*settings.Store) *policy.Policy { return p } diff --git a/internal/api/tenant_test.go b/internal/api/tenant_test.go index 13414041..7dabce29 100644 --- a/internal/api/tenant_test.go +++ b/internal/api/tenant_test.go @@ -132,18 +132,18 @@ func TestNewRouter_HandlersReceiveTheRequestTenantsStore(t *testing.T) { return pipes.Static(&pipes.NamedQuery{Name: "top_pages", SQL: "SELECT 1"}) } reg := testRegistry(t) - ingest := NewIngestHandler(reg, &testutil.MockPublisher{}) + ingest := NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}) ingest.PolicySource = recordPolicy router := NewRouter(Dependencies{ Tenants: tenants, Ingest: ingest, - StructuredQuery: NewStructuredQueryHandler(nil, nil, reg, recordPolicy, func(*settings.Store) int { return 60 }, noTimeout, nil), + StructuredQuery: NewStructuredQueryHandler(nil, nil, fixedRegistry(reg), recordPolicy, func(*settings.Store) int { return 60 }, noTimeout, nil), Pipes: NewPipesHandler(recordPipes, recordPolicy, nil, nil, noTimeout), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, Version: NewVersionHandler("test", "test", "test"), - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, tenants), AuthMW: func(next http.Handler) http.Handler { return next }, PolicySource: policy.Static(&policy.Policy{}), }) @@ -191,7 +191,7 @@ func TestTenantRouteHandlers_NoResolvedTenantIs500(t *testing.T) { t.Parallel() reg := testRegistry(t) handlers := map[string]http.HandlerFunc{ - "ingest": NewIngestHandler(reg, &testutil.MockPublisher{}).Handle, + "ingest": NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}).Handle, "structured query": newStructuredQueryHandler(t).Handle, "pipe execute": NewPipesHandler(staticPipes(), nil, nil, nil, noTimeout).Execute, "stream": NewStreamHandler(stream.NewHub(nil, nil, nil), nil).Handle, @@ -213,12 +213,12 @@ func tenantProbeRouter(t *testing.T, sawStore *[]bool) http.Handler { reg := testRegistry(t) return NewRouter(Dependencies{ Tenants: testTenants(), - Ingest: NewIngestHandler(reg, &testutil.MockPublisher{}), + Ingest: NewIngestHandler(fixedRegistry(reg), &testutil.MockPublisher{}), Query: &QueryHandler{}, SSE: NewStreamHandler(stream.NewHub(nil, nil, nil), nil), Health: &HealthHandler{}, Version: NewVersionHandler("test", "test", "test"), - Schema: NewSchemaHandler(reg), + Schema: schemaHandlerOver(reg, testTenants()), AuthMW: func(next http.Handler) http.Handler { return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { _, ok := StoreFromContext(r.Context()) diff --git a/internal/app/app.go b/internal/app/app.go index 68432fb3..33bf7dfc 100644 --- a/internal/app/app.go +++ b/internal/app/app.go @@ -46,6 +46,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" "github.com/Wave-RF/WaveHouse/internal/stream" + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // BuildInfo is the ldflags-stamped identity of the binary, served by @@ -89,18 +90,21 @@ type App struct { listener net.Listener // tenants is the registry every tenant-aware path resolves through, and - // the owner of every reload. The process-wide resources (ClickHouse, - // MQ) still follow its default tenant, through defaultStore: tenant - // 0's store as of its last adoption (defaultSetting). + // the owner of every reload. The one process-wide resource left, the MQ, + // still follows its default tenant, through defaultStore: tenant 0's + // store as of its last adoption (defaultSetting). tenants *settings.Registry defaultStore atomic.Pointer[settings.Store] // policies is the default tenant's policy, for the ops gate of a flat // directory. policies policy.Source promHandler http.Handler - ch *chconn.Manager + // pools is one ClickHouse pool per tuple the served tenants name, and + // discoveries one schema registry per served tenant, each resolved per + // call by the tenant. + pools *chconn.Pools bootState *api.BootState - registry *discovery.SchemaRegistry + discoveries *discoveries // dedup is one store per tenant, each following its own folder's switch. dedup *dedupe.Stores mq mq.Broker @@ -293,9 +297,10 @@ func closeWithin(ctx context.Context, name string, release func(context.Context) // Handler is the API router, for a harness that serves it itself. func (a *App) Handler() http.Handler { return a.handler } -// Registry is the schema registry, for a harness that refreshes it after -// creating tables. -func (a *App) Registry() *discovery.SchemaRegistry { return a.registry } +// Registry is the default tenant's schema registry, for a harness that +// refreshes it after creating tables; nil over a nested directory serving +// no tenant 0. +func (a *App) Registry() *discovery.SchemaRegistry { return a.discoveries.For(tenant.Default) } // MQ is the broker, for a harness that publishes straight onto the ingest // queue. diff --git a/internal/app/app_test.go b/internal/app/app_test.go index fdae48ea..ad1d503f 100644 --- a/internal/app/app_test.go +++ b/internal/app/app_test.go @@ -626,34 +626,68 @@ func TestNew_DedupeOpenFailure(t *testing.T) { }) } -// Until story 6 every tenant reads the same ClickHouse tables, so an insert -// invalidates a table's cached results under every tenant the registry -// knows, not only under the batch's tenant: the cache the worker is handed fans -// the namespaces out. A rejected tenant is included — it comes back into -// service with the entries it has. -func TestSharedTables_InvalidatesEveryKnownTenant(t *testing.T) { +// The tenants on the writer's ClickHouse address and database read the same +// tables, so an insert invalidates a table's cached results under every one +// of them — whatever their user, so across pools — and under no tenant on +// another address or database: the cache the worker is handed fans the +// namespaces out by the pools' sharing rule. The writer's own tenant is +// bumped even when it is on no pool. +func TestSharedTables_InvalidatesTheTenantsSharingTheTables(t *testing.T) { t.Parallel() - tenants, findings := settings.Open(writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil, "broken": invalidQuery})) - require.NotNil(t, tenants, "findings: %v", findings) + sharing := map[tenant.ID][]tenant.ID{tenant.Default: {tenant.Default, "acme", "globex"}, "initech": {"initech"}} mock := &testutil.MockCache{} - c := sharedTables{Cache: mock, tenants: tenants} + c := sharedTables{Cache: mock, sharing: func(id tenant.ID) []tenant.ID { return sharing[id] }} n, err := c.Invalidate(t.Context(), []cache.Namespace{ {Tenant: tenant.Default, Table: "events"}, {Tenant: tenant.Default, Table: "events", Scope: "org_1"}, }) require.NoError(t, err) - assert.Equal(t, uint64(8), n) + assert.Equal(t, uint64(6), n) assert.ElementsMatch(t, []cache.Namespace{ {Tenant: tenant.Default, Table: "events"}, {Tenant: tenant.Default, Table: "events", Scope: "org_1"}, {Tenant: "acme", Table: "events"}, {Tenant: "acme", Table: "events", Scope: "org_1"}, - {Tenant: "broken", Table: "events"}, - {Tenant: "broken", Table: "events", Scope: "org_1"}, {Tenant: "globex", Table: "events"}, {Tenant: "globex", Table: "events", Scope: "org_1"}, - }, mock.GetNamespaces(), "the batch's tenant and every known one, the rejected one included") + }, mock.GetNamespaces(), "the batch's tenant and the ones sharing its tables; initech reads another database") + + mock = &testutil.MockCache{} + c = sharedTables{Cache: mock, sharing: func(tenant.ID) []tenant.ID { return nil }} + _, err = c.Invalidate(t.Context(), []cache.Namespace{{Tenant: "orphan", Table: "events"}}) + require.NoError(t, err) + assert.Equal(t, []cache.Namespace{{Tenant: "orphan", Table: "events"}}, mock.GetNamespaces(), "a writer on no pool still bumps its own") +} + +// A tenant back on a pool after an absence — its folder rejected, then +// repaired; removed, then restored — was out of the fan-out while away, so +// the wiring orphans its table-keyed cache as it comes back; a tenant that stayed +// is never touched, and a reload that changes nothing bumps nobody. +func TestReload_ReadmittedTenantCacheIsOrphaned(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) + a := newApp(t, testConfig(t, root), Options{}) + // The hooks read a.cache at reload time: a recording cache from here on. + mock := &testutil.MockCache{} + a.cache = mock + + _, adopted := a.tenants.Reload("test") + require.True(t, adopted) + assert.Empty(t, mock.GetTenants(), "nothing readmitted, nothing orphaned") + + rewriteSettings(t, filepath.Join(root, "globex"), invalidQuery) + a.tenants.Reload("test") + assert.Empty(t, mock.GetTenants(), "a rejection releases; it orphans nothing yet") + rewriteSettings(t, filepath.Join(root, "globex"), nil) + _, adopted = a.tenants.Reload("test") + require.True(t, adopted) + assert.Equal(t, []tenant.ID{"globex"}, mock.GetTenants(), "repaired: back on a pool, its cache orphaned") + + require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) + a.tenants.Reload("test") + require.NoError(t, os.Rename(writeSettings(t, nil), filepath.Join(root, "acme"))) + a.tenants.Reload("test") + assert.Equal(t, []tenant.ID{"globex", "acme"}, mock.GetTenants(), "restored: the same") } // keepalive is a config.json patch setting the stream block's keepalive pair. @@ -1196,18 +1230,248 @@ func TestReload_PoolAboveTheCeilingKeepsTheConnection(t *testing.T) { cfg := testConfig(t, dir) cfg.ClickHouse.MaxTotalConns = 10 a := newApp(t, cfg, Options{}) - require.Equal(t, boot, a.ch.Addr()) + require.Equal(t, boot, a.pools.For(tenant.Default).Identity().Addr) logs := logtest.Capture(t, slog.LevelError) - rewriteSettings(t, dir, poolSettings(moved, 20)) + refused := poolSettings(moved, 20) + refused["clickhouse"].(map[string]any)["database"] = "moved_db" + rewriteSettings(t, dir, refused) _, adopted := a.tenants.Reload("test") require.True(t, adopted) - assert.Equal(t, boot, a.ch.Addr(), "the refused reload leaves the connection as it was") - assert.Contains(t, logs.String(), "clickhouse reconfigure refused") + assert.Equal(t, boot, a.pools.For(tenant.Default).Identity().Addr, "the refused reload leaves the connection as it was") + _, database := a.discoverySource(tenant.Default)() + assert.Equal(t, a.pools.For(tenant.Default).Identity().Database, database, "discovery reads the kept pool's database") + assert.NotEqual(t, "moved_db", database, "not the adopted document's") + assert.Contains(t, logs.String(), "clickhouse pools reconciled in part") assert.Contains(t, logs.String(), "clickhouse.max_open_conns 20") + assert.Contains(t, logs.String(), "tenant 0 keeps its previous pool") rewriteSettings(t, dir, poolSettings(moved, 10)) _, adopted = a.tenants.Reload("test") require.True(t, adopted) - assert.Equal(t, moved, a.ch.Addr(), "the next reload that fits applies") + assert.Equal(t, moved, a.pools.For(tenant.Default).Identity().Addr, "the next reload that fits applies") +} + +// chSettings is a config.json patch: the seed's clickhouse block pointed at +// addr as user, with the native pool sized to open. +func chSettings(addr, user string, open int) map[string]any { + p := poolSettings(addr, open) + p["clickhouse"].(map[string]any)["username"] = user + return p +} + +// A nested directory gets one pool per tuple among its tenants (#583 story +// 6): tenants naming the same address, database, user and tls share one +// Manager, sized to their largest ask; a tenant naming another gets its own. +// A reload that changes one sharer's username moves that tenant to a pool of +// its own and leaves the other on the very same Manager, resized to its own +// ask — the worked example of the story. +func TestNew_NestedPoolsFollowEachTenantsTuple(t *testing.T) { + shared, other := closedAddr(t), closedAddr(t) + root := writeNestedSettings(t, map[string]map[string]any{ + "acme": chSettings(shared, "default", 10), + "globex": chSettings(shared, "default", 20), + "initech": chSettings(other, "default", 10), + }) + a := newApp(t, testConfig(t, root), Options{}) + + acme, globex, initech := a.pools.For("acme"), a.pools.For("globex"), a.pools.For("initech") + require.NotNil(t, acme) + assert.Same(t, acme, globex, "one tuple, one pool") + assert.NotSame(t, acme, initech) + assert.Equal(t, 20, acme.Sizes().MaxOpenConns, "the largest ask among the sharers") + assert.Equal(t, []tenant.ID{"acme", "globex"}, a.pools.SharingTables("acme")) + assert.NotNil(t, a.discoveries.For("acme")) + assert.NotNil(t, a.discoveries.For("initech")) + assert.NotSame(t, a.discoveries.For("acme"), a.discoveries.For("globex"), "one registry per tenant, shared pool or not") + + rewriteSettings(t, filepath.Join(root, "globex"), chSettings(shared, "reporting", 20)) + _, adopted := a.tenants.Reload("test") + require.True(t, adopted) + assert.Same(t, acme, a.pools.For("acme"), "acme keeps its Manager") + assert.NotSame(t, acme, a.pools.For("globex"), "globex moved to a pool of its own") + assert.Equal(t, "reporting", a.pools.For("globex").Identity().Username) + assert.Equal(t, 10, acme.Sizes().MaxOpenConns, "acme's pool shrank to acme's ask") + assert.Same(t, initech, a.pools.For("initech")) + assert.Equal(t, []tenant.ID{"acme", "globex"}, a.pools.SharingTables("acme"), "same address and database: still the same tables") +} + +// A nested directory's pools must fit the ceiling together: boot is refused +// naming the sum and the ceiling, like a flat directory's one pool. +func TestNew_NestedRefusesPoolsAboveTheCeiling(t *testing.T) { + guardGlobals(t) + root := writeNestedSettings(t, map[string]map[string]any{ + "acme": chSettings(closedAddr(t), "default", 10), + "globex": chSettings(closedAddr(t), "default", 10), + }) + cfg := testConfig(t, root) + cfg.ClickHouse.MaxTotalConns = 15 + _, err := New(t.Context(), Options{Config: cfg}) + require.ErrorContains(t, err, "clickhouse.max_open_conns 10") + require.ErrorContains(t, err, "at 20, above clickhouse.max_total_conns 15") +} + +// A reload whose new tenant's pool would put the pools over the ceiling +// leaves that tenant on no pool — it fails closed, its schema never +// discovered — and the open pools untouched; the next reload that frees the +// budget opens it. +func TestReload_CeilingRefusesAThirdTupleThenOpensIt(t *testing.T) { + a1, a2, a3 := closedAddr(t), closedAddr(t), closedAddr(t) + root := writeNestedSettings(t, map[string]map[string]any{ + "acme": chSettings(a1, "default", 10), + "globex": chSettings(a2, "default", 10), + }) + cfg := testConfig(t, root) + cfg.ClickHouse.MaxTotalConns = 25 + a := newApp(t, cfg, Options{}) + acme, globex := a.pools.For("acme"), a.pools.For("globex") + + logs := logtest.Capture(t, slog.LevelError) + require.NoError(t, os.Rename(writeSettings(t, chSettings(a3, "default", 10)), filepath.Join(root, "initech"))) + _, adopted := a.tenants.Reload("test") + require.True(t, adopted) + assert.Nil(t, a.pools.For("initech"), "not opened") + assert.Contains(t, logs.String(), "clickhouse pools reconciled in part") + assert.Contains(t, logs.String(), "not opened for tenant initech") + assert.Contains(t, logs.String(), "at 30, above clickhouse.max_total_conns 25") + assert.Same(t, acme, a.pools.For("acme")) + assert.Same(t, globex, a.pools.For("globex")) + + // The tenant is served — its settings are fine — but fails closed on + // its ClickHouse side: no pool, so no discovery, so no table is known. + req := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ingest?table=clicks", strings.NewReader(`{"page": "/"}`)) + req.Header.Set("Content-Type", "application/json") + req.Header.Set(tenant.Header, "initech") + rec := httptest.NewRecorder() + a.Handler().ServeHTTP(rec, req) + assert.Equal(t, http.StatusServiceUnavailable, rec.Code, "body: %s", rec.Body.String()) + assert.Equal(t, "5", rec.Header().Get("Retry-After")) + + rewriteSettings(t, filepath.Join(root, "acme"), chSettings(a1, "default", 5)) + _, adopted = a.tenants.Reload("test") + require.True(t, adopted) + require.NotNil(t, a.pools.For("initech"), "opened once a sharer made room") + assert.Same(t, acme, a.pools.For("acme")) + assert.Equal(t, 5, acme.Sizes().MaxOpenConns) +} + +// A tenant the registry stops serving — its folder rejected, then removed — +// releases its pool and its schema registry; the tenant beside it keeps +// both; restoring the folder restores both. +func TestReload_TenantGoneReleasesItsPoolAndRegistry(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) + a := newApp(t, testConfig(t, root), Options{}) + acme, acmeRegistry := a.pools.For("acme"), a.discoveries.For("acme") + require.NotNil(t, acme) + require.NotNil(t, acmeRegistry) + require.NotNil(t, a.pools.For("globex")) + + rewriteSettings(t, filepath.Join(root, "globex"), invalidQuery) + _, adopted, known := a.tenants.ReloadTenant("globex", "test") + require.True(t, known) + require.False(t, adopted) + assert.Nil(t, a.pools.For("globex"), "a rejected tenant is on no pool") + assert.Nil(t, a.discoveries.For("globex"), "and has no registry") + assert.Same(t, acme, a.pools.For("acme")) + assert.Same(t, acmeRegistry, a.discoveries.For("acme")) + + // Adopted in part from here on: globex's folder stays rejected. + require.NoError(t, os.RemoveAll(filepath.Join(root, "acme"))) + a.tenants.Reload("test") + assert.Nil(t, a.pools.For("acme")) + assert.Nil(t, a.discoveries.For("acme")) + + require.NoError(t, os.Rename(writeSettings(t, nil), filepath.Join(root, "acme"))) + a.tenants.Reload("test") + assert.NotNil(t, a.pools.For("acme")) + assert.NotNil(t, a.discoveries.For("acme"), "back, over a fresh registry") + assert.NotSame(t, acmeRegistry, a.discoveries.For("acme")) +} + +// Over a nested directory the probes read every tenant together: /livez is +// degraded while no tenant has completed a first discovery, names the tenant +// in its diagnostic, and turns 200 for good at the first success, whatever +// another tenant's discovery does afterwards; /readyz then pings every open +// pool and names each one that does not answer. +func TestNew_NestedProbesFollowTheFirstTenantToLoad(t *testing.T) { + root := writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil}) + a := newApp(t, testConfig(t, root), Options{}) + + rec := get(t, a.Handler(), "/livez") + require.Equal(t, http.StatusServiceUnavailable, rec.Code) + assert.Contains(t, rec.Body.String(), "schema discovery") + // The loops' first attempts fail at once against closed ports and + // name their tenant; boot itself starts with the no-tenant diagnostic. + assert.Eventually(t, func() bool { + body := get(t, a.Handler(), "/livez").Body.String() + return strings.Contains(body, "tenant acme") || strings.Contains(body, "tenant globex") + }, 5*time.Second, 10*time.Millisecond) + assert.Equal(t, http.StatusServiceUnavailable, get(t, a.Handler(), "/readyz").Code, "not ready while degraded, before any ping") + + // The first success anywhere, as the loops report it. + a.discoveries.onLoaded("acme") + assert.Equal(t, http.StatusOK, get(t, a.Handler(), "/livez").Code) + a.discoveries.onAttempt("globex", errors.New("connection refused")) + assert.Equal(t, http.StatusOK, get(t, a.Handler(), "/livez").Code, "sticky: another tenant's outage is not a probe failure") + online := httptest.NewRequestWithContext(t.Context(), http.MethodGet, "/v1/health", nil) + online.Header.Set(tenant.Header, "globex") + rec = httptest.NewRecorder() + a.Handler().ServeHTTP(rec, online) + assert.Equal(t, http.StatusOK, rec.Code, "the SDK ping mirrors /livez, for the tenant still failing too") + + rec = get(t, a.Handler(), "/readyz") + assert.Equal(t, http.StatusServiceUnavailable, rec.Code) + for _, id := range []tenant.ID{"acme", "globex"} { + assert.Contains(t, rec.Body.String(), a.pools.For(id).Identity().Addr, "every pool that did not answer is named") + } +} + +// Before a tenant's first discovery a table lookup is a 503 with +// Retry-After, not a 404: in a flat directory during the degraded boot, and +// in a nested one per tenant. +func TestNew_SchemaNotLoadedIs503(t *testing.T) { + ingest := func(t *testing.T, a *App, id string) *httptest.ResponseRecorder { + t.Helper() + req := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/v1/ingest?table=clicks", strings.NewReader(`{"page": "/"}`)) + req.Header.Set("Content-Type", "application/json") + if id != "" { + req.Header.Set(tenant.Header, id) + } + rec := httptest.NewRecorder() + a.Handler().ServeHTTP(rec, req) + return rec + } + t.Run("flat, degraded boot", func(t *testing.T) { + a := newApp(t, testConfig(t, writeSettings(t, nil)), Options{}) + rec := ingest(t, a, "") + assert.Equal(t, http.StatusServiceUnavailable, rec.Code, "body: %s", rec.Body.String()) + assert.Equal(t, "5", rec.Header().Get("Retry-After")) + assert.Contains(t, rec.Body.String(), "schema not loaded yet") + }) + t.Run("nested, per tenant", func(t *testing.T) { + a := newApp(t, testConfig(t, writeNestedSettings(t, map[string]map[string]any{"acme": nil})), Options{}) + rec := ingest(t, a, "acme") + assert.Equal(t, http.StatusServiceUnavailable, rec.Code, "body: %s", rec.Body.String()) + assert.Equal(t, "5", rec.Header().Get("Retry-After")) + rec = get(t, a.Handler(), "/v1/ops/schema?tenant=acme") + assert.Equal(t, http.StatusForbidden, rec.Code, "the ops tree keeps its gate") + }) +} + +// Close stops every tenant's discovery loop within the release budget, and +// the pools after them. +func TestClose_StopsTheDiscoveryLoops(t *testing.T) { + a := newApp(t, testConfig(t, writeNestedSettings(t, map[string]map[string]any{"acme": nil, "globex": nil})), Options{}) + loops := *a.discoveries.cur.Load() + require.Len(t, loops, 2) + require.NoError(t, a.Close(context.Background())) + for id, td := range loops { + select { + case <-td.done: + case <-time.After(5 * time.Second): + t.Fatalf("the loop of tenant %s did not stop", id) + } + } + assert.Nil(t, a.discoveries.For("acme")) + assert.Nil(t, a.pools.For("acme")) } diff --git a/internal/app/discoveries.go b/internal/app/discoveries.go new file mode 100644 index 00000000..60fbee41 --- /dev/null +++ b/internal/app/discoveries.go @@ -0,0 +1,132 @@ +package app + +import ( + "context" + "fmt" + "maps" + "slices" + "sync" + "sync/atomic" + "time" + + "github.com/Wave-RF/WaveHouse/internal/discovery" + "github.com/Wave-RF/WaveHouse/internal/settings" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// discoveries is one schema registry per served tenant (#583 story 6), each +// kept fresh by a loop of its own — RetryRefresh until its first success, +// then StartAutoRefresh at the tenant's cadence, first tick at a random +// offset so tenants adopted together do not refresh together. Reconciled +// from the settings registry's AfterAdopt hook, after the pools: a newly +// served tenant gets a registry over the pool it is on and a loop; a tenant +// no longer served — rejected or removed — has its loop stopped and its +// registry dropped, and starts over when it is back. Every loop stops under +// App.Close within the release budget. A lookup is one lock-free load. +type discoveries struct { + // ctx is the loops' parent: the App's stop context. + ctx context.Context + // build makes a tenant's registry over the pool it is on. + build func(tenant.ID, *settings.Store) *discovery.SchemaRegistry + // onAttempt reports a loop's failed attempt before its first success, + // and onLoaded the first success: what /livez is driven by. + onAttempt func(tenant.ID, error) + onLoaded func(tenant.ID) + + mu sync.Mutex // serializes reconcile, adopt and close + cur atomic.Pointer[map[tenant.ID]*tenantDiscovery] +} + +// tenantDiscovery is one tenant's registry and its loop. +type tenantDiscovery struct { + registry *discovery.SchemaRegistry + cancel context.CancelFunc + done chan struct{} +} + +func newDiscoveries(ctx context.Context, build func(tenant.ID, *settings.Store) *discovery.SchemaRegistry, onAttempt func(tenant.ID, error), onLoaded func(tenant.ID)) *discoveries { + d := &discoveries{ctx: ctx, build: build, onAttempt: onAttempt, onLoaded: onLoaded} + d.cur.Store(&map[tenant.ID]*tenantDiscovery{}) + return d +} + +// For returns tenant id's registry, or nil when it has none: it is not +// served, or its registry is not built yet. +func (d *discoveries) For(id tenant.ID) *discovery.SchemaRegistry { + if td, ok := (*d.cur.Load())[id]; ok { + return td.registry + } + return nil +} + +// reconcile starts a loop, over a registry built now, for every served +// tenant without one, and stops the loop of every tenant no longer served. +func (d *discoveries) reconcile(tenants *settings.Registry) { + d.mu.Lock() + defer d.mu.Unlock() + cur := *d.cur.Load() + next := maps.Clone(cur) + served := map[tenant.ID]bool{} + for id, store := range tenants.All() { + served[id] = true + if _, ok := next[id]; !ok { + next[id] = d.start(id, d.build(id, store)) + } + } + for id, td := range cur { + if !served[id] { + td.cancel() + delete(next, id) + } + } + d.cur.Store(&next) +} + +// adopt registers reg as tenant id's registry and starts its loop: for the +// registry boot refreshed synchronously before any loop ran. +func (d *discoveries) adopt(id tenant.ID, reg *discovery.SchemaRegistry) { + d.mu.Lock() + defer d.mu.Unlock() + next := maps.Clone(*d.cur.Load()) + next[id] = d.start(id, reg) + d.cur.Store(&next) +} + +// start runs reg's loop: the boot retry until the first success, skipped +// for a registry already loaded, then the periodic refresh. +func (d *discoveries) start(id tenant.ID, reg *discovery.SchemaRegistry) *tenantDiscovery { + ctx, cancel := context.WithCancel(d.ctx) //nolint:gosec // G118: held on the tenantDiscovery, called by reconcile or close + td := &tenantDiscovery{registry: reg, cancel: cancel, done: make(chan struct{})} + go func() { + defer close(td.done) + if !reg.Loaded() { + err := reg.RetryRefresh(ctx, 2*time.Second, 60*time.Second, func(err error) { d.onAttempt(id, err) }) + if err != nil { + // ctx cancelled before success — the process is stopping, or + // the tenant is no longer served. + return + } + d.onLoaded(id) + } + reg.StartAutoRefresh(ctx) + }() + return td +} + +// close stops every loop and waits for them within ctx, the release budget. +func (d *discoveries) close(ctx context.Context) error { + d.mu.Lock() + defer d.mu.Unlock() + cur := *d.cur.Swap(&map[tenant.ID]*tenantDiscovery{}) + for _, td := range cur { + td.cancel() + } + for _, id := range slices.Sorted(maps.Keys(cur)) { + select { + case <-cur[id].done: + case <-ctx.Done(): + return fmt.Errorf("schema discovery loop of tenant %s not stopped: %w", id, ctx.Err()) + } + } + return nil +} diff --git a/internal/app/wire.go b/internal/app/wire.go index c3f45e24..313209ba 100644 --- a/internal/app/wire.go +++ b/internal/app/wire.go @@ -17,6 +17,8 @@ import ( "syscall" "time" + "github.com/ClickHouse/clickhouse-go/v2/lib/driver" + "github.com/Wave-RF/WaveHouse/internal/api" "github.com/Wave-RF/WaveHouse/internal/auth" "github.com/Wave-RF/WaveHouse/internal/cache" @@ -74,7 +76,7 @@ func (a *App) wireSettings() error { slog.Warn("no policy adopted — every token-based request is denied until policies.json defines one (fail closed)") } case !served: - slog.Warn("nested settings directory with no tenant 0 being served: the ClickHouse connection, the MQ byte budget, and the schema refresh cadence are still configured from tenant 0's config.json, so they run unconfigured — no ClickHouse address, /livez degraded — until a 0 folder is adopted") + slog.Warn("nested settings directory with no tenant 0 being served: the MQ byte budget is still configured from tenant 0's config.json, so it runs unconfigured until a 0 folder is adopted") } return nil } @@ -89,11 +91,11 @@ func (a *App) trackDefaultStore() { } } -// defaultSetting reads one setting of the default tenant, which the -// process-wide resources (ClickHouse, MQ) follow until #583 gives each -// tenant its own. It reads tenant 0's last adopted document, so a -// 0 folder a reload rejected or removed leaves every one of them as it was — -// the ones a hook reconciles and the one read per request (the ops +// defaultSetting reads one setting of the default tenant, which the one +// process-wide resource left (the MQ) follows until #583 gives each tenant +// its own. It reads tenant 0's last adopted document, so a +// 0 folder a reload rejected or removed leaves every reader as it was — +// the MQ's byte budget a hook reconciles and the one read per request (the ops // gate's admin role) alike. A nested directory that has never served a tenant // 0 reads T's zero value, which wireSettings warned about at boot. func defaultSetting[T any](a *App, get func(*settings.Store) T) T { @@ -153,8 +155,8 @@ func longestGapWindow(tenants *settings.Registry) time.Duration { // perTenant adapts a store accessor to the tenant-keyed getter the async // paths take: they hold a tenant id — the one each message's topic names -// for the stream hub and the ingest worker (#583 story 5), tenant.Default -// for the schema registry until story 6 — not a request's resolved store. A +// for the stream hub and the ingest worker (#583 story 5) — not a request's +// resolved store. A // miss — a nested directory with no 0 folder, or with a rejected or removed // one — is logged and read as T's zero value; what a removed tenant means to // each async path is story 3's to decide. @@ -185,29 +187,32 @@ func dlqFor(tenants *settings.Registry) func(tenant.ID, string) bool { } } -// sharedTables is the cache the ingest worker invalidates through until #583 -// story 6 gives each tenant its own ClickHouse. Every tenant reads the same -// tables today, so an insert into one changes what every tenant would read: -// the worker names one tenant's namespaces (the batch's), and this bumps them -// under every tenant the registry knows, the named one included. Known, not -// served: a rejected tenant keeps its cache entries and comes back into -// service with them, so leaving it out would let a folder repaired inside a -// TTL serve pre-insert rows. A tenant removed and restored inside a TTL still -// can — the registry forgets a removed tenant, and what becomes of its cache -// is story 3's — and reads are untouched: a tenant's cached results stay its -// own. Goes away with story 6, when a table is one tenant's. +// sharedTables is the cache the ingest worker invalidates through. A +// tenant's tables are the ones on its ClickHouse address and database, and +// the tenants naming the same address and database — whatever their user or +// tls block, so across pools — read the same tables: an insert into one +// changes what every one of them would read. The worker names one tenant's +// namespaces (the batch's), and this bumps them under every tenant sharing +// its tables (chconn.Pools.SharingTables), the named one included. Reads are +// untouched: a tenant's cached results stay its own. A tenant on no pool — +// rejected, removed, or one no pool could be opened for, such as by the +// connection ceiling — is out of the fan-out, and its table-keyed cache is +// orphaned when it gets one (wireClickHouse, Cache.InvalidateTenant), so a +// folder repaired or restored inside a TTL never serves pre-insert +// structured-query rows; a pipe result names no table, so no insert +// invalidates it and it stays until its TTL expires (#343). type sharedTables struct { cache.Cache - tenants *settings.Registry + sharing func(tenant.ID) []tenant.ID } func (s sharedTables) Invalidate(ctx context.Context, namespaces []cache.Namespace) (uint64, error) { ids := map[tenant.ID]bool{} for _, ns := range namespaces { ids[ns.Tenant] = true - } - for id := range s.tenants.Known() { - ids[id] = true + for _, id := range s.sharing(ns.Tenant) { + ids[id] = true + } } all := make([]cache.Namespace, 0, len(ids)*len(namespaces)) for _, id := range slices.Sorted(maps.Keys(ids)) { @@ -274,96 +279,177 @@ func (a *App) wireObservability(ctx context.Context) { } } -// wireClickHouse opens the one driver.Conn every consumer holds. The wiring -// is the settings directory's clickhouse block plus the boot-config -// password; a reload that changes it swaps the connection behind the -// manager unconditionally — the adopted settings are the authority, and -// reachability surfaces where it already does (schema discovery retries, -// /readyz, query errors). Two exceptions keep the connection it has: a -// certificate file that cannot be read or parsed, and a pool sized above the boot -// config's clickhouse.max_total_conns — capacity is sized once, per -// process, so a settings pool above it is refused at boot like the rest of -// an impossible boot config (#530) and logged on a reload, which the next -// reload retries. The HTTP-side consumers read Target/QueryTimeout per -// request. +// wireClickHouse opens the ClickHouse pools: one per distinct address, +// database, user, password and tls tuple among the served tenants' clickhouse +// blocks (with the boot-config password), shared by the tenants naming it and +// sized to their largest ask (#583 story 6). Every reload reconciles them — +// a new tuple opens (no dial), a tenant whose tuple changed is repointed, a +// tuple no tenant names is released after its grace — under the boot +// config's clickhouse.max_total_conns, the ceiling on the open pools' sizes +// together: capacity is sized once, per process, so pools above it refuse +// boot like the rest of an impossible boot config (#530), and at a reload a +// resize above it is refused with the pool kept at its size, and a tuple +// that cannot be opened — the ceiling, a certificate file that cannot be +// read, or options the driver refuses — leaves its tenants on the pool they +// had, or on none when they had none; both logged, and retried by the next +// reload. Reachability surfaces +// where it already does (schema discovery retries, /readyz, query errors). +// Every consumer resolves its tenant's pool per call (chConn, chTargetFor). func (a *App) wireClickHouse() error { - params := func() chconn.Params { - c := defaultSetting(a, (*settings.Store).ClickHouse) - return chconn.Params{ - Addr: c.Addr, HTTPPort: c.HTTPPort, HTTPScheme: c.HTTPScheme, - Database: c.Database, Username: c.Username, Password: a.cfg.ClickHouse.Password, - QueryTimeout: c.QueryTimeout, - TLS: chconn.TLS(c.TLS), - Headers: c.Headers, - MaxOpenConns: c.MaxOpenConns, MaxIdleConns: c.MaxIdleConns, - } - } - ceiling := a.cfg.ClickHouse.MaxTotalConns - withinCeiling := func(p chconn.Params) error { - if ceiling > 0 && p.MaxOpenConns > ceiling { - return fmt.Errorf("clickhouse.max_open_conns %d (settings) exceeds clickhouse.max_total_conns %d (boot config)", p.MaxOpenConns, ceiling) + members := func() []chconn.Member { + var ms []chconn.Member + for id, store := range a.tenants.All() { + c := store.ClickHouse() + ms = append(ms, chconn.Member{Tenant: id, Params: chconn.Params{ + Addr: c.Addr, HTTPPort: c.HTTPPort, HTTPScheme: c.HTTPScheme, + Database: c.Database, Username: c.Username, Password: a.cfg.ClickHouse.Password, + QueryTimeout: c.QueryTimeout, + TLS: chconn.TLS(c.TLS), + Headers: c.Headers, + MaxOpenConns: c.MaxOpenConns, MaxIdleConns: c.MaxIdleConns, + }}) } - return nil + return ms } - p := params() - if err := withinCeiling(p); err != nil { - return err - } - ch, err := chconn.Open(p) + pools, err := chconn.NewPools(a.cfg.ClickHouse.MaxTotalConns, members()) if err != nil { return fmt.Errorf("clickhouse open: %w", err) } - a.ch = ch - a.add(component{name: "clickhouse", close: withoutContext(ch.Close)}) - a.onDefaultAdopt(func() { - p := params() - if err := withinCeiling(p); err != nil { - slog.Error("clickhouse reconfigure refused; the connection is unchanged", "error", err) - return + a.pools = pools + a.add(component{name: "clickhouse", close: withoutContext(pools.Close)}) + a.tenants.AfterAdopt(func([]tenant.ID) { + stale, err := pools.Reconcile(members()) + if err != nil { + slog.Error("clickhouse pools reconciled in part; the next reload retries", "error", err) } - if err := ch.Reconfigure(p); err != nil { - slog.Error("clickhouse reconfigure", "error", err) + // A tenant back on a pool after an absence was out of the cache + // fan-out (sharedTables) while away, and one moved to another + // address or database now reads other tables: either way what it + // cached is stale, so all of it is orphaned at once. + for _, id := range stale { + if err := a.cache.InvalidateTenant(a.stopCtx, id); err != nil { + slog.Error("cache invalidation of a stale tenant failed; it may serve stale rows until they expire", "tenant", id, "error", err) + } } }) return nil } -// wireDiscovery runs the boot-time schema discovery — non-fatal. If the -// first Refresh fails (ClickHouse unreachable, database missing, etc.) the -// binary is marked degraded via bootState (which /livez surfaces as 503 + -// diagnostic) and Run retries in the background with exponential backoff. -// The process still binds its port so operators can `curl /livez` instead -// of grepping a restart-loop log. Once a Refresh succeeds, bootState flips -// to nil and /livez returns 200. The periodic auto-refresh starts only after -// the first successful Refresh (boot or retry) so it never races -// RetryRefresh on Refresh calls or on bootState writes. +// chConn is the connection of tenant id, or an untyped nil when the tenant +// is on no pool — never a nil *Manager inside a non-nil driver.Conn, which +// would pass a nil check and panic on use. +func (a *App) chConn(id tenant.ID) driver.Conn { + m := a.pools.For(id) + if m == nil { + return nil + } + return m +} + +// discoverySource is what tenant id's schema registry discovers from, read +// per refresh so a reload that repoints the tenant or moves its database +// applies to the next one: its pool's connection and the database that pool +// was opened for — never the adopted document's, which a refused move would +// pair with the pool the tenant kept, discovering a database its queries and +// inserts do not use. +func (a *App) discoverySource(id tenant.ID) discovery.Source { + return func() (driver.Conn, string) { + m := a.pools.For(id) + if m == nil { + return nil, "" + } + return m, m.Identity().Database + } +} + +// The store-keyed getters the handlers take: each resolves the request +// tenant's pool or registry per call, so a reload that repoints the tenant +// applies to the next request. + +func (a *App) chConnFor(s *settings.Store) driver.Conn { return a.chConn(s.Tenant()) } + +func (a *App) chTargetFor(s *settings.Store) chconn.Target { return a.pools.Target(s.Tenant()) } + +func (a *App) registryFor(s *settings.Store) *discovery.SchemaRegistry { + return a.discoveries.For(s.Tenant()) +} + +// queryTimeout is the tenant's read deadline, a per-call setting rather +// than a property of the pool it shares. +func queryTimeout(s *settings.Store) time.Duration { return s.ClickHouse().QueryTimeout } + +// wireDiscovery builds one schema registry per served tenant, each with a +// refresh loop of its own (discoveries), and the boot state /livez reports: +// 503 with the latest discovery failure while no tenant has completed a +// first discovery, then 200 for the rest of the process lifetime — with one +// tenant, the rule there always was. Non-fatal either way. A flat +// directory's tenant 0 is refreshed synchronously here, as before, so the +// port binds with the state known; a failure marks the binary degraded and +// leaves the retry (backoff 2s → 60s) to its loop. A nested directory's +// tenants refresh in their loops from the start, so boot never waits on a +// tenant's ClickHouse, and a nested directory serving no tenant stays +// degraded until a reload adopts one that loads. The process still binds its +// port so operators can `curl /livez` instead of grepping a restart-loop +// log; once a tenant has loaded, another tenant's outage is that tenant's +// log line and counter, never a probe failure. func (a *App) wireDiscovery(ctx context.Context) { a.bootState = api.NewBootState(nil) - // Both sources are read per refresh, so a settings reload retunes the - // cadence and a ClickHouse reconfigure moves the database without a restart. - registry := discovery.NewSchemaRegistry(a.ch, a.ch.Database, tenant.Default, perTenant(a.tenants, (*settings.Store).SchemaRefreshInterval)) - a.registry = registry - bootErr := registry.Refresh(ctx) - if bootErr != nil { - slog.Warn("schema discovery failed on boot, retrying in background", "error", bootErr) - a.bootState.Set(fmt.Errorf("schema discovery: %w", bootErr)) - } - a.add(component{name: "schema discovery", run: func(ctx context.Context) error { - if bootErr != nil { - err := registry.RetryRefresh(ctx, 2*time.Second, 60*time.Second, func(attemptErr error) { - slog.Warn("schema discovery retry failed", "error", attemptErr) - a.bootState.Set(fmt.Errorf("schema discovery: %w", attemptErr)) - }) - if err != nil { - // ctx cancelled before success — the process is shutting down. - return nil + nested := a.tenants.Nested() + // loaded flips once, on the first tenant's first success. The check and + // the BootState write happen under one lock, so a failure reported + // while another tenant's success lands can never overwrite the cleared + // state for good and pin /livez at 503. + var ( + mu sync.Mutex + loaded bool + ) + diagnostic := func(id tenant.ID, err error) error { + if nested { + return fmt.Errorf("schema discovery: tenant %s: %w", id, err) + } + return fmt.Errorf("schema discovery: %w", err) + } + d := newDiscoveries(a.stopCtx, + func(id tenant.ID, _ *settings.Store) *discovery.SchemaRegistry { + return discovery.NewSchemaRegistry(a.discoverySource(id), id, perTenant(a.tenants, (*settings.Store).SchemaRefreshInterval)) + }, + func(id tenant.ID, err error) { + slog.Warn("schema discovery retry failed", "tenant", id, "error", err) + mu.Lock() + defer mu.Unlock() + if !loaded { + a.bootState.Set(diagnostic(id, err)) + } + }, + func(id tenant.ID) { + mu.Lock() + defer mu.Unlock() + if loaded { + slog.Info("schema discovery succeeded after retry", "tenant", id) + return } - slog.Info("schema discovery succeeded after retry, /livez now 200") + loaded = true + slog.Info("schema discovery succeeded after retry, /livez now 200", "tenant", id) a.bootState.Set(nil) + }) + a.discoveries = d + if nested { + a.bootState.Set(errors.New("schema discovery: no tenant has completed a first discovery yet")) + d.reconcile(a.tenants) + } else { + // A flat registry always serves tenant 0: Open refused boot otherwise. + store, _ := a.tenants.For(tenant.Default) + reg := d.build(tenant.Default, store) + if err := reg.Refresh(ctx); err != nil { + slog.Warn("schema discovery failed on boot, retrying in background", "error", err) + a.bootState.Set(diagnostic(tenant.Default, err)) + } else { + loaded = true // no loop has started yet } - registry.StartAutoRefresh(ctx) - return nil - }}) + d.adopt(tenant.Default, reg) + } + a.tenants.AfterAdopt(func([]tenant.ID) { d.reconcile(a.tenants) }) + a.add(component{name: "schema discovery", close: d.close}) } // legacyDedupeDir is where the one store lived before #583 story 7 gave each @@ -545,7 +631,7 @@ func (a *App) wireSweeper() { // that role's subscribers; the MQ → Hub bridge; and the keepalive wheel. func (a *App) wireStreaming() { a.sseMetrics = stream.NewMetrics() - a.hub = stream.NewHub(perTenant(a.tenants, (*settings.Store).Policy), a.registry, a.sseMetrics) + a.hub = stream.NewHub(perTenant(a.tenants, (*settings.Store).Policy), a.discoveries.For, a.sseMetrics) // Hub bridge: MQ → broadcast to connected SSE clients. The Hub decodes and // projects each event itself (skipping malformed payloads) under the @@ -585,7 +671,7 @@ func (a *App) wireStreaming() { // drain within the shutdown timeout. func (a *App) wireIngestWorker() { a.add(component{name: "ingest worker", run: func(ctx context.Context) error { - stop, failed, err := ingest.StartIngestWorker(ctx, a.mq, sharedTables{Cache: a.cache, tenants: a.tenants}, a.ch.Target, dlqFor(a.tenants)) + stop, failed, err := ingest.StartIngestWorker(ctx, a.mq, sharedTables{Cache: a.cache, sharing: a.pools.SharingTables}, a.pools.Target, dlqFor(a.tenants)) if err != nil { return err } @@ -758,12 +844,14 @@ func (a *App) wireReloadTriggers() { // prometheus.port set — the metrics sidecar. Same-port Prometheus mounts on // the API router instead. func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { - ingestHandler := api.NewIngestHandler(a.registry, a.mq) + ingestHandler := api.NewIngestHandler(a.registryFor, a.mq) ingestHandler.PolicySource = (*settings.Store).Policy ingestHandler.Dedup = func(s *settings.Store) dedupe.Deduplicator { return a.dedup.For(s.Tenant()) } ingestHandler.DedupeSettings = (*settings.Store).DedupeFor - healthHandler := api.NewHealthHandler(a.ch) + // Readiness pings every open pool at once and is ready at the first + // answer: one tenant's ClickHouse outage is not the process's. + healthHandler := api.NewHealthHandler(a.pools.Ping) healthHandler.Boot = a.bootState streamHandler := api.NewStreamHandler(a.hub, a.mq) @@ -774,22 +862,28 @@ func (a *App) wireHTTP(authMW func(http.Handler) http.Handler) { closing := make(chan struct{}) streamHandler.Closing = closing - pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.ch, a.cache, a.ch.QueryTimeout) + pipesHandler := api.NewPipesHandler(func(s *settings.Store) pipes.Source { return s }, (*settings.Store).Policy, a.chConnFor, a.cache, queryTimeout) pipesHandler.Tenants = a.tenants + schemaHandler := api.NewSchemaHandler(a.registryFor) + schemaHandler.Tenants = a.tenants + + // /v1/ops/query proxies straight to ClickHouse over HTTP — no native + // driver involvement. The HTTP target of the tenant ?tenant= names, + // resolved per request like the ingest worker's. + queryHandler := api.NewQueryHandler(a.chTargetFor, queryTimeout) + queryHandler.Tenants = a.tenants + deps := api.Dependencies{ - Ingest: ingestHandler, - // /v1/ops/query proxies straight to ClickHouse over HTTP — no native - // driver involvement. Same HTTP target as the ingest worker, resolved - // per request. - Query: api.NewQueryHandler(a.ch.Target, a.ch.QueryTimeout), + Ingest: ingestHandler, + Query: queryHandler, SSE: streamHandler, Health: healthHandler, Version: api.NewVersionHandler(a.build.Version, a.build.GitCommit, a.build.BuildTime), - Schema: api.NewSchemaHandler(a.registry), + Schema: schemaHandler, DLQ: api.NewDLQHandler(a.mq), Pipes: pipesHandler, - StructuredQuery: api.NewStructuredQueryHandler(a.ch, a.cache, a.registry, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, a.ch.QueryTimeout, (*settings.Store).DefaultMaxRows), + StructuredQuery: api.NewStructuredQueryHandler(a.chConnFor, a.cache, a.registryFor, (*settings.Store).Policy, (*settings.Store).TimestampBucketSeconds, queryTimeout, (*settings.Store).DefaultMaxRows), AuthMW: authMW, Tenants: a.tenants, diff --git a/internal/cache/cache.go b/internal/cache/cache.go index 26a0f7dd..be62bbb4 100644 --- a/internal/cache/cache.go +++ b/internal/cache/cache.go @@ -3,6 +3,8 @@ package cache import ( "context" "time" + + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // Cache provides versioned query-result storage with TTL support. @@ -27,6 +29,16 @@ type Cache interface { // another tenant keeps its versions. Returns the number of namespaces processed. Invalidate(ctx context.Context, namespaces []Namespace) (uint64, error) + // InvalidateTenant orphans every cached query of one tenant that is keyed + // by its tables in one step — every table and scope, bumped or not; a + // pipe result names no table, so neither this nor any insert + // invalidates it and it stays until its TTL expires (#343) — for a + // tenant that comes back after an absence from the invalidation fan-out + // (its settings folder rejected or removed, #583 story 6), stale by every + // insert it missed, or that moved to another ClickHouse address or + // database, whose cached results were read from other tables. + InvalidateTenant(ctx context.Context, id tenant.ID) error + // TODO: for local cache, we can just store the versions in memory, but for distributed/L2 cache, we will need to be able to either have stored procedures/pipelines etc to query them and attach them to a query, or sync them to each edge api server. // Close releases resources. diff --git a/internal/cache/local.go b/internal/cache/local.go index db3ac26d..4959242c 100644 --- a/internal/cache/local.go +++ b/internal/cache/local.go @@ -6,6 +6,8 @@ import ( "time" "github.com/dgraph-io/ristretto/v2" + + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // LocalCache is an L1 in-process cache backed by Ristretto: one pool for @@ -76,6 +78,14 @@ func (l *LocalCache) Invalidate(_ context.Context, namespaces []Namespace) (uint return uint64(len(namespaces)), nil } +// InvalidateTenant orphans every cached query of tenant id keyed by its +// tables (a pipe result names none and keeps its TTL): one version +// bump, nothing enumerated (see VersionManager.BumpTenant). +func (l *LocalCache) InvalidateTenant(_ context.Context, id tenant.ID) error { + l.versionManager.BumpTenant(id) + return nil +} + // Wait blocks until all buffered writes have been applied. // Exposed for testing; production callers rarely need this. func (l *LocalCache) Wait() { diff --git a/internal/cache/local_test.go b/internal/cache/local_test.go index 14dae75d..ed3bc14c 100644 --- a/internal/cache/local_test.go +++ b/internal/cache/local_test.go @@ -203,3 +203,39 @@ func TestLocalCache_KeyedByTenant(t *testing.T) { require.NoError(t, err) assert.Equal(t, []byte("globex rows"), val) } + +// A tenant back after an absence from the invalidation fan-out has its every +// entry orphaned at once — every table, bumped before or not — and the other +// tenants keep theirs. +func TestLocalCache_InvalidateTenant(t *testing.T) { + t.Parallel() + c, err := NewLocal(1 << 20) + require.NoError(t, err) + defer func() { _ = c.Close() }() + + ctx := context.Background() + acmeEvents := []Namespace{{Tenant: "acme", Table: "events"}} + acmeOrders := []Namespace{{Tenant: "acme", Table: "orders", Scope: "org_1"}} + globex := []Namespace{{Tenant: "globex", Table: "events"}} + require.NoError(t, c.Set(ctx, "q", acmeEvents, []byte("acme events"), 10*time.Second)) + require.NoError(t, c.Set(ctx, "q", acmeOrders, []byte("acme orders"), 10*time.Second)) + require.NoError(t, c.Set(ctx, "q", globex, []byte("globex events"), 10*time.Second)) + c.Wait() + + require.NoError(t, c.InvalidateTenant(ctx, "acme")) + for name, deps := range map[string][]Namespace{"events": acmeEvents, "orders": acmeOrders} { + val, _, err := c.Get(ctx, "q", deps) + require.NoError(t, err) + assert.Nil(t, val, "acme's %s entry is orphaned", name) + } + val, _, err := c.Get(ctx, "q", globex) + require.NoError(t, err) + assert.Equal(t, []byte("globex events"), val, "another tenant's entry stays") + + // Entries cached after the bump are served: it is a generation, not a lock. + require.NoError(t, c.Set(ctx, "q", acmeEvents, []byte("acme again"), 10*time.Second)) + c.Wait() + val, _, err = c.Get(ctx, "q", acmeEvents) + require.NoError(t, err) + assert.Equal(t, []byte("acme again"), val) +} diff --git a/internal/cache/version_manager.go b/internal/cache/version_manager.go index 73cd5509..933fa81d 100644 --- a/internal/cache/version_manager.go +++ b/internal/cache/version_manager.go @@ -15,13 +15,18 @@ import ( type VersionManager struct { mu sync.RWMutex - tableVersions map[string]uint64 // .
-> table_version - namespaceVersions map[string]uint64 // .
.. -> namespace_version + // tenantVersions leads every key of a tenant, so BumpTenant orphans the + // tenant's every namespace and query in one step — the ones no bump ever + // keyed included, which is what an enumeration of the maps would miss. + tenantVersions map[tenant.ID]uint64 // -> tenant_version + tableVersions map[string]uint64 // ..
-> table_version + namespaceVersions map[string]uint64 // ..
.. -> namespace_version } // NewVersionManager initializes the thread-safe version store. func NewVersionManager() *VersionManager { return &VersionManager{ + tenantVersions: make(map[tenant.ID]uint64), tableVersions: make(map[string]uint64), namespaceVersions: make(map[string]uint64), } @@ -36,22 +41,24 @@ type Namespace struct { Scope string } -// tableKey renders the table-versions key, ".
". A tenant id -// cannot contain a dot and callers encode the table dot-free, so the two -// tokens can never run together. -func tableKey(id tenant.ID, table string) string { - return string(id) + "." + table +// tableKeyLocked renders the table-versions key, +// "..
"; caller must hold vm.mu. A tenant id +// cannot contain a dot and callers encode the table dot-free, so the tokens +// can never run together. +func (vm *VersionManager) tableKeyLocked(id tenant.ID, table string) string { + return fmt.Sprintf("%s.%d.%s", id, vm.tenantVersions[id], table) } // namespaceKeyLocked builds the namespace-table key; caller must hold vm.mu. func (vm *VersionManager) namespaceKeyLocked(ns Namespace) string { - tk := tableKey(ns.Tenant, ns.Table) + tk := vm.tableKeyLocked(ns.Tenant, ns.Table) return fmt.Sprintf("%s.%d.%s", tk, vm.tableVersions[tk], ns.Scope) } -// NamespaceKey renders the namespace-table key for ns at its table's current -// version: ".
.." (scopeless scope is "", -// so e.g. ".
.."). +// NamespaceKey renders the namespace-table key for ns at its tenant's and +// table's current versions: +// "..
.." (scopeless +// scope is "", so e.g. ".0.
.."). func (vm *VersionManager) NamespaceKey(ns Namespace) string { vm.mu.RLock() defer vm.mu.RUnlock() @@ -87,7 +94,18 @@ func (vm *VersionManager) QueryKey(sha string, deps []Namespace) string { func (vm *VersionManager) BumpTable(id tenant.ID, table string) { vm.mu.Lock() defer vm.mu.Unlock() - vm.tableVersions[tableKey(id, table)]++ + vm.tableVersions[vm.tableKeyLocked(id, table)]++ +} + +// BumpTenant advances a tenant's version, orphaning its every namespace — +// and every cached query keyed by one — in one step (the whole-tenant +// nuke): every namespace key of the tenant carries the version, so nothing +// has to be enumerated, and a table no bump ever keyed is orphaned like the +// rest. Other tenants are untouched. +func (vm *VersionManager) BumpTenant(id tenant.ID) { + vm.mu.Lock() + defer vm.mu.Unlock() + vm.tenantVersions[id]++ } // BumpNamespace advances one (tenant, table, scope) namespace plus the table's diff --git a/internal/cache/version_manager_test.go b/internal/cache/version_manager_test.go index 56b79c24..36da8bdf 100644 --- a/internal/cache/version_manager_test.go +++ b/internal/cache/version_manager_test.go @@ -12,19 +12,28 @@ func TestVersionManager_NamespaceKey(t *testing.T) { t.Parallel() vm := NewVersionManager() - // The tenant leads, then the table at its default version (0); a scopeless - // namespace renders a trailing dot. The flat directory's tenant is "0". - assert.Equal(t, "acme.users.0.", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users"})) - assert.Equal(t, "acme.users.0.org_1", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users", Scope: "org_1"})) - assert.Equal(t, "0.users.0.", vm.NamespaceKey(Namespace{Tenant: tenant.Default, Table: "users"})) + // The tenant leads at its default version (0), then the table at its + // default version (0); a scopeless namespace renders a trailing dot. The + // flat directory's tenant is "0". + assert.Equal(t, "acme.0.users.0.", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users"})) + assert.Equal(t, "acme.0.users.0.org_1", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users", Scope: "org_1"})) + assert.Equal(t, "0.0.users.0.", vm.NamespaceKey(Namespace{Tenant: tenant.Default, Table: "users"})) // The table version is embedded in every namespace key for that tenant's // table, so a BumpTable is reflected across all its scopes at once — and // nowhere else: the same table under another tenant keeps its version. vm.BumpTable("acme", "users") - assert.Equal(t, "acme.users.1.", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users"})) - assert.Equal(t, "acme.users.1.org_1", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users", Scope: "org_1"})) - assert.Equal(t, "globex.users.0.", vm.NamespaceKey(Namespace{Tenant: "globex", Table: "users"})) + assert.Equal(t, "acme.0.users.1.", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users"})) + assert.Equal(t, "acme.0.users.1.org_1", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users", Scope: "org_1"})) + assert.Equal(t, "globex.0.users.0.", vm.NamespaceKey(Namespace{Tenant: "globex", Table: "users"})) + + // The tenant version leads every key of the tenant, so a BumpTenant moves + // every table of acme's — the never-bumped orders table included — to a + // fresh key space, at table version 0 again, and no other tenant's. + vm.BumpTenant("acme") + assert.Equal(t, "acme.1.users.0.", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "users"})) + assert.Equal(t, "acme.1.orders.0.", vm.NamespaceKey(Namespace{Tenant: "acme", Table: "orders"})) + assert.Equal(t, "globex.0.users.0.", vm.NamespaceKey(Namespace{Tenant: "globex", Table: "users"})) } func TestVersionManager_QueryKey(t *testing.T) { @@ -33,7 +42,7 @@ func TestVersionManager_QueryKey(t *testing.T) { // One dependency at default versions: sha | .
.... key := vm.QueryKey("hash123", []Namespace{{Tenant: "acme", Table: "users", Scope: "org_1"}}) - assert.Equal(t, "hash123|acme.users.0.org_1.0", key) + assert.Equal(t, "hash123|acme.0.users.0.org_1.0", key) // Dependency order must not change the key (segments are sorted). deps1 := []Namespace{{Tenant: "acme", Table: "a"}, {Tenant: "acme", Table: "b"}} @@ -89,3 +98,25 @@ func TestVersionManager_BumpNamespace(t *testing.T) { assert.Equal(t, otherBefore, vm.QueryKey("h", otherScope)) assert.Equal(t, otherTenantBefore, vm.QueryKey("h", otherTenant)) } + +// TestVersionManager_BumpTenant: a tenant's every namespace is orphaned in +// one step — a table that was never bumped (so has no key of its own to bump) +// included — and no other tenant's is touched. +func TestVersionManager_BumpTenant(t *testing.T) { + t.Parallel() + vm := NewVersionManager() + + users := []Namespace{{Tenant: "acme", Table: "users", Scope: "org_1"}} + orders := []Namespace{{Tenant: "acme", Table: "orders"}} + globexUsers := []Namespace{{Tenant: "globex", Table: "users", Scope: "org_1"}} + vm.BumpTable("acme", "users") + + usersBefore := vm.QueryKey("h", users) + ordersBefore := vm.QueryKey("h", orders) + globexBefore := vm.QueryKey("h", globexUsers) + + vm.BumpTenant("acme") + assert.NotEqual(t, usersBefore, vm.QueryKey("h", users)) + assert.NotEqual(t, ordersBefore, vm.QueryKey("h", orders), "a table no bump ever keyed is orphaned too") + assert.Equal(t, globexBefore, vm.QueryKey("h", globexUsers)) +} diff --git a/internal/chconn/chconn.go b/internal/chconn/chconn.go index a7d957fb..b53949f2 100644 --- a/internal/chconn/chconn.go +++ b/internal/chconn/chconn.go @@ -1,10 +1,13 @@ -// Package chconn owns the process's ClickHouse connection so the wiring — -// address, database, user, timeout, TLS, pool size — can follow a settings -// reload without a restart. Manager implements driver.Conn by delegating -// every call to the connection current at that instant, so consumers hold -// one driver.Conn for the process lifetime and never learn a reconnect -// happened; the HTTP-side consumers (ingest INSERTs, the raw-SQL proxy) read -// Target per request and take their client from an HTTPClients. +// Package chconn owns the process's ClickHouse connections: one pool per +// distinct address, database, user, password and tls tuple (ClickHouse +// authenticates per connection, so different credentials never share one), +// shared by the tenants whose settings name it and reconciled after every +// settings reload under the boot config's connection ceiling. Manager +// implements driver.Conn by delegating every call to the connection current +// at that instant, so a consumer resolves its tenant's Manager per call and +// never learns a resize happened; the HTTP-side consumers (ingest INSERTs, +// the raw-SQL proxy) read their tenant's Target per request and take their +// client from an HTTPClients. package chconn import ( @@ -18,13 +21,17 @@ import ( "net" "net/http" "os" + "slices" "strconv" + "strings" "sync" "sync/atomic" "time" "github.com/ClickHouse/clickhouse-go/v2" "github.com/ClickHouse/clickhouse-go/v2/lib/driver" + + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // TLS is the settings directory's clickhouse.tls block. Enabled switches @@ -73,8 +80,48 @@ func (t TLS) config() (*tls.Config, error) { return cfg, nil } -// Params is everything a connection is built from: the settings -// directory's clickhouse block plus the boot-config password. +// Identity is the tuple that decides which pool a tenant shares. Comparable +// as it stands — the TLS block is a value of strings and bools — so it is +// the map key as well. +type Identity struct { + Addr string + Database string + Username string + Password string + TLS TLS +} + +// String names the tuple for logs and errors: the address, database and +// user, never the password. +func (id Identity) String() string { return id.name() } + +// name is String for this package's own errors. It is not a String method, +// which static analysis reads as carrying every field of its receiver — the +// password included — into whatever formats it; name reads three fields and +// no other. +func (id Identity) name() string { + return id.Addr + " database " + id.Database + " user " + id.Username +} + +// Sizes is a pool's size: the driver's MaxOpenConns and MaxIdleConns. +type Sizes struct { + MaxOpenConns int + MaxIdleConns int +} + +// max is the larger ask in each dimension: what a pool shared by tenants +// asking s and o is sized to (the #597 shortest-keepalive precedent, +// inverted — the pool must hold the largest ask). Every tenant's own +// open >= idle keeps the result's open >= idle. +func (s Sizes) max(o Sizes) Sizes { + return Sizes{MaxOpenConns: max(s.MaxOpenConns, o.MaxOpenConns), MaxIdleConns: max(s.MaxIdleConns, o.MaxIdleConns)} +} + +// Params is what one tenant asks of its connection: the settings +// directory's clickhouse block plus the boot-config password. The Identity +// picks the pool; the rest is the tenant's own — the HTTP-interface wiring +// and headers, the query deadline, and the pool sizes the shared pool is at +// least as large as. type Params struct { Addr string HTTPPort int @@ -90,14 +137,14 @@ type Params struct { MaxIdleConns int } -// equal reports whether p and q describe the same connection. Field by -// field, since the headers map keeps Params from being comparable; -// TestParams_EqualCoversEveryField keeps the list complete. -func (p Params) equal(q Params) bool { - return p.Addr == q.Addr && p.HTTPPort == q.HTTPPort && p.HTTPScheme == q.HTTPScheme && - p.Database == q.Database && p.Username == q.Username && p.Password == q.Password && - p.QueryTimeout == q.QueryTimeout && p.TLS == q.TLS && maps.Equal(p.Headers, q.Headers) && - p.MaxOpenConns == q.MaxOpenConns && p.MaxIdleConns == q.MaxIdleConns +// Identity returns the tuple p names. +func (p Params) Identity() Identity { + return Identity{Addr: p.Addr, Database: p.Database, Username: p.Username, Password: p.Password, TLS: p.TLS} +} + +// Sizes returns the pool size p asks for. +func (p Params) Sizes() Sizes { + return Sizes{MaxOpenConns: p.MaxOpenConns, MaxIdleConns: p.MaxIdleConns} } // Target is what the HTTP-interface consumers need per request. @@ -108,141 +155,145 @@ type Target struct { Password string Database string // TLS configures an https URL; nil means net/http's defaults. The - // pointer changes only when a reload changes the tls block, which is - // what HTTPClients keys on. + // pointer is the tuple's, so it changes only when the tenant moves to a + // tuple with another tls block, which is what HTTPClients keys on. TLS *tls.Config // Headers go on every request ahead of the consumer's own; read-only. Headers map[string]string } -type state struct { - params Params - conn driver.Conn - tlsCfg *tls.Config -} - -// target derives the HTTP wiring from the native address's host. Addr is -// host:port by settings.Validate's contract, so the split cannot fail. -func (s *state) target() Target { - host, _, _ := net.SplitHostPort(s.params.Addr) +// target derives the HTTP wiring from the native address's host and the +// tuple's TLS config. Addr is host:port by settings.Validate's contract, so +// the split cannot fail. +func (p Params) target(tlsCfg *tls.Config) Target { + host, _, _ := net.SplitHostPort(p.Addr) return Target{ - URL: fmt.Sprintf("%s://%s", s.params.HTTPScheme, net.JoinHostPort(host, strconv.Itoa(s.params.HTTPPort))), - Username: s.params.Username, - Password: s.params.Password, - Database: s.params.Database, - TLS: s.tlsCfg, - Headers: s.params.Headers, + URL: fmt.Sprintf("%s://%s", p.HTTPScheme, net.JoinHostPort(host, strconv.Itoa(p.HTTPPort))), + Username: p.Username, + Password: p.Password, + Database: p.Database, + TLS: tlsCfg, + Headers: p.Headers, } } -// Manager is a driver.Conn whose backing connection is swapped by Reconfigure. +// dialer is the connection factory: clickhouse.Open in production, a fake +// in tests. Like clickhouse.Open it does not dial. +type dialer func(Identity, Sizes, *tls.Config) (driver.Conn, error) + +type state struct { + sizes Sizes + conn driver.Conn +} + +// Manager is a driver.Conn over one tuple's pool, whose backing connection +// is swapped by Resize. type Manager struct { - // dial is the connection factory; tests substitute it. - dial func(Params, *tls.Config) (driver.Conn, error) - // grace is how long a replaced connection stays open for in-flight - // queries before it is closed. - grace time.Duration + dial dialer + id Identity + tlsCfg *tls.Config - mu sync.Mutex // serializes Reconfigure/Close against each other + mu sync.Mutex // serializes Resize/Release/Close against each other cur atomic.Pointer[state] } var _ driver.Conn = (*Manager)(nil) -// Open builds the boot-time connection. Like clickhouse.Open it does not +// Open builds the pool for id at sizes s. Like clickhouse.Open it does not // dial — boot tolerates an unreachable ClickHouse (schema discovery degrades // and retries) — so only a malformed option or a certificate file that // cannot be read or parsed errors here. -func Open(p Params) (*Manager, error) { - m := &Manager{dial: dial, grace: p.QueryTimeout} - st, err := m.open(p, nil) +func Open(id Identity, s Sizes) (*Manager, error) { return open(dial, id, s) } + +func open(d dialer, id Identity, s Sizes) (*Manager, error) { + tlsCfg, err := id.TLS.config() if err != nil { return nil, err } - m.cur.Store(st) - return m, nil + return openWith(d, id, s, tlsCfg) } -// open builds p's state. The TLS config is carried over from prev while the -// tls block is unchanged, so a reload that moves only the database neither -// re-reads the certificate files nor makes the HTTP consumers rebuild their -// transports; a changed block reads the files again. -func (m *Manager) open(p Params, prev *state) (*state, error) { - var tlsCfg *tls.Config - if prev != nil && prev.params.TLS == p.TLS { - tlsCfg = prev.tlsCfg - } else { - var err error - if tlsCfg, err = p.TLS.config(); err != nil { - return nil, err - } - } - conn, err := m.dial(p, tlsCfg) +// openWith is open with the tuple's TLS config already built. +func openWith(d dialer, id Identity, s Sizes, tlsCfg *tls.Config) (*Manager, error) { + conn, err := d(id, s, tlsCfg) if err != nil { return nil, err } - return &state{params: p, conn: conn, tlsCfg: tlsCfg}, nil + m := &Manager{dial: d, id: id, tlsCfg: tlsCfg} + m.cur.Store(&state{sizes: s, conn: conn}) + return m, nil } // options is what the driver opens: the native address and credentials, // the pool sizes, and the TLS config when the native hop is TLS. -func options(p Params, tlsCfg *tls.Config) *clickhouse.Options { +func options(id Identity, s Sizes, tlsCfg *tls.Config) *clickhouse.Options { o := &clickhouse.Options{ - Addr: []string{p.Addr}, - Auth: clickhouse.Auth{Database: p.Database, Username: p.Username, Password: p.Password}, - MaxOpenConns: p.MaxOpenConns, - MaxIdleConns: p.MaxIdleConns, + Addr: []string{id.Addr}, + Auth: clickhouse.Auth{Database: id.Database, Username: id.Username, Password: id.Password}, + MaxOpenConns: s.MaxOpenConns, + MaxIdleConns: s.MaxIdleConns, } - if p.TLS.Enabled { + if id.TLS.Enabled { o.TLS = tlsCfg } return o } -func dial(p Params, tlsCfg *tls.Config) (driver.Conn, error) { - return clickhouse.Open(options(p, tlsCfg)) +func dial(id Identity, s Sizes, tlsCfg *tls.Config) (driver.Conn, error) { + return clickhouse.Open(options(id, s, tlsCfg)) } -// Reconfigure swaps in a connection built from p when p differs from the -// current wiring. The adopted settings are the authority: the swap is -// unconditional and, like Open, does not dial — an unreachable address -// surfaces where reachability is already handled (schema discovery -// retries, /readyz, query errors) and is fixed by the next reload. Only a -// malformed option, which settings.Validate already excludes, or a -// certificate file that cannot be read or parsed errors, and then the -// current connection stays. The replaced connection is closed after the grace -// period so in-flight queries on it finish. -func (m *Manager) Reconfigure(p Params) error { +// Identity returns the tuple this pool is for. +func (m *Manager) Identity() Identity { return m.id } + +// Sizes returns the pool's current size. +func (m *Manager) Sizes() Sizes { return m.cur.Load().sizes } + +// TLSConfig returns the tuple's TLS config, nil for a zero tls block: what +// the https hop of the HTTP interface is configured with. +func (m *Manager) TLSConfig() *tls.Config { return m.tlsCfg } + +// Resize swaps in a connection sized s when s differs from the current +// size — the tls block is the tuple's and never changes under a Manager, so +// no file is re-read. The replaced connection is closed after grace, so the +// queries in flight on it finish; grace is the longest query timeout among +// the tenants sharing the pool. +func (m *Manager) Resize(s Sizes, grace time.Duration) error { m.mu.Lock() defer m.mu.Unlock() old := m.cur.Load() - if old != nil && old.params.equal(p) { + if old.sizes == s { return nil } - st, err := m.open(p, old) + conn, err := m.dial(m.id, s, m.tlsCfg) if err != nil { - return fmt.Errorf("open %s: %w", p.Addr, err) - } - m.cur.Store(st) - if old != nil { - grace := m.grace - time.AfterFunc(grace, func() { _ = old.conn.Close() }) + return fmt.Errorf("open %s: %w", m.id.Addr, err) } - m.grace = p.QueryTimeout + m.cur.Store(&state{sizes: s, conn: conn}) + time.AfterFunc(grace, func() { _ = old.conn.Close() }) return nil } -// Target returns the current HTTP-interface wiring. -func (m *Manager) Target() Target { return m.cur.Load().target() } - -// Database returns the current database name. -func (m *Manager) Database() string { return m.cur.Load().params.Database } - -// QueryTimeout returns the current read deadline. -func (m *Manager) QueryTimeout() time.Duration { return m.cur.Load().params.QueryTimeout } +// Release closes the pool after grace, for a tuple no tenant names any +// more: a consumer that resolved this Manager just before the reload's +// swap finishes its query on it. +func (m *Manager) Release(grace time.Duration) { + m.mu.Lock() + defer m.mu.Unlock() + st := m.cur.Load() + time.AfterFunc(grace, func() { _ = st.conn.Close() }) +} -// Addr returns the current native address (for logs). -func (m *Manager) Addr() string { return m.cur.Load().params.Addr } +// Close closes the current connection now. Replaced and released +// connections close on their own grace timers. +func (m *Manager) Close() error { + m.mu.Lock() + defer m.mu.Unlock() + if s := m.cur.Load(); s != nil { + return s.conn.Close() + } + return errors.New("chconn: not open") +} func (m *Manager) conn() driver.Conn { return m.cur.Load().conn } @@ -284,22 +335,451 @@ func (m *Manager) AsyncInsert(ctx context.Context, query string, wait bool, args func (m *Manager) Ping(ctx context.Context) error { return m.conn().Ping(ctx) } func (m *Manager) Stats() driver.Stats { return m.conn().Stats() } -// Close closes the current connection. Replaced connections close on their -// own grace timers. -func (m *Manager) Close() error { - m.mu.Lock() - defer m.mu.Unlock() - if s := m.cur.Load(); s != nil { - return s.conn.Close() +// Member is one tenant's ask of the pools: the tuple it shares, and its own +// sizes, deadline and HTTP wiring. +type Member struct { + Tenant tenant.ID + Params Params +} + +// tuple is one open pool and the tenants on it, each with the Params last +// applied for it — what its Target is read from, and what the pool's size +// and close grace are the maxima of. manager is always open: a tuple +// planned in a reconcile had its pool opened when its first tenant joined. +type tuple struct { + manager *Manager + members map[tenant.ID]Params +} + +// sizes is the pool size the members ask for together. +func (t *tuple) sizes() Sizes { + var s Sizes + for _, p := range t.members { + s = s.max(p.Sizes()) } - return errors.New("chconn: not open") + return s } -// HTTPClients hands an HTTP-interface consumer the client for the current -// target: one per TLS config, made by the consumer's own factory (the -// worker's tuned transport, the proxy's redirect policy) and replaced when -// a reload changes the tls block, with the replaced transport's idle -// connections closed. Requests in flight on the old client finish. +// grace is how long a connection the members may be querying stays open +// once replaced or released: the longest of their query timeouts. +func (t *tuple) grace() time.Duration { + var g time.Duration + for _, p := range t.members { + g = max(g, p.QueryTimeout) + } + return g +} + +func (t *tuple) tenants() []tenant.ID { + return slices.Sorted(maps.Keys(t.members)) +} + +// snapshot is the pools at one instant, replaced whole by a reconcile so a +// lookup is one lock-free load. +type snapshot struct { + tuples map[Identity]*tuple + tenants map[tenant.ID]Identity // each tenant's tuple +} + +func (s *snapshot) clone() *snapshot { + next := &snapshot{tuples: make(map[Identity]*tuple, len(s.tuples)), tenants: maps.Clone(s.tenants)} + for id, t := range s.tuples { + next.tuples[id] = &tuple{manager: t.manager, members: maps.Clone(t.members)} + } + return next +} + +// Pools holds one Manager per tuple the served tenants name, reconciled +// after every settings reload: a new tuple opens a pool (no dial), a tenant +// whose tuple changed is repointed, a tuple no tenant names any more is +// released after its grace, and a pool shared by several tenants is sized to +// their largest ask. The ceiling — the boot config's +// clickhouse.max_total_conns, 0 for none — bounds the sum of the open pools' +// MaxOpenConns: at boot it refuses to open, and at a reload a resize above it +// is refused with the pool kept at its size, and a tuple that cannot be +// opened leaves its tenants where they were, on their previous pool, or with +// none when they had none. Both are logged by the caller and retried by the +// next reload. Resolution is per tenant: For, Target, SharingTables. +type Pools struct { + ceiling int + dial dialer + + mu sync.Mutex // serializes Reconcile and Close against each other + cur atomic.Pointer[snapshot] +} + +// NewPools opens the pools want names, under ceiling. Any refusal — a +// certificate file that cannot be read, options the driver refuses, or pools +// that would add up to more than the ceiling, named with the sum — refuses +// boot: nothing is left open. +func NewPools(ceiling int, want []Member) (*Pools, error) { + p := newPools(ceiling, dial) + if _, err := p.Reconcile(want); err != nil { + _ = p.Close() + return nil, err + } + return p, nil +} + +func newPools(ceiling int, d dialer) *Pools { + p := &Pools{ceiling: ceiling, dial: d} + p.cur.Store(&snapshot{tuples: map[Identity]*tuple{}, tenants: map[tenant.ID]Identity{}}) + return p +} + +// Reconcile sets the pools to what want names (the served tenants, in id +// order), and returns the tenants whose cache is stale, for the caller to +// orphan, with every refusal joined: the ones it admitted that were not on a +// pool before — new, or back after a rejection or removal, so out of the +// fan-out (SharingTables) while away — and the ones it moved to another +// address or database, whose cached results were read from other tables. A +// move that keeps the address and database (a username or tls change) reads +// the same tables and is not stale. The walk keeps the ceiling at every +// step: tenants no longer wanted leave first, then each wanted tenant is +// placed in turn, and a refused placement — the ceiling, or a pool that +// cannot be opened — is undone before the next, so a refused move leaves the +// tenant on the pool it had, with the Params it had. What was refused is +// placed once more at the end, since a shrink or a move placed after it may +// have freed the budget it needed; only what the second pass refuses is +// reported. +func (p *Pools) Reconcile(want []Member) (stale []tenant.ID, err error) { + p.mu.Lock() + defer p.mu.Unlock() + cur := p.cur.Load() + w := &walk{ceiling: p.ceiling, dial: p.dial, next: cur.clone(), planned: map[Identity]Sizes{}, opened: map[Identity]bool{}} + for id, t := range w.next.tuples { + w.planned[id] = t.manager.Sizes() + } + + wanted := make(map[tenant.ID]bool, len(want)) + w.asks = make(map[Identity]Sizes, len(want)) + for _, m := range want { + wanted[m.Tenant] = true + id := m.Params.Identity() + w.asks[id] = w.asks[id].max(m.Params.Sizes()) + } + for _, id := range slices.Sorted(maps.Keys(w.next.tenants)) { + if !wanted[id] { + w.leave(id) + } + } + var refused []Member + for _, m := range want { + if !w.place(m) { + refused = append(refused, m) + } + } + if len(refused) > 0 { + w.errs, w.refused = nil, nil + for _, m := range refused { + w.place(m) + } + } + + // Apply: release the emptied tuples and resize the rest to their plan. + // A connection the previous members may still be querying stays open + // for the longest of their timeouts; one this walk opened has had no + // consumer, so it goes at once. + for _, id := range w.order() { + t := w.next.tuples[id] + s := w.planned[id] + grace := w.graceBefore(cur, id) + if w.opened[id] { + grace = 0 + } + switch { + case len(t.members) == 0: + if w.opened[id] { + _ = t.manager.Close() + } else { + t.manager.Release(grace) + } + delete(w.next.tuples, id) + case s != t.manager.Sizes(): + if err := t.manager.Resize(s, grace); err != nil { + w.errs = append(w.errs, fmt.Errorf("clickhouse pool %s kept at %d open connections: %w", id.name(), t.manager.Sizes().MaxOpenConns, err)) + } + } + } + p.cur.Store(w.next) + for _, id := range slices.Sorted(maps.Keys(w.next.tenants)) { + prev, had := cur.tenants[id] + now := w.next.tenants[id] + if !had || prev.Addr != now.Addr || prev.Database != now.Database { + stale = append(stale, id) + } + } + return stale, errors.Join(w.errs...) +} + +// walk is one Reconcile's working state: the snapshot being built, and the +// size each of its tuples is planned to be resized to, which the ceiling +// check sums over the tuples that still have members. +type walk struct { + ceiling int + dial dialer + next *snapshot + planned map[Identity]Sizes + // refused is the tuples whose growth this pass already refused, so a + // shared pool's refusal is reported once, naming every member. + refused map[Identity]bool + // opened is the tuples this walk opened: a pool opens when a tenant + // first joins its tuple (reading its certificate files; the driver + // never dials), so a pool that cannot be opened is a refusal to undo in + // place, like the ceiling's. It opens at asks, the largest ask among + // the wanted tenants naming it, when that fits the ceiling, so it opens + // once when they all fit; otherwise at the joining tenant's own ask. + // Until the walk is applied no consumer holds one, so an opened pool + // left with no members is closed at once and one the walk settled on + // another size for is resized with no grace. + opened map[Identity]bool + asks map[Identity]Sizes + errs []error +} + +// order is the tuples in a fixed order, for logs and errors. +func (w *walk) order() []Identity { + ids := slices.Collect(maps.Keys(w.next.tuples)) + slices.SortFunc(ids, func(a, b Identity) int { return strings.Compare(a.String(), b.String()) }) + return ids +} + +// graceBefore is the close grace of tuple id as it was before this +// reconcile: the longest query timeout among the members it had, since they +// are who may still be querying its connection. +func (w *walk) graceBefore(cur *snapshot, id Identity) time.Duration { + if prev, ok := cur.tuples[id]; ok { + return prev.grace() + } + return w.next.tuples[id].grace() +} + +// fits reports whether tuple id at size s keeps the open pools within the +// ceiling, and the sum they would be at: the planned sizes of the other +// tuples that have members, plus s. +func (w *walk) fits(id Identity, s Sizes) (bool, int) { + sum := s.MaxOpenConns + for other, t := range w.next.tuples { + if other != id && len(t.members) > 0 { + sum += w.planned[other].MaxOpenConns + } + } + return w.ceiling <= 0 || sum <= w.ceiling, sum +} + +// leave takes id off its tuple. The tuple's plan shrinks to its remaining +// members' ask when that is smaller; a shrink is always within the ceiling. +func (w *walk) leave(id tenant.ID) { + ident, ok := w.next.tenants[id] + if !ok { + return + } + delete(w.next.tenants, id) + t := w.next.tuples[ident] + delete(t.members, id) + if s := t.sizes(); s.MaxOpenConns <= w.planned[ident].MaxOpenConns { + w.planned[ident] = s + } +} + +// place puts m on the tuple its Params name — its current one, updated in +// place, or another, which it moves to when that tuple's pool opens under +// the ceiling — and reports whether it was placed. A refused move leaves m +// on the tuple it had, with the Params it had. +func (w *walk) place(m Member) bool { + ident := m.Params.Identity() + prev, had := w.next.tenants[m.Tenant] + if had && prev == ident { + w.next.tuples[ident].members[m.Tenant] = m.Params + return w.grow(ident) + } + var prevParams Params + var prevPlanned Sizes + if had { + prevParams, prevPlanned = w.next.tuples[prev].members[m.Tenant], w.planned[prev] + w.leave(m.Tenant) + } + err := w.join(m, ident) + if err == nil { + return true + } + if had { + // Back where it was, with the Params it had: the step before this + // was within the ceiling, so restoring it is too. + w.next.tuples[prev].members[m.Tenant] = prevParams + w.next.tenants[m.Tenant] = prev + w.planned[prev] = prevPlanned + err = fmt.Errorf("%w; tenant %s keeps its previous pool %s", err, m.Tenant, prev.name()) + } + w.errs = append(w.errs, err) + return false +} + +// join adds m to tuple ident, opening the pool when it is new. A refusal — +// over the ceiling, or a pool that cannot be opened — leaves m off it. +func (w *walk) join(m Member, ident Identity) error { + t, exists := w.next.tuples[ident] + if !exists { + s := m.Params.Sizes() + if ok, sum := w.fits(ident, s); !ok { + return fmt.Errorf("clickhouse pool %s not opened for tenant %s: its clickhouse.max_open_conns %d would put the open pools at %d, above clickhouse.max_total_conns %d (boot config)", ident.name(), m.Tenant, s.MaxOpenConns, sum, w.ceiling) + } + tlsCfg, err := ident.TLS.config() + if err != nil { + return fmt.Errorf("clickhouse pool %s not opened for tenant %s: %w", ident.name(), m.Tenant, err) + } + // Open at the largest ask among the tenants naming the tuple when + // the ceiling allows it, so a shared tuple opens once; otherwise at + // this tenant's own, which it just fit. + open := w.asks[ident] + if ok, _ := w.fits(ident, open); !ok { + open = s + } + mgr, err := openWith(w.dial, ident, open, tlsCfg) + if err != nil { + return fmt.Errorf("clickhouse pool %s not opened for tenant %s: %w", ident.name(), m.Tenant, err) + } + w.next.tuples[ident] = &tuple{manager: mgr, members: map[tenant.ID]Params{m.Tenant: m.Params}} + w.next.tenants[m.Tenant] = ident + w.planned[ident] = s + w.opened[ident] = true + return nil + } + // An existing pool takes a tenant whose ask fits its planned size, and + // grows for a larger one when the ceiling allows. + s := w.planned[ident].max(m.Params.Sizes()) + if s.MaxOpenConns > w.planned[ident].MaxOpenConns { + if ok, sum := w.fits(ident, s); !ok { + return fmt.Errorf("clickhouse pool %s not grown for tenant %s: its clickhouse.max_open_conns %d would put the open pools at %d, above clickhouse.max_total_conns %d (boot config)", ident.name(), m.Tenant, s.MaxOpenConns, sum, w.ceiling) + } + } + w.planned[ident] = s + t.members[m.Tenant] = m.Params + w.next.tenants[m.Tenant] = ident + return nil +} + +// grow re-plans tuple ident's size after a member's Params changed in place: +// its members' ask when that fits the ceiling, else the size it has, which +// the next reload retries. Reports whether the ask fit. +func (w *walk) grow(ident Identity) bool { + t := w.next.tuples[ident] + s := t.sizes() + if s.MaxOpenConns <= w.planned[ident].MaxOpenConns { + w.planned[ident] = s + return true + } + if ok, sum := w.fits(ident, s); !ok { + if !w.refused[ident] { + if w.refused == nil { + w.refused = map[Identity]bool{} + } + w.refused[ident] = true + w.errs = append(w.errs, fmt.Errorf("clickhouse pool %s not resized for tenants %v: their clickhouse.max_open_conns %d would put the open pools at %d, above clickhouse.max_total_conns %d (boot config); it keeps %d", ident.name(), t.tenants(), s.MaxOpenConns, sum, w.ceiling, w.planned[ident].MaxOpenConns)) + } + return false + } + w.planned[ident] = s + return true +} + +// For returns the pool of tenant id, or nil when the tenant is on none: it +// is not served, or its tuple could not be opened. +func (p *Pools) For(id tenant.ID) *Manager { + snap := p.cur.Load() + ident, ok := snap.tenants[id] + if !ok { + return nil + } + return snap.tuples[ident].manager +} + +// Target returns the HTTP-interface wiring of tenant id — its own HTTP +// port, scheme and headers over its pool's host, credentials, database and +// TLS config — or the zero Target when it is on no pool. +func (p *Pools) Target(id tenant.ID) Target { + snap := p.cur.Load() + ident, ok := snap.tenants[id] + if !ok { + return Target{} + } + t := snap.tuples[ident] + return t.members[id].target(t.manager.tlsCfg) +} + +// SharingTables returns the tenants reading the same tables as tenant id — +// the same address and database, whatever their user or tls — id included, +// in id order; nil when id is on no pool. +func (p *Pools) SharingTables(id tenant.ID) []tenant.ID { + snap := p.cur.Load() + ident, ok := snap.tenants[id] + if !ok { + return nil + } + var ids []tenant.ID + for tid, tident := range snap.tenants { + if tident.Addr == ident.Addr && tident.Database == ident.Database { + ids = append(ids, tid) + } + } + slices.Sort(ids) + return ids +} + +// Ping pings every open pool at once and returns nil at the first answer, +// or every pool's error joined when none answers — including when none is +// open. At once, not in turn: the driver waits up to its dial timeout on a +// host that does not answer, longer than a readiness probe's, so one such +// pool must not hide the ones that do answer. +func (p *Pools) Ping(ctx context.Context) error { + snap := p.cur.Load() + ids := slices.Collect(maps.Keys(snap.tuples)) + if len(ids) == 0 { + return errors.New("no ClickHouse pool is open") + } + slices.SortFunc(ids, func(a, b Identity) int { return strings.Compare(a.String(), b.String()) }) + ctx, cancel := context.WithCancel(ctx) + defer cancel() + type result struct { + i int + err error + } + results := make(chan result, len(ids)) + for i, id := range ids { + go func() { results <- result{i, snap.tuples[id].manager.Ping(ctx)} }() + } + errs := make([]error, len(ids)) + for range ids { + r := <-results + if r.err == nil { + return nil + } + errs[r.i] = fmt.Errorf("%s: %w", ids[r.i].name(), r.err) + } + return errors.Join(errs...) +} + +// Close closes every open pool now and leaves none. +func (p *Pools) Close() error { + p.mu.Lock() + defer p.mu.Unlock() + snap := p.cur.Swap(&snapshot{tuples: map[Identity]*tuple{}, tenants: map[tenant.ID]Identity{}}) + var errs []error + for _, t := range snap.tuples { + if err := t.manager.Close(); err != nil { + errs = append(errs, err) + } + } + return errors.Join(errs...) +} + +// HTTPClients hands an HTTP-interface consumer the client for a target: one +// per TLS config, made by the consumer's own factory (the worker's tuned +// transport, the proxy's redirect policy) and kept for the process lifetime. +// A config is built each time a pool opens on a tls block, so the set grows +// with the pools ever opened on one — a block whose pool closed and +// reopened adds another — not with requests; a client whose pool is gone +// keeps only its transport, whose idle connections time out on their own. // // The factory gets a copy of the config, never the target's own: net/http // appends its HTTP/2 protocols to TLSClientConfig.NextProtos in place when @@ -309,26 +789,24 @@ func (m *Manager) Close() error { type HTTPClients struct { build func(*tls.Config) *http.Client - mu sync.Mutex - cfg *tls.Config - cur *http.Client + mu sync.Mutex + clients map[*tls.Config]*http.Client } -// NewHTTPClients returns a cache whose clients build makes from the -// target's TLS config, nil meaning net/http's defaults. +// NewHTTPClients returns a cache whose clients build makes from a target's +// TLS config, nil meaning net/http's defaults. func NewHTTPClients(build func(*tls.Config) *http.Client) *HTTPClients { - return &HTTPClients{build: build} + return &HTTPClients{build: build, clients: map[*tls.Config]*http.Client{}} } // For returns the client for t. func (c *HTTPClients) For(t Target) *http.Client { c.mu.Lock() defer c.mu.Unlock() - if c.cur == nil || c.cfg != t.TLS { - if c.cur != nil { - c.cur.CloseIdleConnections() - } - c.cfg, c.cur = t.TLS, c.build(t.TLS.Clone()) + if cl, ok := c.clients[t.TLS]; ok { + return cl } - return c.cur + cl := c.build(t.TLS.Clone()) + c.clients[t.TLS] = cl + return cl } diff --git a/internal/chconn/chconn_test.go b/internal/chconn/chconn_test.go index f15bf4b4..96b1c538 100644 --- a/internal/chconn/chconn_test.go +++ b/internal/chconn/chconn_test.go @@ -1,6 +1,7 @@ package chconn import ( + "context" "crypto/ecdsa" "crypto/elliptic" "crypto/rand" @@ -14,7 +15,6 @@ import ( "net/http/httptest" "os" "path/filepath" - "reflect" "sync" "sync/atomic" "testing" @@ -23,32 +23,78 @@ import ( "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/tenant" ) -// fakeConn records closes; embedding the interface leaves the unused -// methods nil, which is fine — they are never called here. +// fakeConn records closes and answers Ping as told; embedding the interface +// leaves the unused methods nil, which is fine — they are never called here. type fakeConn struct { driver.Conn - name string + id Identity + sizes Sizes closed atomic.Bool + // ping answers Ping; nil means success at once. + ping func(context.Context) error } func (f *fakeConn) Close() error { f.closed.Store(true); return nil } -func params(addr string) Params { - return Params{Addr: addr, HTTPPort: 8123, HTTPScheme: "http", Database: "db", Username: "u", Password: "p", QueryTimeout: time.Second, MaxOpenConns: 10, MaxIdleConns: 5} +func (f *fakeConn) Ping(ctx context.Context) error { + if f.ping == nil { + return nil + } + return f.ping(ctx) } // fakeDial stands in for the driver; the TLS config still comes through // the real path (TLS.config), which is what the TLS tests exercise. -func fakeDial(p Params, _ *tls.Config) (driver.Conn, error) { return &fakeConn{name: p.Addr}, nil } +func fakeDial(id Identity, s Sizes, _ *tls.Config) (driver.Conn, error) { + return &fakeConn{id: id, sizes: s}, nil +} + +// dialRecorder is a fakeDial that keeps every connection it made, by +// address, latest last. +type dialRecorder struct { + mu sync.Mutex + conns map[string][]*fakeConn + // ping is given to every connection made. + ping func(context.Context) error +} + +func (r *dialRecorder) dial(id Identity, s Sizes, _ *tls.Config) (driver.Conn, error) { + r.mu.Lock() + defer r.mu.Unlock() + if r.conns == nil { + r.conns = map[string][]*fakeConn{} + } + c := &fakeConn{id: id, sizes: s, ping: r.ping} + r.conns[id.Addr] = append(r.conns[id.Addr], c) + return c, nil +} + +// latest is the newest connection made for addr. +func (r *dialRecorder) latest(addr string) *fakeConn { + r.mu.Lock() + defer r.mu.Unlock() + cs := r.conns[addr] + return cs[len(cs)-1] +} -func newManager(t *testing.T, dial func(Params, *tls.Config) (driver.Conn, error)) *Manager { +func (r *dialRecorder) count(addr string) int { + r.mu.Lock() + defer r.mu.Unlock() + return len(r.conns[addr]) +} + +func params(addr string) Params { + return Params{Addr: addr, HTTPPort: 8123, HTTPScheme: "http", Database: "db", Username: "u", Password: "p", QueryTimeout: time.Second, MaxOpenConns: 10, MaxIdleConns: 5} +} + +func newManager(t *testing.T, d dialer) *Manager { t.Helper() - m := &Manager{dial: dial, grace: 10 * time.Millisecond} - st, err := m.open(params("a:9000"), nil) + m, err := open(d, params("a:9000").Identity(), Sizes{MaxOpenConns: 10, MaxIdleConns: 5}) require.NoError(t, err) - m.cur.Store(st) return m } @@ -85,66 +131,116 @@ func writeTestPKI(t *testing.T) (caFile, certFile, keyFile string) { return write("ca.pem", "CERTIFICATE", caDER), write("client.pem", "CERTIFICATE", leafDER), write("client.key", "EC PRIVATE KEY", keyDER) } -func TestManager_TargetDerivesHTTPURL(t *testing.T) { +func TestParams_TargetDerivesHTTPURL(t *testing.T) { t.Parallel() - m := newManager(t, fakeDial) - assert.Equal(t, Target{URL: "http://a:8123", Username: "u", Password: "p", Database: "db"}, m.Target()) - assert.Equal(t, "db", m.Database()) - assert.Equal(t, time.Second, m.QueryTimeout()) + assert.Equal(t, Target{URL: "http://a:8123", Username: "u", Password: "p", Database: "db"}, params("a:9000").target(nil)) + p := params("a:9000") + p.HTTPScheme, p.HTTPPort, p.Headers = "https", 8443, map[string]string{"X-Proxy-Token": "abc"} + cfg := &tls.Config{MinVersion: tls.VersionTLS12} + tgt := p.target(cfg) + assert.Equal(t, "https://a:8443", tgt.URL) + assert.Same(t, cfg, tgt.TLS) + assert.Equal(t, map[string]string{"X-Proxy-Token": "abc"}, tgt.Headers) } -func TestManager_ReconfigureSwapsAndClosesOldAfterGrace(t *testing.T) { +// TestIdentity_IsTheComparableTuple: the tuple is a plain value, so two +// Params naming the same pool compare equal and index the same map entry, +// and every field of the tuple — the tls block included — tells them apart. +func TestIdentity_IsTheComparableTuple(t *testing.T) { t.Parallel() - conns := map[string]*fakeConn{} - m := newManager(t, func(p Params, _ *tls.Config) (driver.Conn, error) { - c := &fakeConn{name: p.Addr} - conns[p.Addr] = c - return c, nil - }) - require.NoError(t, m.Reconfigure(params("b:9000"))) - assert.Equal(t, "b:9000", m.Addr()) - assert.Equal(t, "http://b:8123", m.Target().URL) - assert.Same(t, conns["b:9000"], m.conn()) - assert.Eventually(t, func() bool { return conns["a:9000"].closed.Load() }, time.Second, 5*time.Millisecond, "old connection closes after the grace period") - assert.False(t, conns["b:9000"].closed.Load()) + base := params("a:9000") + base.TLS = TLS{Enabled: true, ServerName: "ch.internal"} + same := base + same.HTTPPort, same.QueryTimeout, same.MaxOpenConns, same.Headers = 9999, time.Hour, 99, map[string]string{"X-A": "1"} + assert.Equal(t, base.Identity(), same.Identity(), "the HTTP wiring, the deadline, the sizes and the headers are the tenant's own") + + for name, change := range map[string]func(*Params){ + "addr": func(p *Params) { p.Addr = "b:9000" }, + "database": func(p *Params) { p.Database = "other" }, + "username": func(p *Params) { p.Username = "reporting" }, + "password": func(p *Params) { p.Password = "x" }, + "tls": func(p *Params) { p.TLS.InsecureSkipVerify = true }, + } { + changed := base + change(&changed) + assert.NotEqual(t, base.Identity(), changed.Identity(), name) + } + assert.Equal(t, "a:9000 database db user u", base.Identity().String(), "named without the password") +} + +func TestSizes_MaxIsPerDimension(t *testing.T) { + t.Parallel() + a, b := Sizes{MaxOpenConns: 10, MaxIdleConns: 5}, Sizes{MaxOpenConns: 8, MaxIdleConns: 8} + assert.Equal(t, Sizes{MaxOpenConns: 10, MaxIdleConns: 8}, a.max(b)) + assert.Equal(t, a.max(b), b.max(a)) +} + +func TestManager_ResizeSwapsAndClosesOldAfterGrace(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + m := newManager(t, rec.dial) + first := rec.latest("a:9000") + require.NoError(t, m.Resize(Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, 10*time.Millisecond)) + assert.Equal(t, Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, m.Sizes()) + assert.Same(t, rec.latest("a:9000"), m.conn()) + assert.Equal(t, Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, rec.latest("a:9000").sizes, "the driver is opened at the new size") + assert.Eventually(t, func() bool { return first.closed.Load() }, time.Second, 5*time.Millisecond, "old connection closes after the grace period") + assert.False(t, rec.latest("a:9000").closed.Load()) + assert.Equal(t, "a:9000", m.Identity().Addr) } -func TestManager_ReconfigureSameParamsIsNoop(t *testing.T) { +func TestManager_ResizeSameSizesIsNoop(t *testing.T) { t.Parallel() - dials := 0 - m := newManager(t, func(p Params, _ *tls.Config) (driver.Conn, error) { dials++; return &fakeConn{name: p.Addr}, nil }) - require.NoError(t, m.Reconfigure(params("a:9000"))) - assert.Equal(t, 1, dials, "identical wiring must not re-dial") + rec := &dialRecorder{} + m := newManager(t, rec.dial) + require.NoError(t, m.Resize(m.Sizes(), time.Millisecond)) + assert.Equal(t, 1, rec.count("a:9000"), "an unchanged size must not re-dial") } -// TestManager_ReconfigureDialError pins one of the two ways a swap can -// fail: a malformed option (excluded by settings.Validate) leaves the -// current connection in place. Reachability is never checked here. -func TestManager_ReconfigureDialError(t *testing.T) { +// TestManager_ResizeDialError: a malformed option (excluded by +// settings.Validate) leaves the current connection in place. +func TestManager_ResizeDialError(t *testing.T) { t.Parallel() - m := newManager(t, func(p Params, _ *tls.Config) (driver.Conn, error) { - if p.Addr == "bad:9000" { + var fail atomic.Bool + m := newManager(t, func(id Identity, s Sizes, _ *tls.Config) (driver.Conn, error) { + if fail.Load() { return nil, errors.New("bad options") } - return &fakeConn{name: p.Addr}, nil + return &fakeConn{id: id, sizes: s}, nil }) - require.ErrorContains(t, m.Reconfigure(params("bad:9000")), "open bad:9000") - assert.Equal(t, "a:9000", m.Addr()) + fail.Store(true) + require.ErrorContains(t, m.Resize(Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, time.Millisecond), "open a:9000") + assert.Equal(t, Sizes{MaxOpenConns: 10, MaxIdleConns: 5}, m.Sizes()) } -// TestManager_ReconfigureUnreadableCertificateKeepsTheConnection pins the -// other: a certificate file that cannot be read, which Validate does not -// open, errors here and leaves the current connection in place. -func TestManager_ReconfigureUnreadableCertificateKeepsTheConnection(t *testing.T) { +func TestManager_ReleaseClosesAfterGrace(t *testing.T) { t.Parallel() - m := newManager(t, fakeDial) - p := params("b:9000") - p.TLS = TLS{Enabled: true, CAFile: "/nowhere/ca.pem"} - err := m.Reconfigure(p) - require.ErrorContains(t, err, "open b:9000") + rec := &dialRecorder{} + m := newManager(t, rec.dial) + c := rec.latest("a:9000") + m.Release(20 * time.Millisecond) + assert.False(t, c.closed.Load(), "a consumer that resolved the pool before the swap still finishes on it") + assert.Eventually(t, func() bool { return c.closed.Load() }, time.Second, 5*time.Millisecond) +} + +func TestManager_CloseIsImmediate(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + m := newManager(t, rec.dial) + require.NoError(t, m.Close()) + assert.True(t, rec.latest("a:9000").closed.Load()) +} + +// TestOpen_UnreadableCertificateNamesTheKeyAndPath: a certificate file that +// cannot be read, which Validate does not open, is the one thing Open +// refuses. +func TestOpen_UnreadableCertificateNamesTheKeyAndPath(t *testing.T) { + t.Parallel() + id := params("b:9000").Identity() + id.TLS = TLS{Enabled: true, CAFile: "/nowhere/ca.pem"} + _, err := open(fakeDial, id, Sizes{MaxOpenConns: 1, MaxIdleConns: 1}) require.ErrorContains(t, err, "clickhouse.tls.ca_file") require.ErrorContains(t, err, "/nowhere/ca.pem") - assert.Equal(t, "a:9000", m.Addr()) } func TestTLS_ConfigReadsTheFiles(t *testing.T) { @@ -195,83 +291,425 @@ func TestTLS_UnreadableFilesNameTheKeyAndPath(t *testing.T) { func TestOptions_ReachTheDriver(t *testing.T) { t.Parallel() cfg := &tls.Config{MinVersion: tls.VersionTLS12} - p := params("a:9000") - o := options(p, cfg) + id := params("a:9000").Identity() + o := options(id, Sizes{MaxOpenConns: 10, MaxIdleConns: 5}, cfg) assert.Equal(t, []string{"a:9000"}, o.Addr) + assert.Equal(t, "db", o.Auth.Database) + assert.Equal(t, "u", o.Auth.Username) + assert.Equal(t, "p", o.Auth.Password) assert.Equal(t, 10, o.MaxOpenConns) assert.Equal(t, 5, o.MaxIdleConns) assert.Nil(t, o.TLS, "the native hop stays plain while tls.enabled is false") - p.TLS.Enabled = true - assert.Same(t, cfg, options(p, cfg).TLS) + id.TLS.Enabled = true + assert.Same(t, cfg, options(id, Sizes{}, cfg).TLS) } -func TestManager_TargetCarriesTLSAndHeaders(t *testing.T) { +// TestManager_TLSConfigIsTheTuplesAndBuiltOnce: the tls block is part of +// the tuple, so a Manager reads the files once and every resize keeps the +// same config — the HTTP transports keyed on it are never rebuilt. +func TestManager_TLSConfigIsTheTuplesAndBuiltOnce(t *testing.T) { t.Parallel() ca, _, _ := writeTestPKI(t) p := params("a:9000") p.HTTPScheme = "https" p.TLS = TLS{CAFile: ca} - p.Headers = map[string]string{"X-Proxy-Token": "abc"} - m := &Manager{dial: fakeDial} - st, err := m.open(p, nil) - require.NoError(t, err) - m.cur.Store(st) - tgt := m.Target() - assert.Equal(t, "https://a:8123", tgt.URL) - require.NotNil(t, tgt.TLS, "the material applies to the https hop even with tls.enabled false") - assert.NotNil(t, tgt.TLS.RootCAs) - assert.Equal(t, map[string]string{"X-Proxy-Token": "abc"}, tgt.Headers) + m, err := open(fakeDial, p.Identity(), p.Sizes()) + require.NoError(t, err) + first := m.TLSConfig() + require.NotNil(t, first, "the material applies to the https hop even with tls.enabled false") + assert.NotNil(t, first.RootCAs) + require.NoError(t, m.Resize(Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, time.Millisecond)) + assert.Same(t, first, m.TLSConfig()) + assert.Same(t, first, p.target(m.TLSConfig()).TLS) +} + +// member is one tenant asking for the pool at addr, with its own sizes. +func member(id tenant.ID, addr string, open, idle int) Member { + p := params(addr) + p.MaxOpenConns, p.MaxIdleConns = open, idle + return Member{Tenant: id, Params: p} } -func TestManager_ReconfigureKeepsTLSConfigWhileTheBlockIsUnchanged(t *testing.T) { +func TestPools_DifferentTuplesGetDifferentPools(t *testing.T) { t.Parallel() - ca, _, _ := writeTestPKI(t) - p := params("a:9000") - p.TLS = TLS{Enabled: true, CAFile: ca} - m := &Manager{dial: fakeDial, grace: 10 * time.Millisecond} - st, err := m.open(p, nil) + p := newPools(0, fakeDial) + stale, err := p.Reconcile([]Member{member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5)}) require.NoError(t, err) - m.cur.Store(st) - first := m.Target().TLS - require.NotNil(t, first) + assert.Equal(t, []tenant.ID{"acme", "globex"}, stale) + require.NotNil(t, p.For("acme")) + require.NotNil(t, p.For("globex")) + assert.NotSame(t, p.For("acme"), p.For("globex")) + assert.Equal(t, "http://a:8123", p.Target("acme").URL) + assert.Equal(t, "http://b:8123", p.Target("globex").URL) + assert.Equal(t, []tenant.ID{"acme"}, p.SharingTables("acme")) + assert.Nil(t, p.For("initech"), "a tenant on no pool") + assert.Equal(t, Target{}, p.Target("initech")) + assert.Nil(t, p.SharingTables("initech")) +} - q := p - q.Database = "other" - require.NoError(t, m.Reconfigure(q)) - assert.Same(t, first, m.Target().TLS, "an unchanged tls block keeps the config, so the HTTP transports are not rebuilt") +// TestPools_SharedTupleGetsOnePool: tenants naming the same tuple share one +// Manager, sized to the largest ask in each dimension, each with a Target of +// its own HTTP wiring. +func TestPools_SharedTupleGetsOnePool(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(0, rec.dial) + globex := member("globex", "a:9000", 8, 8) + globex.Params.HTTPPort, globex.Params.Headers = 8443, map[string]string{"X-Proxy-Token": "g"} + _, err := p.Reconcile([]Member{member("acme", "a:9000", 10, 5), globex}) + require.NoError(t, err) + require.NotNil(t, p.For("acme")) + assert.Same(t, p.For("acme"), p.For("globex")) + assert.Equal(t, 1, rec.count("a:9000"), "one pool, opened once") + assert.Equal(t, Sizes{MaxOpenConns: 10, MaxIdleConns: 8}, p.For("acme").Sizes(), "the largest ask in each dimension") + assert.Equal(t, "http://a:8123", p.Target("acme").URL) + assert.Equal(t, "http://a:8443", p.Target("globex").URL) + assert.Equal(t, map[string]string{"X-Proxy-Token": "g"}, p.Target("globex").Headers) + assert.Equal(t, []tenant.ID{"acme", "globex"}, p.SharingTables("acme")) + assert.Equal(t, []tenant.ID{"acme", "globex"}, p.SharingTables("globex")) +} - q.TLS.ServerName = "ch.internal" - require.NoError(t, m.Reconfigure(q)) - assert.NotSame(t, first, m.Target().TLS) - assert.Equal(t, "ch.internal", m.Target().TLS.ServerName) +// A shared tuple opens at its tenants' largest ask only when that fits the +// ceiling: with a ceiling of 10, acme asking 1 and globex 20 on one tuple, +// the pool opens at acme's 1 — never above the ceiling, not even before the +// walk refuses globex's growth — and stays there. +func TestPools_NewPoolNeverOpensAboveTheCeiling(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(10, rec.dial) + _, err := p.Reconcile([]Member{member("acme", "a:9000", 1, 1), member("globex", "a:9000", 20, 5)}) + require.ErrorContains(t, err, "not grown for tenant globex") + require.Equal(t, 1, rec.count("a:9000")) + assert.Equal(t, Sizes{MaxOpenConns: 1, MaxIdleConns: 1}, rec.latest("a:9000").sizes, "opened at acme's ask, within the ceiling") + assert.Equal(t, Sizes{MaxOpenConns: 1, MaxIdleConns: 1}, p.For("acme").Sizes()) + assert.Nil(t, p.For("globex"), "refused, and it had no pool to keep") } -// TestParams_EqualCoversEveryField mutates each field in turn, so a field -// added to Params without a clause in equal fails here rather than making -// Reconfigure skip a real change. -func TestParams_EqualCoversEveryField(t *testing.T) { +// TestPools_TupleChangeRepointsWithoutTouchingTheOther is the worked +// example: two tenants on one tuple, one changes its username. It moves to a +// pool of its own; the other keeps the very same Manager, resized down to +// its own ask with the old connection closed after the grace; and both still +// read the same tables, so the fan-out still pairs them. +func TestPools_TupleChangeRepointsWithoutTouchingTheOther(t *testing.T) { t.Parallel() - base := params("a:9000") - base.Headers = map[string]string{"X-A": "1"} - require.True(t, base.equal(base)) - rt := reflect.TypeFor[Params]() - for i := range rt.NumField() { - changed := base - f := reflect.ValueOf(&changed).Elem().Field(i) - switch f.Kind() { //nolint:exhaustive // the default names any kind a new field would add - case reflect.String: - f.SetString(f.String() + "x") - case reflect.Int, reflect.Int64: - f.SetInt(f.Int() + 1) - case reflect.Map: - f.Set(reflect.ValueOf(map[string]string{"X-A": "2"})) - case reflect.Struct: - f.Field(0).SetBool(!f.Field(0).Bool()) - default: - t.Fatalf("field %s: kind %s not covered", rt.Field(i).Name, f.Kind()) + rec := &dialRecorder{} + p := newPools(40, rec.dial) + acme, globex := member("acme", "a:9000", 10, 5), member("globex", "a:9000", 20, 8) + acme.Params.QueryTimeout, globex.Params.QueryTimeout = 30*time.Millisecond, 10*time.Millisecond + _, err := p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + shared := p.For("acme") + require.Same(t, shared, p.For("globex")) + require.Equal(t, Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, shared.Sizes()) + before := rec.latest("a:9000") + + globex.Params.Username = "reporting" + stale, err := p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + assert.Empty(t, stale, "a new username reads the same tables: not stale") + assert.Same(t, shared, p.For("acme"), "acme keeps its Manager") + assert.NotSame(t, shared, p.For("globex"), "globex moved to a pool of its own") + assert.Equal(t, "reporting", p.For("globex").Identity().Username) + assert.Equal(t, Sizes{MaxOpenConns: 20, MaxIdleConns: 8}, p.For("globex").Sizes()) + assert.Equal(t, Sizes{MaxOpenConns: 10, MaxIdleConns: 5}, shared.Sizes(), "acme's pool shrinks to acme's ask") + assert.Equal(t, 3, rec.count("a:9000"), "the shrink and the new pool are two new connections") + assert.Eventually(t, func() bool { return before.closed.Load() }, time.Second, 5*time.Millisecond, "the replaced connection closes after the longest member timeout") + assert.Equal(t, []tenant.ID{"acme", "globex"}, p.SharingTables("acme"), "same address and database: still the same tables") + assert.Equal(t, "reporting", p.Target("globex").Username) + assert.Equal(t, "u", p.Target("acme").Username) +} + +// A tenant moved to another address or database reads other tables, so its +// cache is stale like a readmitted tenant's; the tenant left where it was is +// not, and neither is one whose move keeps the address and database. +func TestPools_MoveToOtherTablesIsStale(t *testing.T) { + t.Parallel() + p := newPools(0, fakeDial) + acme, globex := member("acme", "a:9000", 10, 5), member("globex", "a:9000", 10, 5) + _, err := p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + + acme.Params.Addr = "b:9000" + stale, err := p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + assert.Equal(t, []tenant.ID{"acme"}, stale, "another address") + + globex.Params.Database = "other" + stale, err = p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + assert.Equal(t, []tenant.ID{"globex"}, stale, "another database") + + stale, err = p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + assert.Empty(t, stale, "nothing moved") +} + +// A pool that cannot be opened is a refusal like the ceiling's: a tenant +// moving onto it stays on the pool it had, with the Params it had, and that +// pool is not released even when the mover was its only member. +func TestPools_UnopenablePoolKeepsTheMoverWhereItWas(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(0, func(id Identity, s Sizes, c *tls.Config) (driver.Conn, error) { + if id.Addr == "bad:9000" { + return nil, errors.New("malformed option") } - assert.False(t, base.equal(changed), "field %s must take part in equal", rt.Field(i).Name) - } + return rec.dial(id, s, c) + }) + acme := member("acme", "a:9000", 10, 5) + _, err := p.Reconcile([]Member{acme}) + require.NoError(t, err) + before := p.For("acme") + beforeConn := rec.latest("a:9000") + + moved := member("acme", "bad:9000", 10, 5) + moved.Params.HTTPPort = 8124 + stale, err := p.Reconcile([]Member{moved}) + require.ErrorContains(t, err, "clickhouse pool bad:9000 database db user u not opened for tenant acme: malformed option") + assert.ErrorContains(t, err, "tenant acme keeps its previous pool a:9000 database db user u") + assert.Empty(t, stale) + assert.Same(t, before, p.For("acme")) + assert.Equal(t, "http://a:8123", p.Target("acme").URL, "the previous Params, whole") + assert.False(t, beforeConn.closed.Load(), "the pool it kept is not released") +} + +func TestPools_TenantGoneReleasesItsPoolAfterGrace(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(0, rec.dial) + acme := member("acme", "a:9000", 10, 5) + acme.Params.QueryTimeout = 20 * time.Millisecond + _, err := p.Reconcile([]Member{acme, member("globex", "b:9000", 10, 5)}) + require.NoError(t, err) + c := rec.latest("a:9000") + globexPool := p.For("globex") + + stale, err := p.Reconcile([]Member{member("globex", "b:9000", 10, 5)}) + require.NoError(t, err) + assert.Empty(t, stale) + assert.Nil(t, p.For("acme")) + assert.Same(t, globexPool, p.For("globex")) + assert.False(t, c.closed.Load(), "released after the grace, not at once") + assert.Eventually(t, func() bool { return c.closed.Load() }, time.Second, 5*time.Millisecond) + + stale, err = p.Reconcile([]Member{acme, member("globex", "b:9000", 10, 5)}) + require.NoError(t, err) + assert.Equal(t, []tenant.ID{"acme"}, stale, "back after an absence: its cache is stale") + require.NotNil(t, p.For("acme")) + assert.NotSame(t, c, p.For("acme").conn()) +} + +// TestPools_CeilingRefusesBoot: at boot the pools must fit the ceiling +// together, and a refusal names the sum and the ceiling and leaves nothing +// open. +func TestPools_CeilingRefusesBoot(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(15, rec.dial) + _, err := p.Reconcile([]Member{member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5)}) + require.Error(t, err) + assert.ErrorContains(t, err, "not opened for tenant globex") + assert.ErrorContains(t, err, "clickhouse.max_open_conns 10") + assert.ErrorContains(t, err, "at 20, above clickhouse.max_total_conns 15") + assert.NotContains(t, err.Error(), "keeps its previous pool", "a tenant that was on no pool ends up on none") + assert.Nil(t, p.For("globex")) + require.NotNil(t, p.For("acme"), "the walk opened what fit") + require.NoError(t, p.Close()) + assert.True(t, rec.latest("a:9000").closed.Load(), "and boot's refusal closes it") + assert.Nil(t, p.For("acme")) +} + +// TestPools_CeilingRefusesAThirdTupleThenOpensIt: a reload's new tuple over +// the ceiling is not opened and the open pools are untouched; the next +// reload that frees the budget opens it. +func TestPools_CeilingRefusesAThirdTupleThenOpensIt(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(25, rec.dial) + acme, globex, initech := member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5), member("initech", "c:9000", 10, 5) + _, err := p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + acmePool, globexPool := p.For("acme"), p.For("globex") + + stale, err := p.Reconcile([]Member{acme, globex, initech}) + require.ErrorContains(t, err, "clickhouse pool c:9000 database db user u not opened for tenant initech") + assert.ErrorContains(t, err, "at 30, above clickhouse.max_total_conns 25") + assert.Empty(t, stale) + assert.Nil(t, p.For("initech"), "its tenant fails closed") + assert.Same(t, acmePool, p.For("acme")) + assert.Same(t, globexPool, p.For("globex")) + assert.Equal(t, 0, rec.count("c:9000"), "not even opened") + assert.Equal(t, 1, rec.count("a:9000")) + + acme.Params.MaxOpenConns = 5 + stale, err = p.Reconcile([]Member{acme, globex, initech}) + require.NoError(t, err) + assert.Equal(t, []tenant.ID{"initech"}, stale) + require.NotNil(t, p.For("initech")) + assert.Same(t, acmePool, p.For("acme")) + assert.Equal(t, 5, acmePool.Sizes().MaxOpenConns, "the shrink that made room") +} + +// TestPools_RefusedResizeKeepsTheSize: a reload that grows a pool above the +// ceiling is refused and the pool keeps its size — the rest of the tenant's +// Params apply — and the next reload retries. +func TestPools_RefusedResizeKeepsTheSize(t *testing.T) { + t.Parallel() + p := newPools(20, fakeDial) + acme, globex := member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5) + _, err := p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + + acme.Params.MaxOpenConns, acme.Params.HTTPPort = 15, 8124 + _, err = p.Reconcile([]Member{acme, globex}) + require.ErrorContains(t, err, "clickhouse pool a:9000 database db user u not resized for tenants [acme]") + assert.ErrorContains(t, err, "clickhouse.max_open_conns 15") + assert.ErrorContains(t, err, "at 25, above clickhouse.max_total_conns 20") + assert.ErrorContains(t, err, "it keeps 10") + assert.Equal(t, 10, p.For("acme").Sizes().MaxOpenConns) + assert.Equal(t, "http://a:8124", p.Target("acme").URL, "the HTTP wiring applied all the same") + + globex.Params.MaxOpenConns = 5 + _, err = p.Reconcile([]Member{acme, globex}) + require.NoError(t, err) + assert.Equal(t, 15, p.For("acme").Sizes().MaxOpenConns, "retried by the next reload") + assert.Equal(t, 5, p.For("globex").Sizes().MaxOpenConns) +} + +// TestPools_SharedGrowthIsOneRefusal: a shared pool whose members' ask grows +// past the ceiling is refused once, naming every member, not once per member. +func TestPools_SharedGrowthIsOneRefusal(t *testing.T) { + t.Parallel() + p := newPools(20, fakeDial) + acme, globex, initech := member("acme", "a:9000", 10, 5), member("globex", "a:9000", 10, 5), member("initech", "b:9000", 10, 5) + _, err := p.Reconcile([]Member{acme, globex, initech}) + require.NoError(t, err) + acme.Params.MaxOpenConns, globex.Params.MaxOpenConns = 12, 12 + _, err = p.Reconcile([]Member{acme, globex, initech}) + require.Error(t, err) + joined, ok := err.(interface{ Unwrap() []error }) + require.True(t, ok) + assert.Len(t, joined.Unwrap(), 1, "one refusal for the shared pool, not one per member") + assert.ErrorContains(t, err, "for tenants [acme globex]") + assert.Equal(t, 10, p.For("acme").Sizes().MaxOpenConns) +} + +// TestPools_RefusedMoveKeepsThePreviousPool: a tenant whose new tuple the +// ceiling refuses stays where it was, with the Params it had — #603's +// keep-the-previous-wiring rule — and the next reload retries. +func TestPools_RefusedMoveKeepsThePreviousPool(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(15, rec.dial) + acme := member("acme", "a:9000", 10, 5) + acme.Params.QueryTimeout = 10 * time.Millisecond + _, err := p.Reconcile([]Member{acme}) + require.NoError(t, err) + before := p.For("acme") + beforeConn := rec.latest("a:9000") + + moved := member("acme", "b:9000", 20, 5) + moved.Params.HTTPPort = 8124 + stale, err := p.Reconcile([]Member{moved}) + require.ErrorContains(t, err, "clickhouse pool b:9000 database db user u not opened for tenant acme") + assert.ErrorContains(t, err, "tenant acme keeps its previous pool a:9000 database db user u") + assert.Empty(t, stale, "a refused move stays on the tables it had") + assert.Same(t, before, p.For("acme")) + assert.Equal(t, "http://a:8123", p.Target("acme").URL, "the previous Params, whole") + assert.False(t, beforeConn.closed.Load()) + assert.Equal(t, 0, rec.count("b:9000")) + + moved.Params.MaxOpenConns = 15 + _, err = p.Reconcile([]Member{moved}) + require.NoError(t, err) + assert.Equal(t, "b:9000", p.For("acme").Identity().Addr, "the next reload that fits moves it") + assert.Equal(t, "http://b:8124", p.Target("acme").URL) + assert.Eventually(t, func() bool { return beforeConn.closed.Load() }, time.Second, 5*time.Millisecond) +} + +// TestPools_MoveAtTheCeilingIsAllowed: the pool a move leaves frees its +// budget for the pool the move opens, so a flat directory at the ceiling can +// still change its address. +func TestPools_MoveAtTheCeilingIsAllowed(t *testing.T) { + t.Parallel() + p := newPools(10, fakeDial) + _, err := p.Reconcile([]Member{member("0", "a:9000", 10, 5)}) + require.NoError(t, err) + _, err = p.Reconcile([]Member{member("0", "b:9000", 10, 5)}) + require.NoError(t, err) + assert.Equal(t, "b:9000", p.For("0").Identity().Addr) +} + +// TestPools_UnreadableCertificateRefusesTheTuple: a tls block whose files +// cannot be read refuses that tuple like the ceiling does — boot refuses, a +// reload leaves the tenant on its previous pool. +func TestPools_UnreadableCertificateRefusesTheTuple(t *testing.T) { + t.Parallel() + p := newPools(0, fakeDial) + acme := member("acme", "a:9000", 10, 5) + _, err := p.Reconcile([]Member{acme}) + require.NoError(t, err) + before := p.For("acme") + + acme.Params.TLS = TLS{Enabled: true, CAFile: "/nowhere/ca.pem"} + _, err = p.Reconcile([]Member{acme}) + require.ErrorContains(t, err, "clickhouse.tls.ca_file") + assert.ErrorContains(t, err, "/nowhere/ca.pem") + assert.ErrorContains(t, err, "keeps its previous pool") + assert.Same(t, before, p.For("acme")) + + _, err = newPools(0, fakeDial).Reconcile([]Member{acme}) + require.ErrorContains(t, err, "not opened for tenant acme") +} + +func TestPools_PingFirstSuccessWins(t *testing.T) { + t.Parallel() + t.Run("none open", func(t *testing.T) { + t.Parallel() + require.ErrorContains(t, newPools(0, fakeDial).Ping(context.Background()), "no ClickHouse pool is open") + }) + t.Run("every pool down names each", func(t *testing.T) { + t.Parallel() + rec := &dialRecorder{ping: func(context.Context) error { return errors.New("connection refused") }} + p := newPools(0, rec.dial) + _, err := p.Reconcile([]Member{member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5)}) + require.NoError(t, err) + err = p.Ping(context.Background()) + require.Error(t, err) + assert.ErrorContains(t, err, "a:9000 database db user u: connection refused") + assert.ErrorContains(t, err, "b:9000 database db user u: connection refused") + }) + t.Run("one answering pool is ready, whatever the others do", func(t *testing.T) { + t.Parallel() + rec := &dialRecorder{ping: func(ctx context.Context) error { + // A host that does not answer: the driver would wait for its dial + // timeout; here, until the probe gives up on it. + <-ctx.Done() + return ctx.Err() + }} + p := newPools(0, rec.dial) + _, err := p.Reconcile([]Member{member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5)}) + require.NoError(t, err) + rec.latest("b:9000").ping = nil + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + start := time.Now() + require.NoError(t, p.Ping(ctx)) + assert.Less(t, time.Since(start), time.Second, "answered at the first success, not after the hung pool") + }) +} + +func TestPools_CloseClosesEveryPool(t *testing.T) { + t.Parallel() + rec := &dialRecorder{} + p := newPools(0, rec.dial) + _, err := p.Reconcile([]Member{member("acme", "a:9000", 10, 5), member("globex", "b:9000", 10, 5)}) + require.NoError(t, err) + require.NoError(t, p.Close()) + assert.True(t, rec.latest("a:9000").closed.Load()) + assert.True(t, rec.latest("b:9000").closed.Load()) + assert.Nil(t, p.For("acme")) + assert.Error(t, p.Ping(context.Background()), "nothing left to ping") } // closeCounter is a transport that counts CloseIdleConnections, which @@ -283,15 +721,12 @@ func (c *closeCounter) RoundTrip(*http.Request) (*http.Response, error) { } func (c *closeCounter) CloseIdleConnections() { c.closed.Add(1) } -func TestHTTPClients_RebuildOnlyWhenTheTLSConfigChanges(t *testing.T) { +func TestHTTPClients_OneClientPerTLSConfig(t *testing.T) { t.Parallel() var builds int - var transports []*closeCounter clients := NewHTTPClients(func(*tls.Config) *http.Client { builds++ - rt := &closeCounter{} - transports = append(transports, rt) - return &http.Client{Transport: rt} + return &http.Client{Transport: &closeCounter{}} }) plain := Target{} first := clients.For(plain) @@ -302,9 +737,9 @@ func TestHTTPClients_RebuildOnlyWhenTheTLSConfigChanges(t *testing.T) { second := clients.For(Target{TLS: cfg}) assert.NotSame(t, first, second) assert.Equal(t, 2, builds) - assert.Equal(t, int32(1), transports[0].closed.Load(), "the replaced client's idle connections are closed") assert.Same(t, second, clients.For(Target{TLS: cfg})) - assert.Equal(t, int32(0), transports[1].closed.Load()) + assert.Same(t, first, clients.For(plain), "tenants on different configs alternate without rebuilding") + assert.Equal(t, 2, builds) } // TestHTTPClients_HandsEachClientItsOwnTLSConfig: net/http rewrites a diff --git a/internal/discovery/discovery.go b/internal/discovery/discovery.go index 867f38b1..f7bfdbd1 100644 --- a/internal/discovery/discovery.go +++ b/internal/discovery/discovery.go @@ -2,15 +2,39 @@ package discovery import ( "context" + "errors" "fmt" "log/slog" + "math/rand/v2" "sync" + "sync/atomic" "time" "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/Wave-RF/WaveHouse/internal/tenant" "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/metric" +) + +var ( + // ErrNotLoaded is Lookup's answer before the first successful Refresh: + // the table may well exist, the registry just cannot say yet. + ErrNotLoaded = errors.New("schema not loaded yet") + // ErrUnknownTable is Lookup's answer for a table the loaded schema lacks. + ErrUnknownTable = errors.New("unknown table") + // ErrNoConnection is Refresh's answer when the connection getter yields + // none: the tenant has no open ClickHouse pool. + ErrNoConnection = errors.New("no ClickHouse connection") +) + +// refreshFailures counts the refresh loops' failed attempts per tenant: a +// tenant's ClickHouse outage after its first discovery is this counter and a +// log line, never a probe failure. +var refreshFailures, _ = otel.Meter("wavehouse-discovery").Int64Counter( + "wavehouse_schema_refresh_failures_total", + metric.WithDescription("Failed schema refresh attempts of the boot retry and auto-refresh loops, per tenant"), ) // Column describes a single ClickHouse column. @@ -171,34 +195,48 @@ func (ts *TableSchema) InsertableColumnNames() []string { return columnNames(ts.InsertableColumns()) } -// SchemaRegistry discovers and caches ClickHouse table schemas. +// SchemaRegistry discovers and caches one tenant's ClickHouse table schemas. type SchemaRegistry struct { - conn driver.Conn - // database supplies the database to discover from on each Refresh, so a - // ClickHouse reconfigure that changes clickhouse.database is honored by - // the next refresh (chconn.Manager.Database in production). - database func() string + // source supplies the tenant's connection and the database to discover + // from, read together once per Refresh, so a settings reload that moves + // the tenant to another pool or database is honored by the next refresh + // (the tenant's chconn.Pools entry in production). + source Source // tenant is whose tables the registry discovers. tenant tenant.ID // refreshInterval supplies the tenant's auto-refresh interval on each // tick, so a settings reload retunes the cadence without restarting the // loop (settings.Store.SchemaRefreshInterval in production). refreshInterval func(tenant.ID) time.Duration - mu sync.RWMutex - tables map[string]*TableSchema + // firstTick picks how long StartAutoRefresh waits before its first + // refresh, within the interval; rand.N, substituted by tests. + firstTick func(interval time.Duration) time.Duration + // loaded is set by the first successful Refresh and never cleared: the + // line between "no schema known yet" and "this table is unknown". + loaded atomic.Bool + mu sync.RWMutex + tables map[string]*TableSchema // serverVersion is the ClickHouse version string from the last successful // Refresh, guarded by mu alongside tables. serverVersion string } +// Source yields a tenant's connection and the database it discovers from, +// one snapshot: the database is the one the connection's own pool was +// opened for, so the schema discovered always describes the database the +// tenant's queries and inserts run against. A nil connection is a tenant +// with no open pool, which Refresh reports as ErrNoConnection. +type Source func() (driver.Conn, string) + // NewSchemaRegistry creates the registry of tenant id, which discovers -// schemas from system.columns. -func NewSchemaRegistry(conn driver.Conn, database func() string, id tenant.ID, refreshInterval func(tenant.ID) time.Duration) *SchemaRegistry { +// schemas from system.columns over the connection and database source +// yields. +func NewSchemaRegistry(source Source, id tenant.ID, refreshInterval func(tenant.ID) time.Duration) *SchemaRegistry { return &SchemaRegistry{ - conn: conn, - database: database, + source: source, tenant: id, refreshInterval: refreshInterval, + firstTick: rand.N[time.Duration], tables: make(map[string]*TableSchema), } } @@ -211,26 +249,34 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { ctx, span := tracer.Start(ctx, "SchemaRegistry.Refresh") defer span.End() + // One connection and one database per refresh, read together: a reload + // that moves the tenant to another pool or database applies to the NEXT + // refresh, so every query of this one runs against the same server and + // database — reading the database twice would let a reconfigure land + // between the system.columns and system.tables queries and attach DDL + // from the new database to same-named schemas discovered from the old + // one. (The pool's own connection can still be swapped underneath + // mid-refresh by a resize, which stays on the same server.) + conn, database := sr.source() + if conn == nil { + return fmt.Errorf("%w for tenant %s", ErrNoConnection, sr.tenant) + } + // ClickHouse interprets zone-less timestamp strings in the server's default // zone; canonicalization applies the same rule so the instant never changes (#372). var tzName string - if err := sr.conn.QueryRow(ctx, "SELECT timezone()").Scan(&tzName); err != nil { + if err := conn.QueryRow(ctx, "SELECT timezone()").Scan(&tzName); err != nil { return fmt.Errorf("query server timezone: %w", err) } // The server version is metadata about the schema source, probed on the same // refresh so a stale version cannot outlive the schemas it describes. // - // It is NOT a same-server guarantee: chconn.Manager resolves the connection - // per call, so a reload changing clickhouse.addr between this probe and the - // system.columns query below would pair a version from one server with - // schemas from another. Narrow, self-correcting on the next refresh, and - // shared with the timezone probe above — but do not read this as atomic. - // - // Nor is the read side: ServerVersion() and Get()/List() take separate - // RLocks, so a caller doing both across a refresh boundary pairs version N - // with schemas N+1. They are published together; nothing reads them together. + // The read side is not atomic: ServerVersion() and Get()/List() take + // separate RLocks, so a caller doing both across a refresh boundary pairs + // version N with schemas N+1. They are published together; nothing reads + // them together. var serverVersion string - if err := sr.conn.QueryRow(ctx, "SELECT version()").Scan(&serverVersion); err != nil { + if err := conn.QueryRow(ctx, "SELECT version()").Scan(&serverVersion); err != nil { return fmt.Errorf("query server version: %w", err) } @@ -244,14 +290,7 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { "timezone", tzName, "error", err) } - // One database per refresh. `database` is a live getter so a ClickHouse - // reconfigure is honored on the NEXT refresh — reading it twice would let a - // reconfigure land between the system.columns and system.tables queries and - // attach DDL from the new database to same-named schemas discovered from the - // old one. - database := sr.database() - - rows, err := sr.conn.Query(ctx, + rows, err := conn.Query(ctx, `SELECT table, name, type, default_kind, default_expression, position FROM system.columns WHERE database = ? @@ -294,7 +333,7 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { return fmt.Errorf("iterate system.columns: %w", err) } - if err := sr.attachDDL(ctx, database, tables); err != nil { + if err := attachDDL(ctx, conn, database, tables); err != nil { return err } @@ -307,7 +346,8 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { sr.tables = tables sr.serverVersion = serverVersion sr.mu.Unlock() - slog.InfoContext(ctx, "schema registry refreshed", "tables", len(tables), "server_tz", tzName, "server_version", serverVersion) + sr.loaded.Store(true) + slog.InfoContext(ctx, "schema registry refreshed", "tenant", sr.tenant, "tables", len(tables), "server_tz", tzName, "server_version", serverVersion) return nil } @@ -317,8 +357,8 @@ func (sr *SchemaRegistry) Refresh(ctx context.Context) error { // scan didn't return, and one created between the two queries — is skipped rather // than added: a TableSchema with no columns is not a schema, and the two queries // are not a snapshot. -func (sr *SchemaRegistry) attachDDL(ctx context.Context, database string, tables map[string]*TableSchema) error { - rows, err := sr.conn.Query(ctx, +func attachDDL(ctx context.Context, conn driver.Conn, database string, tables map[string]*TableSchema) error { + rows, err := conn.Query(ctx, `SELECT name, create_table_query FROM system.tables WHERE database = ? @@ -356,13 +396,34 @@ func (sr *SchemaRegistry) ServerVersion() string { return sr.serverVersion } -// Get returns the schema for a table, or nil if not found. +// Get returns the schema for a table, or nil if not found — before the first +// refresh as much as for a table the schema lacks, which is the fail-closed +// reading the stream hub wants. A handler that answers 404 uses Lookup. func (sr *SchemaRegistry) Get(name string) *TableSchema { sr.mu.RLock() defer sr.mu.RUnlock() return sr.tables[name] } +// Loaded reports whether a Refresh has ever succeeded: until one has, the +// registry cannot tell an unknown table from one it has not seen yet. +func (sr *SchemaRegistry) Loaded() bool { return sr.loaded.Load() } + +// Lookup is Get for a caller that answers the two misses differently: +// ErrNotLoaded before the first successful Refresh (the table may exist — +// a 503 with Retry-After, not a 404), ErrUnknownTable for a table the loaded +// schema lacks. +func (sr *SchemaRegistry) Lookup(name string) (*TableSchema, error) { + if !sr.Loaded() { + return nil, ErrNotLoaded + } + ts := sr.Get(name) + if ts == nil { + return nil, fmt.Errorf("%w: %s", ErrUnknownTable, name) + } + return ts, nil +} + // List returns all discovered table schemas. func (sr *SchemaRegistry) List() []*TableSchema { sr.mu.RLock() @@ -403,7 +464,7 @@ func (sr *SchemaRegistry) RetryRefresh(ctx context.Context, initialBackoff, maxB for { if err := sr.Refresh(ctx); err == nil { return nil - } else if ctx.Err() == nil && onAttempt != nil { + } else if ctx.Err() == nil { // Skip the callback when Refresh's error is just a downstream // reflection of ctx cancellation — that's a shutdown signal, // not a real diagnostic. Without this guard, a clean shutdown @@ -412,7 +473,10 @@ func (sr *SchemaRegistry) RetryRefresh(ctx context.Context, initialBackoff, maxB // "schema discovery: context canceled" // — visible to anyone curl'ing /livez during the shutdown // window. Not wrong, just noise. - onAttempt(err) + sr.countFailure(ctx) + if onAttempt != nil { + onAttempt(err) + } } select { case <-ctx.Done(): @@ -431,10 +495,13 @@ func (sr *SchemaRegistry) RetryRefresh(ctx context.Context, initialBackoff, maxB const unresolvedRefreshInterval = time.Minute // StartAutoRefresh runs a background goroutine that refreshes schemas -// at the configured interval. Blocks until ctx is cancelled. The interval is -// re-read after every tick, so a changed setting applies from the next cycle -// — an in-flight wait finishes at the old cadence rather than resetting, -// which keeps a reload from ever deferring an imminent refresh. +// at the configured interval. Blocks until ctx is cancelled. The first +// refresh fires at a random point within the interval, so tenants adopted +// together do not refresh together; the cadence runs from there. The +// interval is re-read after every tick, so a changed setting applies from +// the next cycle — an in-flight wait finishes at the old cadence rather +// than resetting, which keeps a reload from ever deferring an imminent +// refresh. // // A validated setting is at least a second, so a non-positive interval is a // tenant the settings registry could not resolve, read as the zero value. @@ -446,24 +513,35 @@ func (sr *SchemaRegistry) StartAutoRefresh(ctx context.Context) { if interval <= 0 { interval = unresolvedRefreshInterval } + first := time.NewTimer(sr.firstTick(interval)) + defer first.Stop() ticker := time.NewTicker(interval) defer ticker.Stop() for { select { case <-ctx.Done(): return + case <-first.C: + // The cadence starts from the first refresh, not from the loop. + ticker.Reset(interval) case <-ticker.C: - if err := sr.Refresh(ctx); err != nil { - slog.ErrorContext(ctx, "schema auto-refresh failed", "error", err) - } - if next := sr.refreshInterval(sr.tenant); next > 0 && next != interval { - interval = next - ticker.Reset(interval) - } + } + if err := sr.Refresh(ctx); err != nil { + sr.countFailure(ctx) + slog.ErrorContext(ctx, "schema auto-refresh failed", "tenant", sr.tenant, "error", err) + } + if next := sr.refreshInterval(sr.tenant); next > 0 && next != interval { + interval = next + ticker.Reset(interval) } } } +// countFailure bumps the tenant's refresh-failure counter. +func (sr *SchemaRegistry) countFailure(ctx context.Context) { + refreshFailures.Add(ctx, 1, metric.WithAttributes(attribute.String("tenant", sr.tenant.String()))) +} + // isNullable checks if a ClickHouse type string is Nullable. func isNullable(chType string) bool { return len(chType) > 9 && chType[:9] == "Nullable(" diff --git a/internal/discovery/discovery_test.go b/internal/discovery/discovery_test.go index a009b953..6e786415 100644 --- a/internal/discovery/discovery_test.go +++ b/internal/discovery/discovery_test.go @@ -18,6 +18,10 @@ import ( "github.com/Wave-RF/WaveHouse/internal/testutil/logtest" ) +// sourceOf wraps a fake connection in the Source NewSchemaRegistry takes, +// with the database "test". +func sourceOf(c driver.Conn) Source { return func() (driver.Conn, string) { return c, "test" } } + func TestTableSchema_ColumnNames(t *testing.T) { t.Parallel() tests := []struct { @@ -50,9 +54,10 @@ func TestTableSchema_ColumnNames(t *testing.T) { func TestNewSchemaRegistry_ConstructorDefaults(t *testing.T) { t.Parallel() - sr := NewSchemaRegistry(nil, func() string { return "wavehouse" }, tenant.Default, func(tenant.ID) time.Duration { return 30 * time.Second }) + sr := NewSchemaRegistry(func() (driver.Conn, string) { return nil, "wavehouse" }, tenant.Default, func(tenant.ID) time.Duration { return 30 * time.Second }) require.NotNil(t, sr) - assert.Equal(t, "wavehouse", sr.database()) + _, db := sr.source() + assert.Equal(t, "wavehouse", db) assert.Equal(t, 30*time.Second, sr.refreshInterval(tenant.Default)) assert.NotNil(t, sr.tables) assert.Empty(t, sr.List()) @@ -78,7 +83,7 @@ func TestRefresh_PopulatesAndLookups(t *testing.T) { {"ghost", "CREATE TABLE test.ghost (`x` String) ENGINE = MergeTree"}, }, } - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) clicks := sr.Get("clicks") @@ -98,6 +103,81 @@ func TestRefresh_PopulatesAndLookups(t *testing.T) { assert.Equal(t, "25.3.1.1", sr.ServerVersion()) } +// TestLookup_NotLoadedThenUnknown: until a Refresh has succeeded, Lookup +// cannot tell an unknown table from one it has not seen, and says so with +// ErrNotLoaded — the handlers' 503 — where Get would answer nil; once loaded, +// a table the schema lacks is ErrUnknownTable, the 404. +func TestLookup_NotLoadedThenUnknown(t *testing.T) { + t.Parallel() + conn := &fakeConn{columns: []fakeColumn{{table: "clicks", name: "id", chType: "String", position: 1}}} + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + assert.False(t, sr.Loaded()) + ts, err := sr.Lookup("clicks") + require.ErrorIs(t, err, ErrNotLoaded) + assert.Nil(t, ts) + assert.Nil(t, sr.Get("clicks"), "Get keeps its fail-closed nil") + + require.NoError(t, sr.Refresh(context.Background())) + assert.True(t, sr.Loaded()) + ts, err = sr.Lookup("clicks") + require.NoError(t, err) + assert.Equal(t, "clicks", ts.Name) + ts, err = sr.Lookup("missing") + require.ErrorIs(t, err, ErrUnknownTable) + assert.ErrorContains(t, err, "missing") + assert.Nil(t, ts) + + // A later failed refresh keeps the prior schema, and with it the loaded state. + conn.versionErr = errors.New("code: 497, not enough privileges") + require.Error(t, sr.Refresh(context.Background())) + assert.True(t, sr.Loaded()) + _, err = sr.Lookup("clicks") + require.NoError(t, err) +} + +// TestRefresh_NoConnection: a source yielding no connection — a tenant with +// no open pool — is ErrNoConnection naming the tenant, not a panic, and the +// registry stays unloaded until the source yields one. +func TestRefresh_NoConnection(t *testing.T) { + t.Parallel() + var conn driver.Conn + sr := NewSchemaRegistry(func() (driver.Conn, string) { return conn, "test" }, "acme", func(tenant.ID) time.Duration { return time.Hour }) + err := sr.Refresh(context.Background()) + require.ErrorIs(t, err, ErrNoConnection) + assert.ErrorContains(t, err, "acme") + assert.False(t, sr.Loaded()) + + conn = &fakeConn{} + require.NoError(t, sr.Refresh(context.Background())) + assert.True(t, sr.Loaded()) +} + +// TestRefresh_ConnectionReadOncePerRefresh: like the database, the connection +// is resolved once per refresh, so every query of one refresh runs against +// the same server even if a reload repoints the tenant halfway through. +func TestRefresh_ConnectionReadOncePerRefresh(t *testing.T) { + t.Parallel() + first := &fakeConn{columns: []fakeColumn{{table: "clicks", name: "id", chType: "String", position: 1}}} + second := &fakeConn{} + var reads atomic.Int32 + conn := func() driver.Conn { + if reads.Add(1) == 1 { + return first + } + return second + } + sr := NewSchemaRegistry(func() (driver.Conn, string) { return conn(), "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, int32(1), reads.Load(), "one read per refresh") + assert.Equal(t, int32(1), first.calls.Load()) + assert.Zero(t, second.calls.Load(), "the repointed connection is for the next refresh") + require.NotNil(t, sr.Get("clicks")) + + require.NoError(t, sr.Refresh(context.Background())) + assert.Equal(t, int32(1), second.calls.Load()) + assert.Nil(t, sr.Get("clicks"), "the next refresh reads the new connection") +} + // TestRefresh_DDLIsNotSerialized: the schema endpoint marshals TableSchema // straight to the client, and an external-engine table renders its wiring in // create_table_query — endpoint, bucket, username, access key id. ClickHouse @@ -111,7 +191,7 @@ func TestRefresh_DDLIsNotSerialized(t *testing.T) { // topology is not. The field is withheld for the topology. tables: [][2]string{{"clicks", "CREATE TABLE test.clicks (`id` String) ENGINE = S3('https://acme-private.s3.amazonaws.com/events.csv', 'AKIAEXAMPLEKEY', '[HIDDEN]', 'CSV')"}}, } - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) encoded, err := json.Marshal(sr.Get("clicks")) @@ -131,7 +211,7 @@ func TestRefresh_ServerVersionQueryFails(t *testing.T) { version: "25.3.2.2", columns: []fakeColumn{{table: "clicks", name: "id", chType: "String", position: 1}}, } - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) require.Equal(t, "25.3.2.2", sr.ServerVersion()) @@ -152,7 +232,7 @@ func TestRefresh_TablesQueryFails(t *testing.T) { columns: []fakeColumn{{table: "clicks", name: "id", chType: "String", position: 1}}, tables: [][2]string{{"clicks", "CREATE TABLE test.clicks (id String) ENGINE = MergeTree"}}, } - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) require.NotNil(t, sr.Get("clicks")) require.NotEmpty(t, sr.Get("clicks").DDL) @@ -354,7 +434,7 @@ func (*fakeTableRows) Err() error { return nil } func newFakeRegistry(t *testing.T, errs []error) (*SchemaRegistry, *fakeConn) { t.Helper() conn := &fakeConn{errsThenSuccess: errs} - return NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }), conn + return NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }), conn } // TestRefresh_UnresolvableServerTimezone_NotFatal: an unresolvable server zone @@ -362,7 +442,7 @@ func newFakeRegistry(t *testing.T, errs []error) (*SchemaRegistry, *fakeConn) { func TestRefresh_UnresolvableServerTimezone_NotFatal(t *testing.T) { t.Parallel() conn := &fakeConn{tz: "Not/AZone"} - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) } @@ -376,7 +456,7 @@ func TestRefresh_RowsIterationError_Fails(t *testing.T) { columns: []fakeColumn{{table: "events", name: "id", chType: "String", position: 1}}, iterErr: errors.New("network drop mid-stream"), } - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) err := sr.Refresh(context.Background()) require.ErrorContains(t, err, "network drop mid-stream") require.Nil(t, sr.Get("events"), "truncated scan must not be published") @@ -484,7 +564,7 @@ func TestRetryRefresh_DoesNotFireOnAttemptDuringCancel(t *testing.T) { // looks real, but ctx.Err() reveals we're shutting down anyway. conn := &fakeConn{errsThenSuccess: []error{errors.New("transient")}} sr, _ := newFakeRegistry(t, nil) - sr.conn = conn // override the no-error conn from newFakeRegistry + sr.source = sourceOf(conn) // override the no-error conn from newFakeRegistry ctx, cancel := context.WithCancel(context.Background()) cancel() // already cancelled before RetryRefresh starts @@ -575,7 +655,7 @@ func TestClampBackoff(t *testing.T) { func TestStartAutoRefresh_ExitsOnContextCancel(t *testing.T) { t.Parallel() // Long interval so the ticker never fires before cancel. - sr := NewSchemaRegistry(nil, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(nil), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) ctx, cancel := context.WithCancel(context.Background()) @@ -594,6 +674,39 @@ func TestStartAutoRefresh_ExitsOnContextCancel(t *testing.T) { } } +// TestStartAutoRefresh_FirstTickWithinTheInterval: the first refresh fires +// at a point within the interval rather than a full interval in, so tenants +// adopted together do not refresh together; the cadence runs from there. +func TestStartAutoRefresh_FirstTickWithinTheInterval(t *testing.T) { + t.Parallel() + conn := &fakeConn{} + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + for range 50 { + offset := sr.firstTick(time.Hour) + assert.GreaterOrEqual(t, offset, time.Duration(0)) + assert.Less(t, offset, time.Hour) + } + + // Pinned at zero: the first refresh fires at once, an hour before the + // ticker would have. + var asked []time.Duration + sr.firstTick = func(interval time.Duration) time.Duration { + asked = append(asked, interval) + return 0 + } + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan struct{}) + go func() { + defer close(done) + sr.StartAutoRefresh(ctx) + }() + assert.Eventually(t, func() bool { return conn.calls.Load() == 1 }, 2*time.Second, time.Millisecond) + cancel() + <-done + assert.Equal(t, []time.Duration{time.Hour}, asked, "the offset is drawn from the interval") + assert.Equal(t, int32(1), conn.calls.Load(), "and the ticker did not add a second refresh") +} + // TestStartAutoRefresh_UnresolvedIntervalDoesNotPanic pins the zero interval a // settings-registry miss reads as: time.NewTicker and Ticker.Reset panic on a // non-positive duration, which would take the process down. The loop keeps @@ -616,7 +729,7 @@ func TestStartAutoRefresh_UnresolvedIntervalDoesNotPanic(t *testing.T) { i := int(reads.Add(1)) - 1 return tt.intervals[min(i, len(tt.intervals)-1)] } - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, interval) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, interval) ctx, cancel := context.WithCancel(context.Background()) done := make(chan struct{}) @@ -657,7 +770,7 @@ func TestStartAutoRefresh_LogsAndContinuesOnError(t *testing.T) { conn := &fakeConn{errsThenSuccess: errs} buf := logtest.Capture(t, slog.LevelDebug) - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return 5 * time.Millisecond }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return 5 * time.Millisecond }) ctx, cancel := context.WithCancel(context.Background()) done := make(chan struct{}) @@ -711,7 +824,7 @@ func TestRefresh_DatabaseSnapshottedForWholeRefresh(t *testing.T) { } return "old" } - sr := NewSchemaRegistry(conn, db, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(func() (driver.Conn, string) { return conn, db() }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) require.Len(t, seen, 2, "both scans should be parameterised by a database") @@ -784,7 +897,7 @@ func TestRefresh_CapturesDefaultKind(t *testing.T) { {table: "t", name: "mat", chType: "String", defaultKind: "MATERIALIZED", defaultExpr: "concat('m', id)", position: 2}, {table: "t", name: "page", chType: "String", defaultKind: "DEFAULT", defaultExpr: "'/'", position: 3}, }} - sr := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + sr := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, sr.Refresh(context.Background())) ts := sr.Get("t") diff --git a/internal/discovery/timestamp_test.go b/internal/discovery/timestamp_test.go index 4f96e49c..669ba620 100644 --- a/internal/discovery/timestamp_test.go +++ b/internal/discovery/timestamp_test.go @@ -301,7 +301,7 @@ func TestResolveTimestampSpecs(t *testing.T) { func TestRefresh_PrecomputesSpecs(t *testing.T) { t.Parallel() conn := &fakeConn{columns: []fakeColumn{{table: "t", name: "ts", chType: "DateTime", position: 1}}} - reg := NewSchemaRegistry(conn, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + reg := NewSchemaRegistry(sourceOf(conn), tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, reg.Refresh(context.Background())) col := reg.Get("t").Columns[0] require.NotNil(t, col.tsSpec) diff --git a/internal/ingest/worker.go b/internal/ingest/worker.go index 36bb9655..4c1ff846 100644 --- a/internal/ingest/worker.go +++ b/internal/ingest/worker.go @@ -62,10 +62,11 @@ type IngestWorker struct { // replaced when a reload changes it (chconn.HTTPClients). clients *chconn.HTTPClients cache cache.Cache - // target resolves the ClickHouse HTTP wiring per insert - // (chconn.Manager.Target in production) so a settings reload that - // re-points ClickHouse applies to the next flush. - target func() chconn.Target + // target resolves a tenant's ClickHouse HTTP wiring per insert + // (chconn.Pools.Target in production) so a settings reload that + // re-points the tenant applies to the next flush; the zero Target is a + // tenant on no pool, whose insert fails like an unreachable one. + target func(tenant.ID) chconn.Target maxBatch int maxWait time.Duration // dlqEnabled reports, per tenant table, whether a row that still fails @@ -132,7 +133,7 @@ const ( // still the caller's to call. func StartIngestWorker( ctx context.Context, queue Queue, cache cache.Cache, - target func() chconn.Target, + target func(tenant.ID) chconn.Target, dlqEnabled func(id tenant.ID, table string) bool, ) (stop func(context.Context) error, failed <-chan error, err error) { if queue == nil { @@ -622,7 +623,13 @@ func (w *IngestWorker) insertToClickHouse(ctx context.Context, tableName string, quoted[i] = chsql.QuoteIdent(c) } - t := w.target() + // A group is one (tenant, table) batch's, so its first row names the + // tenant whose ClickHouse takes the insert. + id := msgs[0].tenant + t := w.target(id) + if t.URL == "" { + return fmt.Errorf("no ClickHouse connection is open for tenant %s", id) + } q := url.Values{} q.Set("database", t.Database) q.Set("param_target_table", tableName) diff --git a/internal/ingest/worker_test.go b/internal/ingest/worker_test.go index e6703437..4c300e45 100644 --- a/internal/ingest/worker_test.go +++ b/internal/ingest/worker_test.go @@ -49,7 +49,7 @@ func newTestWorker(rt http.RoundTripper) (*IngestWorker, *testutil.MockPublisher failed: make(chan error, 1), clients: chconn.NewHTTPClients(func(*tls.Config) *http.Client { return &http.Client{Transport: rt} }), cache: cache, - target: func() chconn.Target { + target: func(tenant.ID) chconn.Target { return chconn.Target{URL: "http://test-clickhouse:8123", Username: "test_user", Password: "test_pass", Database: "test_db"} }, } @@ -136,7 +136,7 @@ func TestStartIngestWorker_Validation(t *testing.T) { t.Parallel() q, c := tt.setup(t) _, _, err := StartIngestWorker(context.Background(), q, c, - func() chconn.Target { return chconn.Target{URL: "http://localhost:8123"} }, nil) + func(tenant.ID) chconn.Target { return chconn.Target{URL: "http://localhost:8123"} }, nil) require.Error(t, err) assert.Contains(t, err.Error(), tt.wantErrSub) }) @@ -197,7 +197,7 @@ func TestStartIngestWorker_EndToEnd(t *testing.T) { dlq: emb, clients: chconn.NewHTTPClients(ingestHTTPClient), cache: cache, - target: func() chconn.Target { + target: func(tenant.ID) chconn.Target { return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db"} }, maxBatch: defaultMaxBatch, @@ -269,7 +269,7 @@ func TestStartIngestWorker_StopFunc_RespectsShutdownDeadline(t *testing.T) { t.Cleanup(cancel) stopFn, _, err := StartIngestWorker(ctx, emb, &testutil.MockCache{}, - func() chconn.Target { + func(tenant.ID) chconn.Target { return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db"} }, nil) require.NoError(t, err) @@ -305,7 +305,7 @@ func TestStartIngestWorker_StopFunc_CleanShutdown(t *testing.T) { // chURL is never dialed: with no messages there is no flush, so a dummy // host/port is fine. stopFn, _, err := StartIngestWorker(context.Background(), emb, &testutil.MockCache{}, - func() chconn.Target { return chconn.Target{URL: "http://localhost:8123"} }, nil) + func(tenant.ID) chconn.Target { return chconn.Target{URL: "http://localhost:8123"} }, nil) require.NoError(t, err) // Nothing to flush, so shutdown drains immediately and returns nil before the @@ -1130,7 +1130,7 @@ func TestDispatchLoop_PerTableBatching_NoCrossTableContamination(t *testing.T) { dlq: emb, clients: chconn.NewHTTPClients(ingestHTTPClient), cache: &testutil.MockCache{}, - target: func() chconn.Target { + target: func(tenant.ID) chconn.Target { return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db"} }, maxBatch: maxBatch, @@ -1225,7 +1225,7 @@ func TestDispatchLoop_PartialBatchWaitsForOwnTrigger(t *testing.T) { dlq: emb, clients: chconn.NewHTTPClients(ingestHTTPClient), cache: &testutil.MockCache{}, - target: func() chconn.Target { + target: func(tenant.ID) chconn.Target { return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db"} }, maxBatch: maxBatch, @@ -1784,8 +1784,9 @@ func TestFlushTable_DLQSwitchIsTheRowsTenants(t *testing.T) { // Two tenants, one table name: each tenant's rows batch on their own, so an // INSERT never mixes tenants — what a per-tenant ClickHouse target and cache // namespace (stories 6 and 8) rely on — and each batch invalidates its own -// tenant's namespaces. Published interleaved, so batching by table alone -// would put both tenants' first rows in one INSERT. +// tenant's namespaces and goes to its own tenant's target. Published +// interleaved, so batching by table alone would put both tenants' first rows +// in one INSERT. func TestDispatchLoop_BatchesPerTenantTable(t *testing.T) { t.Parallel() const maxBatch = 2 @@ -1794,15 +1795,15 @@ func TestDispatchLoop_BatchesPerTenantTable(t *testing.T) { require.NoError(t, err) t.Cleanup(func() { _ = emb.Close() }) - // CH stub: record each INSERT's body. + // CH stub: record each INSERT's body under the database it named. var ( mu sync.Mutex - bodies []string + bodies = map[string]string{} ) chSrv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { body, _ := io.ReadAll(r.Body) mu.Lock() - bodies = append(bodies, string(body)) + bodies[r.URL.Query().Get("database")] = string(body) mu.Unlock() w.WriteHeader(http.StatusOK) })) @@ -1820,8 +1821,9 @@ func TestDispatchLoop_BatchesPerTenantTable(t *testing.T) { dlq: emb, clients: chconn.NewHTTPClients(ingestHTTPClient), cache: mc, - target: func() chconn.Target { - return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db"} + // Each tenant's target names its own database. + target: func(id tenant.ID) chconn.Target { + return chconn.Target{URL: fmt.Sprintf("http://%s:%s", host, port), Username: "u", Password: "p", Database: "db_" + string(id)} }, maxBatch: maxBatch, maxWait: 30 * time.Second, // the size trigger is the one under test @@ -1851,9 +1853,11 @@ func TestDispatchLoop_BatchesPerTenantTable(t *testing.T) { "each batch bumps its own tenant's namespaces") mu.Lock() defer mu.Unlock() - for _, body := range bodies { + for _, id := range []string{"acme", "globex"} { + body, ok := bodies["db_"+id] + require.True(t, ok, "tenant %s's batch reached its own target", id) assert.Equal(t, maxBatch, strings.Count(body, "\n"), "a full batch: %q", body) - assert.NotEqual(t, strings.Contains(body, "acme"), strings.Contains(body, "globex"), "one tenant per INSERT: %q", body) + assert.Equal(t, maxBatch, strings.Count(body, id), "only tenant %s's rows: %q", id, body) } } @@ -1870,7 +1874,7 @@ func TestInsertToClickHouse_SetsConfiguredHeaders(t *testing.T) { }, } w, _, _, _ := newTestWorker(rt) - w.target = func() chconn.Target { + w.target = func(tenant.ID) chconn.Target { return chconn.Target{ URL: "http://test-clickhouse:8123", Username: "test_user", Password: "test_pass", Database: "test_db", Headers: map[string]string{"X-Proxy-Token": "abc", "Content-Type": "text/plain", "X-ClickHouse-User": "someone-else"}, @@ -1891,8 +1895,8 @@ func TestInsertToClickHouse_UsesTargetTLS(t *testing.T) { t.Cleanup(srv.Close) pool := x509.NewCertPool() pool.AddCert(srv.Certificate()) - target := func(cfg *tls.Config) func() chconn.Target { - return func() chconn.Target { + target := func(cfg *tls.Config) func(tenant.ID) chconn.Target { + return func(tenant.ID) chconn.Target { return chconn.Target{URL: srv.URL, Username: "u", Password: "p", Database: "db", TLS: cfg} } } @@ -1908,3 +1912,25 @@ func TestInsertToClickHouse_UsesTargetTLS(t *testing.T) { w.target = target(nil) require.ErrorContains(t, insert(), "unknown authority") } + +// TestInsertToClickHouse_NoTargetIsAnError: a tenant on no pool — its tuple +// could not be opened, such as by the connection ceiling — has no HTTP +// target, and its insert fails naming the tenant before any request is +// built, into the same failure path an unreachable ClickHouse takes. +func TestInsertToClickHouse_NoTargetIsAnError(t *testing.T) { + t.Parallel() + rt := &testutil.MockRoundTripper{Fn: func(*http.Request) (*http.Response, error) { + t.Fatal("no request must be made without a target") + return nil, nil + }} + w, _, _, _ := newTestWorker(rt) + var asked []tenant.ID + w.target = func(id tenant.ID) chconn.Target { + asked = append(asked, id) + return chconn.Target{} + } + err := w.insertToClickHouse(context.Background(), "events", []string{"id"}, []parsedMsg{{row: []byte(`[1]`), tenant: "acme"}}) + require.ErrorContains(t, err, "no ClickHouse connection is open for tenant acme") + assert.Equal(t, []tenant.ID{"acme"}, asked, "the target is the batch's own tenant's") + assert.Zero(t, rt.Hits()) +} diff --git a/internal/stream/hub.go b/internal/stream/hub.go index b299cf6d..1564feac 100644 --- a/internal/stream/hub.go +++ b/internal/stream/hub.go @@ -34,9 +34,9 @@ import ( type Hub struct { mu sync.RWMutex topics map[mq.Topic]*topicRoutes - policy PolicySource // nil ⇒ policy filtering not configured (legacy passthrough) - registry *discovery.SchemaRegistry // nil ⇒ no column types; row-filter comparison degrades fail-closed (see columnSpecs) - metric *Metrics // nil-safe + policy PolicySource // nil ⇒ policy filtering not configured (legacy passthrough) + registry RegistrySource // nil, or yielding nil ⇒ no column types; row-filter comparison degrades fail-closed (see columnSpecs) + metric *Metrics // nil-safe // RowEvaluator is the seam a native type layer will take over: the one // place a row's visibility under a role's row-filter is decided. nil means @@ -86,16 +86,35 @@ type topicRoutes struct { // is a deliberate lockout — a tenant the registry no longer serves included. type PolicySource func(tenant.ID) *policy.Policy +// RegistrySource yields a tenant's schema registry, its own since #583 +// story 6, read per event so a reload that rebuilds it applies to the next +// one. nil is a tenant with no registry, read as no schema. +type RegistrySource func(tenant.ID) *discovery.SchemaRegistry + // NewHub builds the event hub. A nil policy store passes every event through // unfiltered (the unwired-tests case); a non-nil store whose Get returns nil is a -// total lockout (a deleted/absent policy denies everyone). A nil registry leaves +// total lockout (a deleted/absent policy denies everyone). A nil registry source leaves // every column's type unknown, so row-filter comparison degrades FAIL-CLOSED: // equality/set predicates admit only a byte-identical value and ordering/!= admit // nothing (see policy.ColumnKind); metric may be nil. -func NewHub(policyStore PolicySource, registry *discovery.SchemaRegistry, metric *Metrics) *Hub { +func NewHub(policyStore PolicySource, registry RegistrySource, metric *Metrics) *Hub { return &Hub{topics: make(map[mq.Topic]*topicRoutes), policy: policyStore, registry: registry, metric: metric} } +// schema is tenant id's schema for table, nil when the hub has no registry +// source, the tenant no registry, or the registry no such table — every one +// of them the fail-closed reading. +func (h *Hub) schema(id tenant.ID, table string) *discovery.TableSchema { + if h.registry == nil { + return nil + } + reg := h.registry(id) + if reg == nil { + return nil + } + return reg.Get(table) +} + // Add registers sub to receive events for (topic, role), creating the role bucket // (and topic) on first use. func (h *Hub) Add(topic mq.Topic, role string, sub *Subscriber) { @@ -215,7 +234,7 @@ func (h *Hub) Broadcast(topic mq.Topic, raw []byte) { // evaluate visibility per subscriber. Predicates read the full event row (a // filter may key on a column the role can't SELECT), not the projected columns. if !specsResolved { - colSpecs = h.columnSpecs(ev.evt.TableName) + colSpecs = h.columnSpecs(topic.Tenant, ev.evt.TableName) specsResolved = true } for _, sub := range rb.bucket.Snapshot() { @@ -274,11 +293,8 @@ func (h *Hub) rowAdmitted(p *policy.Policy, role string, ev *eventView, claims m // no schema is available (unknown table, or a Hub built without a registry), which // reads as every column Opaque: the fail-closed floor, never a lexicographic // fallback that could admit rows the query path excludes ("9" > "100" as text). -func (h *Hub) columnSpecs(table string) map[string]policy.ColumnSpec { - if h.registry == nil { - return nil - } - schema := h.registry.Get(table) +func (h *Hub) columnSpecs(id tenant.ID, table string) map[string]policy.ColumnSpec { + schema := h.schema(id, table) if schema == nil { return nil } @@ -495,7 +511,7 @@ func (h *Hub) ReplayProjector(id tenant.ID, role string, sub *Subscriber) func(r // One topic ⇒ one table, so this resolves once per replay in practice; the // guard re-resolves if a stream ever mixes tables rather than going stale. if specsFor != ev.evt.TableName { - colSpecs = h.columnSpecs(ev.evt.TableName) + colSpecs = h.columnSpecs(id, ev.evt.TableName) specsFor = ev.evt.TableName } if !h.rowAdmitted(p, role, ev, sub.claims, colSpecs) { @@ -524,10 +540,7 @@ func (h *Hub) ReplayProjector(id tenant.ID, role string, sub *Subscriber) func(r // table, or a role that cannot read it. That is not an error — the event path's // drift check still sends a schema frame before the first data frame. func (h *Hub) SubscribeSchemaFrame(id tenant.ID, table, role string, sub *Subscriber) (Frame, bool) { - if h.registry == nil { - return Frame{}, false - } - schema := h.registry.Get(table) + schema := h.schema(id, table) if schema == nil { return Frame{}, false } diff --git a/internal/stream/hub_test.go b/internal/stream/hub_test.go index b6c83ae3..5bbede8e 100644 --- a/internal/stream/hub_test.go +++ b/internal/stream/hub_test.go @@ -629,7 +629,7 @@ func TestHub_RowFilter_NumericOrdering_SchemaInformed(t *testing.T) { "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"amount": {Gt: new("100")}}}}}, }, } - hub := NewHub(staticPolicy(p), reg, nil) + hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) topic := topicOf("clicks") sub := NewSubscriber(nil, nil) // constant filter value ⇒ no claims needed @@ -669,7 +669,7 @@ func TestHub_RowFilter_FloatNarrowing_SchemaInformed(t *testing.T) { "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"score": {Gt: new("16777216")}}}}}, }, } - hub := NewHub(staticPolicy(p), reg, nil) + hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) topic := topicOf("clicks") sub := NewSubscriber(nil, nil) hub.Add(topic, "viewer", sub) @@ -1003,7 +1003,7 @@ func TestHub_RowFilter_BigIntegerExact(t *testing.T) { "clicks": {"viewer": {Select: &policy.SelectPermissions{Filter: map[string]policy.Filter{"tenant_id": {Eq: new("{{ jwt.tenant }}")}}}}}, }, } - hub := NewHub(staticPolicy(p), reg, nil) + hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) topic := topicOf("clicks") // Claims come from real signed tokens through the production middleware, so a @@ -1048,7 +1048,7 @@ func TestHub_RowFilter_TimestampInstantMatch(t *testing.T) { }, }, } - hub := NewHub(staticPolicy(p), reg, nil) + hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) topic := topicOf("clicks") sub := NewSubscriber(nil, nil) @@ -1341,7 +1341,7 @@ func TestHub_SubscribeSchemaFrame(t *testing.T) { "clicks": {"viewer": {Select: &policy.SelectPermissions{AllowColumns: []string{"page"}}}}, }, } - hub := NewHub(staticPolicy(p), reg, nil) + hub := NewHub(staticPolicy(p), fixedRegistry(reg), nil) sub := NewSubscriber(nil, nil) f, ok := hub.SubscribeSchemaFrame(tenant.Default, "clicks", "viewer", sub) @@ -1378,8 +1378,8 @@ func TestHub_SubscribeSchemaFrame_NothingToAnnounce(t *testing.T) { role string }{ {"no registry", NewHub(staticPolicy(p), nil, nil), "clicks", "viewer"}, - {"unknown table", NewHub(staticPolicy(p), reg, nil), "missing", "viewer"}, - {"role cannot read the table", NewHub(staticPolicy(p), reg, nil), "clicks", "stranger"}, + {"unknown table", NewHub(staticPolicy(p), fixedRegistry(reg), nil), "missing", "viewer"}, + {"role cannot read the table", NewHub(staticPolicy(p), fixedRegistry(reg), nil), "clicks", "stranger"}, } for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { @@ -1472,7 +1472,7 @@ func TestHub_SubscribeSchemaFrame_ExcludesComputedColumns(t *testing.T) { {Name: "country", Type: "String"}, }}, }) - hub := NewHub(nil, reg, nil) + hub := NewHub(nil, fixedRegistry(reg), nil) sub := NewSubscriber(nil, nil) f, ok := hub.SubscribeSchemaFrame(tenant.Default, "clicks", "public", sub) @@ -1488,6 +1488,27 @@ func TestHub_SubscribeSchemaFrame_ExcludesComputedColumns(t *testing.T) { assert.Equal(t, "/a", row["page"]) } +// fixedRegistry is a RegistrySource fixed to reg, whatever the tenant. +func fixedRegistry(reg *discovery.SchemaRegistry) RegistrySource { + return func(tenant.ID) *discovery.SchemaRegistry { return reg } +} + +// TestHub_RegistrySourceYieldingNilIsNoSchema: a tenant with no registry — +// not served, or its registry not built yet — reads exactly like a hub with +// no registry at all: nothing to announce, every column opaque. +func TestHub_RegistrySourceYieldingNilIsNoSchema(t *testing.T) { + t.Parallel() + var asked []tenant.ID + hub := NewHub(nil, func(id tenant.ID) *discovery.SchemaRegistry { + asked = append(asked, id) + return nil + }, nil) + _, ok := hub.SubscribeSchemaFrame("acme", "clicks", "viewer", NewSubscriber(nil, nil)) + assert.False(t, ok) + assert.Nil(t, hub.columnSpecs("globex", "clicks")) + assert.Equal(t, []tenant.ID{"acme", "globex"}, asked, "the source is asked for the tenant the lookup names") +} + // TestHub_TopicsAreTenantScoped: two tenants, one table name (#583). A // broadcast on one tenant's topic reaches that tenant's subscribers alone, // projected under that tenant's policy — the policy source is asked for the @@ -1509,7 +1530,7 @@ func TestHub_TopicsAreTenantScoped(t *testing.T) { defer mu.Unlock() asked = append(asked, id) return policies[id] - }, reg, nil) + }, fixedRegistry(reg), nil) acmeTopic := mq.Topic{Tenant: "acme", Table: "clicks"} globexTopic := mq.Topic{Tenant: "globex", Table: "clicks"} diff --git a/internal/testutil/mocks.go b/internal/testutil/mocks.go index 9904d4a0..6a9dc64a 100644 --- a/internal/testutil/mocks.go +++ b/internal/testutil/mocks.go @@ -11,6 +11,7 @@ import ( "github.com/Wave-RF/WaveHouse/internal/cache" "github.com/Wave-RF/WaveHouse/internal/mq" + "github.com/Wave-RF/WaveHouse/internal/tenant" ) // Compile-time interface assertions. Catch breakage early when an interface @@ -143,8 +144,10 @@ func (m *MockDeduplicator) Close() error { return nil } type MockCache struct { cache.Cache // Embed to satisfy remaining interface methods silently InvNamespaces []cache.Namespace - InvErr error - mu sync.Mutex + // InvTenants records every InvalidateTenant, in order. + InvTenants []tenant.ID + InvErr error + mu sync.Mutex } func (m *MockCache) Invalidate(_ context.Context, namespaces []cache.Namespace) (uint64, error) { @@ -154,6 +157,20 @@ func (m *MockCache) Invalidate(_ context.Context, namespaces []cache.Namespace) return uint64(len(namespaces)), m.InvErr } +func (m *MockCache) InvalidateTenant(_ context.Context, id tenant.ID) error { + m.mu.Lock() + defer m.mu.Unlock() + m.InvTenants = append(m.InvTenants, id) + return m.InvErr +} + +// GetTenants returns the tenants InvalidateTenant was called for, in order. +func (m *MockCache) GetTenants() []tenant.ID { + m.mu.Lock() + defer m.mu.Unlock() + return append([]tenant.ID(nil), m.InvTenants...) +} + func (m *MockCache) GetNamespaces() []cache.Namespace { m.mu.Lock() defer m.mu.Unlock() diff --git a/internal/testutil/testutil.go b/internal/testutil/testutil.go index 80e141de..371cdcd4 100644 --- a/internal/testutil/testutil.go +++ b/internal/testutil/testutil.go @@ -28,7 +28,8 @@ import ( // type string), not the caller's structs. func NewTestSchemaRegistry(t testing.TB, tables []*discovery.TableSchema) *discovery.SchemaRegistry { t.Helper() - reg := discovery.NewSchemaRegistry(&schemaConn{tables: tables}, func() string { return "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) + conn := &schemaConn{tables: tables} + reg := discovery.NewSchemaRegistry(func() (driver.Conn, string) { return conn, "test" }, tenant.Default, func(tenant.ID) time.Duration { return time.Hour }) require.NoError(t, reg.Refresh(context.Background())) return reg } diff --git a/tests/e2e/sdk/admin.test.ts b/tests/e2e/sdk/admin.test.ts index 54589b39..4ffe351f 100644 --- a/tests/e2e/sdk/admin.test.ts +++ b/tests/e2e/sdk/admin.test.ts @@ -58,6 +58,32 @@ describe("Admin", () => { expect(result.error).toBeNull(); }); + // The schema routes and the raw-SQL proxy name their tenant in ?tenant= + // like the pipe reads. This stack's settings directory is the single + // tenant 0. + it("addresses a tenant with the tenant option", async () => { + const named = await wh.schema.list({ tenant: "0" }); + expect(named.error).toBeNull(); + expect(named.data).toHaveProperty(T.clicks); + + const one = await wh.from(T.clicks).schema({ tenant: "0" }); + expect(one.error).toBeNull(); + expect(one.data?.name).toBe(T.clicks); + + const refreshed = await wh.schema.refresh({ tenant: "0" }); + expect(refreshed.error).toBeNull(); + + const rows = await wh.sql("SELECT 1 AS one", { tenant: "0" }); + expect(rows.error).toBeNull(); + + const unknown = await wh.schema.list({ tenant: "acme" }); + expect(unknown.error?.status).toBe(404); + expect(unknown.error?.message).toContain("unknown tenant: acme"); + + const malformed = await wh.sql("SELECT 1", { tenant: "a.b" }); + expect(malformed.error?.status).toBe(400); + }); + it("gets per-table schema", async () => { const result = await wh.from(T.clicks).schema(); expect(result.error).toBeNull(); diff --git a/tests/integration/boot_resilience_test.go b/tests/integration/boot_resilience_test.go index e511953f..5c2f63c1 100644 --- a/tests/integration/boot_resilience_test.go +++ b/tests/integration/boot_resilience_test.go @@ -66,14 +66,14 @@ func TestBootResilience_StickyHealthVsConditionalReady(t *testing.T) { require.NoError(t, err, "reopen driver against stopped CH") bootState := api.NewBootState(nil) - registry := discovery.NewSchemaRegistry(ch.conn, func() string { return testCHDatabase }, tenant.Default, func(tenant.ID) time.Duration { return time.Minute }) + registry := discovery.NewSchemaRegistry(func() (driver.Conn, string) { return ch.conn, testCHDatabase }, tenant.Default, func(tenant.ID) time.Duration { return time.Minute }) // === Row 1: Boot, CH down === err = registry.Refresh(ctx) require.Error(t, err, "Refresh against stopped CH must fail") bootState.Set(fmt.Errorf("schema discovery: %w", err)) - h := api.NewHealthHandler(ch.conn) + h := api.NewHealthHandler(ch.conn.Ping) h.Boot = bootState assertHealth(t, "/livez", h.Liveness, http.StatusServiceUnavailable, `"status":"degraded"`) @@ -88,8 +88,8 @@ func TestBootResilience_StickyHealthVsConditionalReady(t *testing.T) { _ = ch.conn.Close() ch.conn, err = openDriver(ch.nativeAddr()) require.NoError(t, err, "reopen driver against restarted CH") - registry = discovery.NewSchemaRegistry(ch.conn, func() string { return testCHDatabase }, tenant.Default, func(tenant.ID) time.Duration { return time.Minute }) - h.CHConn = ch.conn + registry = discovery.NewSchemaRegistry(func() (driver.Conn, string) { return ch.conn, testCHDatabase }, tenant.Default, func(tenant.ID) time.Duration { return time.Minute }) + h.Ping = ch.conn.Ping require.NoError(t, waitForNativeReady(ctx, ch.conn, 30*time.Second), "CH native should be ready after restart") retryCtx, retryCancel := context.WithTimeout(ctx, 30*time.Second) @@ -123,7 +123,7 @@ func TestBootResilience_StickyHealthVsConditionalReady(t *testing.T) { _ = ch.conn.Close() ch.conn, err = openDriver(ch.nativeAddr()) require.NoError(t, err, "reopen driver against final restart") - h.CHConn = ch.conn + h.Ping = ch.conn.Ping require.NoError(t, waitForNativeReady(ctx, ch.conn, 30*time.Second), "CH native should be ready after second restart") // === Row 4: Post-boot, CH back === diff --git a/tests/integration/query_limits_test.go b/tests/integration/query_limits_test.go index caa2a0ed..cd82af8d 100644 --- a/tests/integration/query_limits_test.go +++ b/tests/integration/query_limits_test.go @@ -11,11 +11,13 @@ import ( "testing" "time" + "github.com/ClickHouse/clickhouse-go/v2/lib/driver" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" "github.com/Wave-RF/WaveHouse/internal/api" "github.com/Wave-RF/WaveHouse/internal/auth" + "github.com/Wave-RF/WaveHouse/internal/discovery" "github.com/Wave-RF/WaveHouse/internal/policy" "github.com/Wave-RF/WaveHouse/internal/settings" ) @@ -99,7 +101,7 @@ func TestStructuredQuery_ResourceCapsEnforcedServerSide(t *testing.T) { }, } h := api.NewStructuredQueryHandler( - e.chConn, nil, e.registry, func(*settings.Store) *policy.Policy { return p }, func(*settings.Store) int { return 60 }, func() time.Duration { return 30 * time.Second }, nil, + func(*settings.Store) driver.Conn { return e.chConn }, nil, func(*settings.Store) *discovery.SchemaRegistry { return e.registry }, func(*settings.Store) *policy.Policy { return p }, func(*settings.Store) int { return 60 }, func(*settings.Store) time.Duration { return 30 * time.Second }, nil, ) req := httptest.NewRequest(http.MethodPost, diff --git a/tests/integration/setup_test.go b/tests/integration/setup_test.go index 470aa344..a01064a2 100644 --- a/tests/integration/setup_test.go +++ b/tests/integration/setup_test.go @@ -45,6 +45,7 @@ const ( // testEnv holds the shared infrastructure available to every test. type testEnv struct { + ch *chInstance chConn driver.Conn chHTTPURL string embeddedMQ mq.Broker @@ -192,6 +193,7 @@ func setup() (int, func()) { } sharedEnv = &testEnv{ + ch: ch, chConn: ch.conn, chHTTPURL: ch.httpURL(), embeddedMQ: a.MQ(), @@ -209,18 +211,32 @@ func setup() (int, func()) { // unit tests and the e2e SDK suite. The stream budget is shrunk to 1 GiB // like the e2e fixture so the scratch directory stays small. func writeTestSettings(ch *chInstance) (string, error) { - files, err := settings.Seed() + files, err := tenantSettings(ch, testCHDatabase) if err != nil { return "", err } + dir := mustTempDir() + if err := writeSettingsFiles(dir, files); err != nil { + return "", err + } + return dir, nil +} + +// tenantSettings is one tenant's four files: the seed with the ClickHouse +// block pointed at the testcontainer's database, and the dev-style policy. +func tenantSettings(ch *chInstance, database string) (map[string][]byte, error) { + files, err := settings.Seed() + if err != nil { + return nil, err + } var doc map[string]json.RawMessage if err := json.Unmarshal(files[settings.FileConfig], &doc); err != nil { - return "", fmt.Errorf("seed config.json: %w", err) + return nil, fmt.Errorf("seed config.json: %w", err) } patch := map[string]any{ "clickhouse": map[string]any{ "addr": ch.nativeAddr(), "http_port": mustAtoi(ch.httpPort), "http_scheme": "http", - "database": testCHDatabase, "username": testCHUser, "query_timeout": 30, + "database": database, "username": testCHUser, "query_timeout": 30, "tls": map[string]any{"enabled": false, "ca_file": "", "cert_file": "", "key_file": "", "insecure_skip_verify": false, "server_name": ""}, "headers": map[string]any{}, "max_open_conns": 10, "max_idle_conns": 5, }, @@ -228,22 +244,28 @@ func writeTestSettings(ch *chInstance) (string, error) { } for key, val := range patch { if doc[key], err = json.Marshal(val); err != nil { - return "", err + return nil, err } } if files[settings.FileConfig], err = json.MarshalIndent(doc, "", " "); err != nil { - return "", err + return nil, err } files[settings.FileRoles] = []byte(`{"roles": ["admin"]}`) files[settings.FilePolicies] = []byte(`{"default_role": "admin"}`) + return files, nil +} - dir := mustTempDir() +// writeSettingsFiles writes one tenant's files into dir. +func writeSettingsFiles(dir string, files map[string][]byte) error { + if err := os.MkdirAll(dir, 0o750); err != nil { + return err + } for name, data := range files { if err := os.WriteFile(filepath.Join(dir, name), data, 0o600); err != nil { - return "", err + return err } } - return dir, nil + return nil } // waitForLive polls /livez until it returns 200 or the timeout elapses, diff --git a/tests/integration/tenants_test.go b/tests/integration/tenants_test.go new file mode 100644 index 00000000..40c5a700 --- /dev/null +++ b/tests/integration/tenants_test.go @@ -0,0 +1,138 @@ +//go:build integration + +package tests + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net" + "net/http" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/Wave-RF/WaveHouse/internal/app" + "github.com/Wave-RF/WaveHouse/internal/config" + "github.com/Wave-RF/WaveHouse/internal/tenant" +) + +// TestNestedDirectory_PerTenantPoolsAndDiscovery boots the real wiring over +// a nested settings directory whose two tenants point at the one ClickHouse +// through two databases — two tuples, so two pools and two schema +// registries (#583 story 6) — and checks each tenant discovers its own +// tables and none of the other's, that the ops routes address a tenant by +// ?tenant=, that /livez turns 200 once a tenant has discovered and /readyz +// finds a pool that answers, and that a tenant's structured query runs +// against its own database. +func TestNestedDirectory_PerTenantPoolsAndDiscovery(t *testing.T) { + e := env(t) + ctx := context.Background() + const operatorKey = "it-operator-key" + + // One database per tenant, each with a table of its own shape. + databases := map[tenant.ID]string{"acme": "it_acme_" + t.Name(), "globex": "it_globex_" + t.Name()} + columns := map[tenant.ID]string{"acme": "id String, page String", "globex": "id String, amount Float64"} + for id, db := range databases { + db = strings.ToLower(strings.NewReplacer("/", "_", " ", "_", "-", "_").Replace(db)) + databases[id] = db + require.NoError(t, e.chConn.Exec(ctx, "CREATE DATABASE IF NOT EXISTS "+db)) + t.Cleanup(func() { _ = e.chConn.Exec(context.Background(), "DROP DATABASE IF EXISTS "+db) }) + require.NoError(t, e.chConn.Exec(ctx, fmt.Sprintf("CREATE TABLE %s.events (%s) ENGINE = MergeTree() ORDER BY id", db, columns[id]))) + require.NoError(t, e.chConn.Exec(ctx, fmt.Sprintf("INSERT INTO %s.events (id) VALUES ('1')", db))) + } + + root := t.TempDir() + for id, db := range databases { + files, err := tenantSettings(e.ch, db) + require.NoError(t, err) + require.NoError(t, writeSettingsFiles(filepath.Join(root, id.String()), files)) + } + + var lc net.ListenConfig + ln, err := lc.Listen(ctx, "tcp", "127.0.0.1:0") + require.NoError(t, err) + cfg := &config.Config{ + DataDir: t.TempDir(), + Server: config.Server{ShutdownTimeout: 10}, + ClickHouse: config.ClickHouse{Password: testCHPassword}, + Auth: config.Auth{OperatorKey: operatorKey}, + Cache: config.Cache{L1MaxCost: 1 << 20}, + Settings: config.Settings{Dir: root}, + } + a, err := app.New(ctx, app.Options{Config: cfg, Listener: ln}) + require.NoError(t, err) + runCtx, stop := context.WithCancel(ctx) + runDone := make(chan error, 1) + go func() { runDone <- a.Run(runCtx) }() + t.Cleanup(func() { + stop() + assert.NoError(t, <-runDone) + closeCtx, cancel := context.WithTimeout(context.Background(), 10*time.Second) + defer cancel() + assert.NoError(t, a.Close(closeCtx)) + }) + baseURL := "http://" + ln.Addr().String() + // A nested directory's tenants discover in the background: /livez turns + // 200 once the first has. + require.NoError(t, waitForLive(ctx, baseURL, 30*time.Second)) + + do := func(method, path string, headers map[string]string, body string) (int, string) { + t.Helper() + req, err := http.NewRequestWithContext(ctx, method, baseURL+path, strings.NewReader(body)) + require.NoError(t, err) + for k, v := range headers { + req.Header.Set(k, v) + } + resp, err := http.DefaultClient.Do(req) + require.NoError(t, err) + defer func() { _ = resp.Body.Close() }() + b, err := io.ReadAll(resp.Body) + require.NoError(t, err) + return resp.StatusCode, string(b) + } + operator := map[string]string{"X-Operator-Key": operatorKey} + + // Each tenant's registry discovers its own database and no other's; + // every tenant's has loaded once its refresh route answers 200. + for id, cols := range map[tenant.ID][]string{"acme": {"id", "page"}, "globex": {"id", "amount"}} { + require.Eventually(t, func() bool { + status, _ := do(http.MethodGet, "/v1/ops/schema?tenant="+id.String(), operator, "") + return status == http.StatusOK + }, 30*time.Second, 200*time.Millisecond, "tenant %s never discovered its schema", id) + status, body := do(http.MethodGet, "/v1/ops/schema?table=events&tenant="+id.String(), operator, "") + require.Equal(t, http.StatusOK, status, body) + var schema struct { + Columns []struct{ Name string } `json:"columns"` + } + require.NoError(t, json.Unmarshal([]byte(body), &schema)) + var names []string + for _, c := range schema.Columns { + names = append(names, c.Name) + } + assert.Equal(t, cols, names, "tenant %s reads its own database's table", id) + } + + status, body := do(http.MethodGet, "/readyz", nil, "") + assert.Equal(t, http.StatusOK, status, body) + + // A structured query runs against the tenant's own database: globex's + // events has no page column, so the same query is a 400 there. + query := `{"columns": ["page"]}` + status, body = do(http.MethodPost, "/v1/query?table=events", map[string]string{tenant.Header: "acme", "Content-Type": "application/json"}, query) + assert.Equal(t, http.StatusOK, status, body) + assert.Contains(t, body, `"page"`) + status, body = do(http.MethodPost, "/v1/query?table=events", map[string]string{tenant.Header: "globex", "Content-Type": "application/json"}, query) + assert.Equal(t, http.StatusBadRequest, status, body) + + // The raw-SQL proxy runs against the ?tenant='s database too. + status, body = do(http.MethodPost, "/v1/ops/query?tenant=globex", operator, `{"sql": "SELECT amount FROM events"}`) + assert.Equal(t, http.StatusOK, status, body) + status, body = do(http.MethodPost, "/v1/ops/query?tenant=acme", operator, `{"sql": "SELECT amount FROM events"}`) + assert.Equal(t, http.StatusBadRequest, status, body) +}